Compare commits

...
39 Commits
Author SHA1 Message Date
Dominique Eifländer 8924e905ad hotfix: Extend Tesseract instead of Tesseract1 2024-01-15 16:13:43 +01:00
Kilian Schüttler fb1fe35bc1 Merge branch 'RED-8155' into 'master'
RED-8155: bold-detection in ocr-service

Closes RED-8155

See merge request redactmanager/ocr-service!33
2024-01-08 13:53:30 +01:00
Kilian Schuettler 912f00aa84 RED-8155: bold-detection in ocr-service
* fix application.yml
2024-01-08 13:49:58 +01:00
Kilian Schüttler bab16ad9b2 Merge branch 'RED-7669-fontstyle' into 'master'
RED-8155: integrate bold-detection into ocr-service

Closes RED-7669

See merge request redactmanager/ocr-service!31
2024-01-05 16:05:53 +01:00
Kilian Schüttler be4656189b RED-8155: integrate bold-detection into ocr-service 2024-01-05 16:05:53 +01:00
Dominique Eifländer 8944b57344 Merge branch 'RED-7669' into 'master'
RED-7669: optimize OCR-module performance

Closes RED-7669

See merge request redactmanager/ocr-service!30
2023-12-22 15:14:42 +01:00
Kilian Schuettler 67540950b8 RED-7669: optimize OCR-module performance
* fix thread handling for PDFs without any images
2023-12-22 15:11:29 +01:00
Kilian Schuettler 6f29270e66 RED-7669: optimize OCR-module performance
* fix thread handling for PDFs without any images
2023-12-22 15:04:52 +01:00
Dominique Eifländer b961c9e324 Merge branch 'RED-1137' into 'master'
RED-1137: Do not observe actuator endpoints

Closes RED-1137

See merge request redactmanager/ocr-service!29
2023-12-20 14:24:49 +01:00
Dominique Eifländer 4b6411161e RED-1137: Do not observe actuator endpoints 2023-12-20 14:17:09 +01:00
Dominique Eifländer 14982eae7c Merge branch 'RED-5223' into 'master'
RED-5223: Use tracing-commons from fforesight

Closes RED-5223

See merge request redactmanager/ocr-service!28
2023-12-13 16:16:29 +01:00
Dominique Eifländer 99fc16130b RED-5223: Use tracing-commons from fforesight 2023-12-13 16:10:10 +01:00
Dominique Eifländer 80d38fb785 Merge branch 'RED-7669' into 'master'
RED-7669: optimize OCR-module performance

Closes RED-7669

See merge request redactmanager/ocr-service!27
2023-12-12 15:27:00 +01:00
Kilian Schüttler c06974ce69 RED-7669: optimize OCR-module performance 2023-12-12 15:27:00 +01:00
Dominique Eifländer 591c7d7fab Merge branch 'RED-5223' into 'master'
RED-5223: Enabled tracing, upgrade spring, use logstash-logback-encoder for json logs

Closes RED-5223

See merge request redactmanager/ocr-service!26
2023-12-12 12:36:01 +01:00
Dominique Eifländer 0300a087d4 RED-5223: Enabled tracing, upgrade spring, use logstash-logback-encoder for json logs 2023-12-12 11:55:01 +01:00
Andrei Isvoran 98752ff1d1 Merge branch 'RED-7714' into 'master'
RED-7715 - Add log4j config to enable switching between json/line logs

Closes RED-7714

See merge request redactmanager/ocr-service!25
2023-12-06 12:42:33 +01:00
Andrei Isvoran ae09a59a7c RED-7715 - Add log4j config to enable switching between json/line logs 2023-12-06 11:52:01 +02:00
Kilian Schüttler 65d818200f Merge branch 'RED-7669' into 'master'
RED-7669: optimize OCR-module performance

Closes RED-7669

See merge request redactmanager/ocr-service!24
2023-11-28 12:35:22 +01:00
Kilian Schuettler 6fe95c6940 RED-7669: optimize OCR-module performance
* dont interrupt threads, use boolean flag instead
2023-11-28 10:04:56 +01:00
Kilian Schüttler 202132e14c Merge branch 'RED-7668' into 'master'
RED-7669: optimize OCR-module performance

Closes RED-7668

See merge request redactmanager/ocr-service!23
2023-11-24 10:57:45 +01:00
Kilian Schuettler 0264e28cc2 RED-7669: optimize OCR-module performance
* enable caches
2023-11-24 10:21:55 +01:00
Dominique Eifländer a50f54676e Merge branch 'RED-7668' into 'master'
RED-7669: optimize OCR-module performance

Closes RED-7668

See merge request redactmanager/ocr-service!22
2023-11-23 16:04:43 +01:00
Kilian Schuettler 1926707ae1 RED-7669: optimize OCR-module performance
* move all critical stuff to its own singleton thread
* make gs process queue any image once the file has been written
2023-11-23 16:00:53 +01:00
Kilian Schuettler d3190844a3 RED-7669: optimize OCR-module performance
* move all critical stuff to its own singleton thread
* make gs process queue any image once the file has been written
2023-11-23 16:00:31 +01:00
Kilian Schuettler c7ccbae6ff RED-7669: optimize OCR-module performance
* move all critical stuff to its own singleton thread
* make gs process queue any image once the file has been written
2023-11-23 16:00:31 +01:00
Kilian Schuettler 880bebcafc RED-7669: optimize OCR-module performance
* move all critical stuff to its own singleton thread
* make gs process queue any image once the file has been written
2023-11-23 16:00:31 +01:00
Kilian Schuettler 955ff6281d RED-7669: optimize OCR-module performance
* move all critical stuff to its own singleton thread
* make gs process queue any image once the file has been written
2023-11-23 16:00:31 +01:00
Kilian Schuettler efd3a1d952 RED-7669: optimize OCR-module performance
* move all non thread safe stuff to separate thread in the middle
2023-11-23 16:00:29 +01:00
Kilian Schuettler bb5b4a2fd8 RED-7669: optimize OCR-module performance
* binarize images after reading
2023-11-23 16:00:22 +01:00
Kilian Schuettler 6f99664906 RED-7669: optimize OCR-module performance
* try and synchronize all malloc calls
2023-11-23 16:00:19 +01:00
Kilian Schuettler 574f7ac25e RED-7669: optimize OCR-module performance
* moar sigsegv
2023-11-23 16:00:01 +01:00
Kilian Schuettler 12217f2459 RED-7669: optimize OCR-module performance
* moar sigsegv
2023-11-23 16:00:01 +01:00
Kilian Schuettler 19747cbca5 RED-7669: optimize OCR-module performance
* moar sigsegv
2023-11-23 15:59:59 +01:00
Kilian Schuettler 2632d2023d RED-7669: optimize OCR-module performance
* reset test and settings
2023-11-23 15:59:16 +01:00
Kilian Schuettler 4c225c2219 RED-7669: optimize OCR-module performance
* cleanup Code
2023-11-23 15:59:16 +01:00
Kilian Schuettler 3d09f46844 RED-7669: optimize OCR-module performance
* don't despeckle small images
2023-11-23 15:59:16 +01:00
Kilian Schuettler 77355b5367 RED-7669: optimize OCR-module performance
* second attempt at thread safety
2023-11-23 15:59:16 +01:00
Kilian Schuettler 57e194fcd0 RED-7669: optimize OCR-module performance
* attempt at thread safety
2023-11-23 15:59:14 +01:00
47 changed files with 1503 additions and 505 deletions
+1 -1
View File
@@ -10,7 +10,7 @@ deploy:
script:
- echo "Building with gradle version ${BUILDVERSION}"
- gradle -Pversion=${BUILDVERSION} publish
- gradle bootBuildImage --cleanCache --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
- gradle bootBuildImage --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
- echo "BUILDVERSION=$BUILDVERSION" >> version.env
artifacts:
reports:
+9 -3
View File
@@ -15,11 +15,17 @@ The service uses PDFTron to attempt the removal of invisible elements and waterm
Extracts all images from the PDF using PDFBox
3. Striped Image Detection and Stitching
Detects if images are striped and stitches them together using Ghostscript.
4. Binarization
Binarizes the resulting images using Leptonica and the Otsu thresholding algorithm.
4. Image Processing
- Convert to grayscale
- Upscale to target DPI
- Filter using Gauss kernel
- Binarizes the resulting images using Leptonica and the Otsu thresholding algorithm.
- Despeckle using various morphological operations
5. OCR Processing
Runs Tesseract on the images to extract text.
6. Text Integration
6. Font style detection
Detection of bold text using stroke width estimation
7. Text Integration
Draws the resulting text onto the original PDF using PDFBox.
Steps 2.-5. happen in parallel and communicate via a blocking queue to limit RAM usage.
@@ -25,6 +25,8 @@ tasks.named<Test>("test") {
reports {
junitXml.outputLocation.set(layout.buildDirectory.dir("reports/junit"))
}
minHeapSize = "512m"
maxHeapSize = "8192m"
}
tasks.test {
@@ -14,13 +14,14 @@ dependencies {
api("net.sourceforge.tess4j:tess4j:5.8.0")
api("com.iqser.red.commons:metric-commons:2.1.0")
api("com.iqser.red.commons:storage-commons:2.45.0")
api("com.knecon.fforesight:tenant-commons:0.13.0")
api("com.knecon.fforesight:tenant-commons:0.19.0")
api("com.pdftron:PDFNet:10.5.0")
api("org.apache.pdfbox:pdfbox:3.0.0")
api("org.apache.pdfbox:jbig2-imageio:3.0.4")
api("com.github.jai-imageio:jai-imageio-core:1.4.0")
api("com.github.jai-imageio:jai-imageio-jpeg2000:1.4.0")
api("io.github.karols:hocr4j:0.1.2")
api("org.apache.commons:commons-math3:3.6.1")
api("io.github.karols:hocr4j:0.2.0")
api("com.amazonaws:aws-java-sdk-kms:1.12.440")
api("com.google.guava:guava:31.1-jre")
api("com.iqser.red.commons:pdftron-logic-commons:2.20.0")
@@ -0,0 +1,36 @@
package com.knecon.fforesight.service.ocr.processor.model;
import java.awt.geom.Rectangle2D;
import java.awt.image.BufferedImage;
import org.apache.pdfbox.pdmodel.graphics.color.PDColorSpace;
import org.apache.pdfbox.util.Matrix;
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
import lombok.AccessLevel;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
public record ExtractedImage(
int pageNumber, QuadPoint position, int height, int width, BufferedImage image, Matrix ctm, int numberOnPage, PDColorSpace colorSpace) implements UnprocessedImage {
@SneakyThrows
public Pix asPix() {
BufferedImage image = ImageProcessingUtils.convertToDeviceColorSpace(this);
ImageProcessingUtils.setAlphaChannelToWhite(image);
return LeptUtils.convertImageToPix(image);
}
public QuadPoint getImageCoordinatesInInitialUserSpace() {
return QuadPoint.fromRectangle2D(new Rectangle2D.Double(0, 0, 1, 1)).getTransformed(ctm.createAffineTransform());
}
}
@@ -1,20 +1,15 @@
package com.knecon.fforesight.service.ocr.processor.model;
import java.awt.AlphaComposite;
import java.awt.Color;
import java.awt.Graphics2D;
import java.awt.Transparency;
import java.awt.Graphics;
import java.awt.geom.AffineTransform;
import java.awt.image.BufferedImage;
import java.io.IOException;
import java.nio.IntBuffer;
import java.util.concurrent.Semaphore;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.util.Matrix;
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
import com.pdftron.sdf.Obj;
import lombok.AccessLevel;
import lombok.Getter;
@@ -23,58 +18,26 @@ import lombok.Setter;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
import net.sourceforge.tess4j.ITessAPI;
@Slf4j
@Getter
@RequiredArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class ExtractedOcrImage implements OcrImage {
final int pageNumber;
final Pix pix;
final int originalHeight;
final int originalWidth;
final int height;
final int width;
final Matrix ctm;
final int numberOnPage;
@Setter
int pageNumber;
int numberOnPage;
int originalHeight;
int originalWidth;
Matrix ctm;
Pix pix;
int height;
int width;
int rotationDegrees;
@SneakyThrows
public ExtractedOcrImage(int pageNumber, int numberOnPage, BufferedImage bufferedImage, Matrix ctm, int targetDpi) {
this.pageNumber = pageNumber;
this.numberOnPage = numberOnPage;
this.ctm = ctm;
this.originalHeight = bufferedImage.getHeight();
this.originalWidth = bufferedImage.getWidth();
float imageDPI = Math.abs(bufferedImage.getWidth() / (ctm.getScalingFactorX() / 72));
this.pix = binarize(bufferedImage, imageDPI, targetDpi);
this.height = pix.h;
this.width = pix.w;
}
@SneakyThrows
private Pix binarize(BufferedImage image, float imageDpi, int targetDpi) {
ImageProcessingUtils.setAlphaChannelToWhite(image);
synchronized (OCRThread.class) { // must synchronize the mallocs here with the mallocs tesseract detection script.
Pix grayScale = ImageProcessingUtils.convertToGrayScale(image);
Pix scaledUp = ImageProcessingUtils.scaleToTargetDpi(imageDpi, targetDpi, grayScale);
return ImageProcessingUtils.despecklePix(scaledUp);
}
}
@Override
public AffineTransform getImageCTM() {
@@ -95,11 +58,4 @@ public class ExtractedOcrImage implements OcrImage {
return affineTransform;
}
@Override
public int getOptimalPageSegmentationMode() {
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
}
}
@@ -2,13 +2,16 @@ package com.knecon.fforesight.service.ocr.processor.model;
import java.awt.geom.AffineTransform;
import java.awt.geom.Point2D;
import java.awt.image.BufferedImage;
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
import com.knecon.fforesight.service.ocr.processor.utils.PdfDpiCalculator;
import lombok.SneakyThrows;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
import net.sourceforge.tess4j.ITessAPI;
public interface OcrImage {
@@ -28,9 +31,19 @@ public interface OcrImage {
int getNumberOnPage();
/**
* Retrieves the height of the original image (not necessarily in pdf coordinates).
*
* @return the height of the image
*/
int getHeight();
/**
* Retrieves the width of the original image (not necessarily in pdf coordinates).
*
* @return the width of the image
*/
int getWidth();
@@ -41,7 +54,7 @@ public interface OcrImage {
*/
default QuadPoint getImageBounds() {
// cannot be solved with a nice rotation matrix, since the after rotating the text coordinates in the image will always start at (0,0) and will therefore always start at (0,0) in the PDF.
// cannot be solved with a nice rotation matrix. After rotating the text coordinates in the image will always start at (0,0) and will therefore always start at (0,0) in the PDF.
// So in order to mimic this behavior we need to start with (0,0) coordinates always.
if (getRotationDegrees() == 90 || getRotationDegrees() == 270) {
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, getWidth()), new Point2D.Double(getHeight(), getWidth()), new Point2D.Double(getHeight(), 0));
@@ -75,17 +88,13 @@ public interface OcrImage {
*
* @return The optimal page segmentation mode.
*/
int getOptimalPageSegmentationMode(); // TODO: evaluate if PSM can be dynamically chosen to increase performance
default int getOptimalPageSegmentationMode() {
/**
* Sets the rotation degree of the OCR image. The rotation degree specifies the amount of rotation applied to the image.
* Currently only quadrant rotations are supported.
* Rotated partial images work, due to the CTM present in the pdf working with any rotation.
*
* @param rotationDegree The rotation degree of the OCR image.
*/
void setRotationDegrees(int rotationDegree);
if (getWidth() < 200 || getHeight() < 200) {
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
}
return ITessAPI.TessPageSegMode.PSM_AUTO;
} // TODO: evaluate if PSM can be dynamically chosen to increase performance
/**
@@ -96,24 +105,6 @@ public interface OcrImage {
Pix getPix();
/**
* Retrieves the rotated image of the OCR image.
*
* @return The rotated BufferedImage object of the OCR image.
*/
default Pix getRotatedPix() {
synchronized (OCRThread.class) {
return switch (360 - getRotationDegrees()) {
case 90 -> Leptonica1.pixRotateOrth(getPix(), 1);
case 180 -> Leptonica1.pixRotateOrth(getPix(), 2);
case 270 -> Leptonica1.pixRotateOrth(getPix(), 3);
default -> getPix();
};
}
}
default int getDpi() {
return PdfDpiCalculator.calculateDpi(getImageBounds(), getImageCTM(), getWidth());
@@ -128,17 +119,6 @@ public interface OcrImage {
AffineTransform getImageCTM();
/**
* Retrieves the size (width * height) of the image.
*
* @return The size of the image.
*/
default int getImageSize() {
return getHeight() * getWidth();
}
default void destroyPix() {
LeptUtils.disposePix(getPix());
@@ -7,27 +7,17 @@ import com.knecon.fforesight.service.ocr.processor.service.HOcrPageParser;
import io.github.karols.hocr4j.Word;
public record OcrResult(Image image, String hOcrPageAbsolutePath) {
public record OcrResult(OcrImage image, String tesseractOutputFilePath) {
public static OcrResult create(OcrImage image, String tesseractResult) {
return new OcrResult(Image.fromOcrImage(image), tesseractResult);
return new OcrResult(image, tesseractResult);
}
public List<Word> getAllWords() {
return HOcrPageParser.extractHocrPage(hOcrPageAbsolutePath).getAllWords();
}
public record Image(Integer pageNumber, AffineTransform ctm, QuadPoint position) {
public static Image fromOcrImage(OcrImage image) {
return new Image(image.getPageNumber(), image.getImageCTM(), image.getImageCoordinatesInInitialUserSpace());
}
return HOcrPageParser.extractHocrPage(tesseractOutputFilePath).getAllWords();
}
}
@@ -0,0 +1,35 @@
package com.knecon.fforesight.service.ocr.processor.model;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.FontStyleDetectionModel;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
public record OcrResultToWrite(List<TextPositionInImage> textPositionInImage, QuadPoint imageBoundingBox) {
public static OcrResultToWrite fromFontStyleDetectionModel(FontStyleDetectionModel fontStyleDetectionModel) {
return new OcrResultToWrite(fontStyleDetectionModel.getTextPositionInImages(), fontStyleDetectionModel.getImageBounds());
}
public static Map<Integer, List<OcrResultToWrite>> buildOcrResultsToWrite(List<OcrResult> ocrResults, FontMetricsFactory fontMetricsFactory) {
return ocrResults.stream()
.collect(Collectors.groupingBy(ocrResult -> ocrResult.image().getPageNumber()))
.entrySet()
.stream()
.collect(Collectors.toMap(Map.Entry::getKey,
entry -> entry.getValue()
.stream()
.map(ocrResult -> new OcrResultToWrite(ocrResult.getAllWords()
.stream()
.filter(word -> !word.isBlank())
.map(word -> new TextPositionInImage(word, ocrResult.image().getImageCTM(), fontMetricsFactory, FontStyle.REGULAR))
.toList(), ocrResult.image().getImageCoordinatesInInitialUserSpace()))
.toList()));
}
}
@@ -0,0 +1,12 @@
package com.knecon.fforesight.service.ocr.processor.model;
import org.apache.pdfbox.pdmodel.PDPage;
public record PageInformation(int height, int width, int number, int rotationDegrees) {
public static PageInformation fromPDPage(int pageNum, PDPage page) {
return new PageInformation((int) page.getMediaBox().getHeight(), (int) page.getMediaBox().getWidth(), pageNum, page.getRotation());
}
}
@@ -97,4 +97,10 @@ public record QuadPoint(Point2D a, Point2D b, Point2D c, Point2D d) {
d().getY());
}
public double size() {
return a().distance(b()) * a().distance(d());
}
}
@@ -1,5 +1,14 @@
package com.knecon.fforesight.service.ocr.processor.model;
public record RenderedPageImageFile(int pageNumber, String absoluteFilePath) {
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
public record RenderedPageImageFile(int pageNumber, String absoluteFilePath) implements UnprocessedImage {
@Override
public Pix asPix() {
return Leptonica1.pixRead(absoluteFilePath);
}
}
@@ -8,6 +8,7 @@ import org.apache.pdfbox.pdmodel.PDPage;
import lombok.AccessLevel;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.Setter;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
@@ -16,36 +17,17 @@ import net.sourceforge.lept4j.Pix;
import net.sourceforge.tess4j.ITessAPI;
@Getter
@FieldDefaults(level = AccessLevel.PRIVATE)
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class RenderedPageOcrImage implements OcrImage {
final String absoluteImagePath;
final int height;
final int width;
final PageInformation pageInformation;
final Pix pix;
@Setter
int height;
int width;
PageInformation pageInformation;
Pix pix;
int rotationDegrees;
@SneakyThrows
public RenderedPageOcrImage(RenderedPageImageFile renderedPageImageFile, PDDocument document) {
this.pageInformation = PageInformation.fromPDPage(renderedPageImageFile.pageNumber(), document.getPage(renderedPageImageFile.pageNumber() - 1));
this.absoluteImagePath = renderedPageImageFile.absoluteFilePath();
this.pix = Leptonica1.pixRead(absoluteImagePath);
this.height = getPix().h;
this.width = getPix().w;
}
@Override
public int getOptimalPageSegmentationMode() {
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
}
@Override
public AffineTransform getImageCTM() {
@@ -78,17 +60,6 @@ public class RenderedPageOcrImage implements OcrImage {
}
@Override
public QuadPoint getImageBounds() {
if (rotationDegrees == 90 || rotationDegrees == 270) {
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, width), new Point2D.Double(height, width), new Point2D.Double(height, 0));
} else {
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, height), new Point2D.Double(width, height), new Point2D.Double(width, 0));
}
}
@Override
public int getPageNumber() {
@@ -107,7 +78,7 @@ public class RenderedPageOcrImage implements OcrImage {
// PDFBox always returns page height and width based on rotation
double pageWidth;
if (pageInformation.rotationDegrees == 90 || pageInformation.rotationDegrees == 270) {
if (pageInformation.rotationDegrees() == 90 || pageInformation.rotationDegrees() == 270) {
pageWidth = pageInformation.height();
} else {
pageWidth = pageInformation.width();
@@ -116,14 +87,4 @@ public class RenderedPageOcrImage implements OcrImage {
return pageWidth / width;
}
private record PageInformation(int height, int width, int number, int rotationDegrees) {
public static PageInformation fromPDPage(int pageNum, PDPage page) {
return new PageInformation((int) page.getCropBox().getHeight(), (int) page.getCropBox().getWidth(), pageNum, page.getRotation());
}
}
}
@@ -7,29 +7,35 @@ import org.apache.pdfbox.pdmodel.font.PDFont;
import org.apache.pdfbox.util.Matrix;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
import io.github.karols.hocr4j.Bounds;
import io.github.karols.hocr4j.Word;
import lombok.AccessLevel;
import lombok.Getter;
import lombok.Setter;
import lombok.experimental.FieldDefaults;
@Getter
@FieldDefaults(level = AccessLevel.PRIVATE, makeFinal = true)
@FieldDefaults(level = AccessLevel.PRIVATE)
public class TextPositionInImage {
QuadPoint position;
String text;
AffineTransform imageCTM;
final QuadPoint position;
final String text;
final AffineTransform imageCTM;
@Setter
FontMetricsFactory fontMetricsFactory;
@Setter
FontStyle fontStyle;
public TextPositionInImage(Word word, AffineTransform imageCTM, FontMetricsFactory fontMetricsFactory) {
public TextPositionInImage(Word word, AffineTransform imageCTM, FontMetricsFactory fontMetricsFactory, FontStyle fontStyle) {
this.position = QuadPoint.fromBounds(word.getBounds());
this.text = word.getText();
this.imageCTM = imageCTM;
this.fontMetricsFactory = fontMetricsFactory;
this.fontStyle = fontStyle;
}
@@ -90,6 +96,13 @@ public class TextPositionInImage {
}
public double getTextHeight() {
var metrics = fontMetricsFactory.calculateMetrics(text, getTransformedWidth(), getTransformedHeight());
return fontMetricsFactory.calculateFontSize(text, getTransformedWidth()) * metrics.getHeightScaling();
}
public double getHeight() {
return position.a().distance(position.b());
@@ -0,0 +1,9 @@
package com.knecon.fforesight.service.ocr.processor.model;
import net.sourceforge.lept4j.Pix;
public interface UnprocessedImage {
Pix asPix();
}
@@ -0,0 +1,58 @@
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
import java.util.List;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import lombok.AccessLevel;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.experimental.FieldDefaults;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
@Getter
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public final class FontStyleDetectionModel {
QuadPoint imageBounds;
Pix image;
List<TextPositionAndWordImage> textPositionsAndWordImages;
public static FontStyleDetectionModel fromOcrResult(OcrResult ocrResult, FontMetricsFactory fontMetricsFactory, OcrServiceSettings settings) {
var image = Leptonica1.pixRead(ocrResult.tesseractOutputFilePath() + ".tiff");
var wordPixes = ocrResult.getAllWords().stream().filter(word -> !word.isBlank()).map(word -> TextPositionAndWordImage.create(ocrResult.image().getImageCTM(), word, image, settings, fontMetricsFactory)).toList();
return new FontStyleDetectionModel(ocrResult.image().getImageCoordinatesInInitialUserSpace(), image, wordPixes);
}
public List<TextPositionInImage> getTextPositionInImages() {
return textPositionsAndWordImages.stream().map(TextPositionAndWordImage::getTextPositionInImage).toList();
}
public List<WordImage> getWordImages() {
return textPositionsAndWordImages.stream().map(TextPositionAndWordImage::getWordImage).toList();
}
public void dispose() {
LeptUtils.disposePix(image);
getWordImages().forEach(WordImage::dispose);
}
}
@@ -0,0 +1,52 @@
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
import java.awt.geom.AffineTransform;
import java.util.Objects;
import org.apache.commons.math3.ml.clustering.Clusterable;
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import io.github.karols.hocr4j.Word;
import lombok.Getter;
import net.sourceforge.lept4j.Pix;
@Getter
public final class TextPositionAndWordImage implements Clusterable {
private final TextPositionInImage textPositionInImage;
private final WordImage wordImage;
public TextPositionAndWordImage(TextPositionInImage textPositionInImage, WordImage wordImage) {
this.textPositionInImage = textPositionInImage;
this.wordImage = wordImage;
}
public static TextPositionAndWordImage create(AffineTransform imageCTM, Word word, Pix image, OcrServiceSettings settings, FontMetricsFactory fontMetricsFactory) {
TextPositionInImage textPositionInImage = new TextPositionInImage(word, imageCTM, fontMetricsFactory, FontStyle.REGULAR);
WordImage wordImage = new WordImage(textPositionInImage.getTextHeight(), word, image, settings);
return new TextPositionAndWordImage(textPositionInImage, wordImage);
}
@Override
public double[] getPoint() {
return wordImage.getPoint();
}
public double getTextHeight() {
return wordImage.getTextHeight();
}
}
@@ -0,0 +1,71 @@
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
import org.apache.commons.math3.ml.clustering.Clusterable;
import com.knecon.fforesight.service.ocr.processor.service.scriptdetection.StrokeWidthCalculator;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
import io.github.karols.hocr4j.Word;
import lombok.AccessLevel;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.experimental.FieldDefaults;
import net.sourceforge.lept4j.Box;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
@Getter
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class WordImage implements Clusterable {
Pix image;
String text;
double textHeight;
OcrServiceSettings settings;
public WordImage(double textHeight, Word word, Pix originalImage, OcrServiceSettings settings) {
Box box = new Box(word.getBounds().getLeft(), word.getBounds().getTop(), word.getBounds().getWidth(), word.getBounds().getHeight(), 1);
this.image = Leptonica1.pixClipRectangle(originalImage, box, null);
box.clear();
this.text = word.getText();
this.textHeight = textHeight;
this.settings = settings;
}
public boolean hasLargerStrokeWidth(double strokeWidth) {
int roundedStrokeWidth = (int) Math.round(strokeWidth);
double roundingError = (roundedStrokeWidth - strokeWidth) / strokeWidth;
// add 1 to open a bit bigger than the estimated regular stroke width
Pix openedPix = Leptonica1.pixOpenBrick(null, image, roundedStrokeWidth + 1, roundedStrokeWidth + 1);
double openedPixelDensity = ImageProcessingUtils.calculatePixelDensity(openedPix);
double pixelDensity = ImageProcessingUtils.calculatePixelDensity(image);
LeptUtils.disposePix(openedPix);
return (openedPixelDensity * (1 + roundingError)) / pixelDensity > (settings.getBoldThreshold());
}
@Override
public double[] getPoint() {
return new double[]{textHeight};
}
public void dispose() {
LeptUtils.disposePix(image);
}
}
@@ -3,19 +3,21 @@ package com.knecon.fforesight.service.ocr.processor.service;
import java.io.InputStream;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Collections;
import java.util.HashMap;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.concurrent.BlockingQueue;
import java.util.concurrent.LinkedBlockingDeque;
import java.util.stream.Collectors;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageOcrImage;
import com.knecon.fforesight.service.ocr.processor.service.threads.ProcessIOLogger;
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
import com.knecon.fforesight.service.ocr.processor.service.threads.BlockingQueueFiller;
import com.knecon.fforesight.service.ocr.processor.service.threads.GhostScriptOutputHandler;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import com.knecon.fforesight.service.ocr.processor.utils.ListSplittingUtils;
@@ -42,17 +44,19 @@ public class GhostScriptService {
String documentAbsolutePath,
Path tmpImageDir,
PDDocument document,
BlockingQueue<OcrImage> imageOutputQueue,
BlockingQueue<UnprocessedImage> imageProcessingQueue,
Statistics stats) {
BlockingQueue<RenderedPageImageFile> imageFileCollectorQueue = new LinkedBlockingDeque<>();
BlockingQueueFiller asyncTransferThread = new BlockingQueueFiller(imageFileCollectorQueue, imageProcessingQueue);
asyncTransferThread.start();
int numOfProcesses = Math.min(settings.getGsProcessCount(), stitchedPageNumbers.size());
List<List<ProcessInfo>> processInfoBatches = buildSubListForEachProcess(stitchedPageNumbers,
numOfProcesses,
2 * settings.getOcrThreadCount()); // use 2 times the thread count as batch size, such that GS generates the rendered pages as needed by the OCR Threads
256 * numOfProcesses); // GS has a limit on how many pageIndices per call are possible, so we limit it to 256 pages per process
for (int batchIdx = 0; batchIdx < processInfoBatches.size(); batchIdx++) {
long timestamp = System.currentTimeMillis();
List<RenderedPageImageFile> renderedPageImageFiles = Collections.synchronizedList(new LinkedList<>());
List<ProcessInfo> processInfos = processInfoBatches.get(batchIdx);
log.info("Batch {}: Running {} gs processes with ({}) pages each",
@@ -63,9 +67,9 @@ public class GhostScriptService {
int finalBatchIdx = batchIdx;
List<Process> processes = processInfos.stream()
.parallel()
.map(info -> buildCmdArgs(info.processIdx(), finalBatchIdx, info.stitchedPageNumbers(), tmpImageDir, documentAbsolutePath, renderedPageImageFiles))
.peek(s -> log.debug(String.join(" ", s)))
.map(this::executeProcess)
.map(info -> buildCmdArgs(info.processIdx(), finalBatchIdx, info.stitchedPageNumbers(), tmpImageDir, documentAbsolutePath))
.peek(s -> log.debug(String.join(" ", s.cmdArgs())))
.map(processInfo -> executeProcess(processInfo, imageFileCollectorQueue))
.toList();
List<Integer> processExitCodes = new LinkedList<>();
@@ -73,14 +77,9 @@ public class GhostScriptService {
processExitCodes.add(process.waitFor());
}
stats.increasePDF2ImgDuration(System.currentTimeMillis() - timestamp);
log.info("Batch {}: Ghostscript processes finished with exit codes " + processExitCodes, batchIdx);
for (RenderedPageImageFile renderedPageImageFile : renderedPageImageFiles) {
OcrImage image = new RenderedPageOcrImage(renderedPageImageFile, document);
imageOutputQueue.put(image);
}
}
asyncTransferThread.setAllImagesQueued(true);
}
@@ -107,20 +106,28 @@ public class GhostScriptService {
@SneakyThrows
private String[] buildCmdArgs(Integer processIdx,
Integer batchIdx,
List<Integer> stitchedImagePageIndices,
Path outputDir,
String documentAbsolutePath,
List<RenderedPageImageFile> fullPageImages) {
private ProcessCmdsAndRenderedImageFiles buildCmdArgs(Integer processIdx,
Integer batchIdx,
List<Integer> stitchedImagePageIndices,
Path outputDir,
String documentAbsolutePath) {
String imagePathFormat = outputDir.resolve("output_" + processIdx + "_" + batchIdx + ".%04d" + FORMAT).toFile().toString();
Map<Integer, RenderedPageImageFile> fullPageImages = new HashMap<>();
for (int i = 0; i < stitchedImagePageIndices.size(); i++) {
Integer pageNumber = stitchedImagePageIndices.get(i);
fullPageImages.add(new RenderedPageImageFile(pageNumber, String.format(imagePathFormat, i + 1)));
fullPageImages.put(pageNumber, new RenderedPageImageFile(pageNumber, String.format(imagePathFormat, i + 1)));
}
String[] cmdArgs = buildCmdArgs(stitchedImagePageIndices, documentAbsolutePath, imagePathFormat);
return new ProcessCmdsAndRenderedImageFiles(cmdArgs, fullPageImages);
}
private String[] buildCmdArgs(List<Integer> stitchedImagePageIndices, String documentAbsolutePath, String imagePathFormat) {
StringBuilder sPageList = new StringBuilder();
int i = 1;
for (Integer integer : stitchedImagePageIndices) {
@@ -131,18 +138,19 @@ public class GhostScriptService {
i++;
}
return new String[]{"gs", "-dNOPAUSE", "-sDEVICE=" + DEVICE, "-r" + settings.getDpi(), "-sPageList=" + sPageList, "-sOutputFile=" + imagePathFormat, documentAbsolutePath, "-c", "quit"};
String[] cmdArgs = new String[]{"gs", "-dNOPAUSE", "-sDEVICE=" + DEVICE, "-r" + settings.getDpi(), "-sPageList=" + sPageList, "-sOutputFile=" + imagePathFormat, documentAbsolutePath, "-c", "quit"};
return cmdArgs;
}
@SneakyThrows
private Process executeProcess(String[] cmdArgs) {
private Process executeProcess(ProcessCmdsAndRenderedImageFiles processInfo, BlockingQueue<RenderedPageImageFile> imageFileCollectorQueue) {
Process p = Runtime.getRuntime().exec(cmdArgs);
Process p = Runtime.getRuntime().exec(processInfo.cmdArgs());
InputStream stdOut = p.getInputStream();
ProcessIOLogger stdOutLogger = new ProcessIOLogger(stdOut, "GS", ProcessIOLogger.Type.STD_OUT);
GhostScriptOutputHandler stdOutLogger = GhostScriptOutputHandler.stdOut(stdOut, processInfo.renderedPageImageFiles(), imageFileCollectorQueue);
InputStream stdError = p.getErrorStream();
ProcessIOLogger stdErrorLogger = new ProcessIOLogger(stdError, "GS", ProcessIOLogger.Type.ERROR);
GhostScriptOutputHandler stdErrorLogger = GhostScriptOutputHandler.errorHandler(stdError);
stdOutLogger.start();
stdErrorLogger.start();
@@ -150,6 +158,10 @@ public class GhostScriptService {
}
private record ProcessCmdsAndRenderedImageFiles(String[] cmdArgs, Map<Integer, RenderedPageImageFile> renderedPageImageFiles) {
}
private record ProcessInfo(Integer processIdx, List<Integer> stitchedPageNumbers) {
}
@@ -1,7 +1,6 @@
package com.knecon.fforesight.service.ocr.processor.service;
import java.awt.Graphics;
import java.awt.image.BufferedImage;
import java.awt.geom.Rectangle2D;
import java.io.IOException;
import java.util.LinkedList;
import java.util.List;
@@ -18,13 +17,12 @@ import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.util.Matrix;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import lombok.Getter;
@@ -33,8 +31,7 @@ import lombok.SneakyThrows;
@Getter
public class ImageStreamEngine extends PDFStreamEngine {
private ExtractedOcrImage currentImageOnPage;
private List<ExtractedOcrImage> imagesOnCurrentPage;
private List<ExtractedImage> imagesOnCurrentPage;
private OcrServiceSettings settings;
private int pageNum;
@@ -69,22 +66,15 @@ public class ImageStreamEngine extends PDFStreamEngine {
}
Matrix imageCTM = getGraphicsState().getCurrentTransformationMatrix();
if (imageXObject.getColorSpace() instanceof PDDeviceRGB) {
BufferedImage image = imageXObject.getImage();
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi());
} else if (imageXObject.getColorSpace() instanceof PDDeviceGray) {
BufferedImage image = imageXObject.getImage();
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi());
} else {
BufferedImage pdfImage = imageXObject.getImage();
BufferedImage image = new BufferedImage(pdfImage.getWidth(), pdfImage.getHeight(), BufferedImage.TYPE_BYTE_GRAY);
Graphics g = image.getGraphics();
g.drawImage(pdfImage, 0, 0, null);
g.dispose();
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi());
}
this.imagesOnCurrentPage.add(this.currentImageOnPage);
//imagesOnPages.add(this.currentImageOnPage);
this.imagesOnCurrentPage.add(new ExtractedImage(pageNum,
QuadPoint.fromRectangle2D(new Rectangle2D.Double(0, 0, imageXObject.getWidth(), imageXObject.getHeight())),
imageXObject.getHeight(),
imageXObject.getWidth(),
imageXObject.getImage(),
imageCTM,
imagesOnCurrentPage.size(),
imageXObject.getColorSpace()));
} else if (xobject instanceof PDFormXObject) {
PDFormXObject form = (PDFormXObject) xobject;
showForm(form);
@@ -9,6 +9,7 @@ import java.io.OutputStream;
import java.nio.file.Path;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.concurrent.ArrayBlockingQueue;
import java.util.concurrent.BlockingQueue;
import java.util.stream.IntStream;
@@ -20,8 +21,10 @@ import org.springframework.util.FileSystemUtils;
import com.iqser.red.pdftronlogic.commons.InvisibleElementRemovalService;
import com.iqser.red.pdftronlogic.commons.WatermarkRemovalService;
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.service.scriptdetection.FontStyleDetector;
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
@@ -44,6 +47,7 @@ public class OCRService {
InvisibleElementRemovalService invisibleElementRemovalService;
OcrResultWriter ocrResultWriter;
GhostScriptService ghostScriptService;
FontStyleDetector boldDetector;
/**
@@ -107,7 +111,7 @@ public class OCRService {
int numberOfOcrThreads = Math.min(settings.getOcrThreadCount(), document.getNumberOfPages());
stats = new Statistics(numberOfExtractThreads, numberOfOcrThreads);
BlockingQueue<OcrImage> ocrImageQueue = new ArrayBlockingQueue<>(numberOfOcrThreads);
BlockingQueue<OcrImage> ocrImageQueue = new ArrayBlockingQueue<>((int) (1.5 * numberOfOcrThreads));
OcrImageFactory ocrImageFactory = new OcrImageFactory(document,
documentFile,
@@ -128,16 +132,21 @@ public class OCRService {
.toList();
log.info("Started {} OCR consumer threads, listening for images on the queue", ocrThreads.size());
ocrImageFactory.join();
log.info("Extracted all images, interrupting ocr threads");
log.info("Processed all images, interrupting ocr threads");
ocrThreads.forEach(Thread::interrupt);
for (OCRThread ocrThread : ocrThreads) {
ocrThread.join();
}
log.info("OCR processing has finished, writing results");
log.info("Tesseract OCR has finished for file {} and dossier {}", fileId, dossierId);
timestamp = System.currentTimeMillis();
var dictionariesToUpdate = ocrResultWriter.drawOcrResultsToPdf(document, ocrResults);
Map<Integer, List<OcrResultToWrite>> imageWithTextPositionsPerPage = boldDetector.detectBold(ocrResults, document);
stats.increaseFontStyleDetectionDuration(System.currentTimeMillis() - timestamp);
timestamp = System.currentTimeMillis();
var dictionariesToUpdate = ocrResultWriter.drawOcrResultsToPdf(document, imageWithTextPositionsPerPage);
log.info("Saving document");
document.saveIncremental(out, dictionariesToUpdate);
stats.increaseWritingTextDuration(System.currentTimeMillis() - timestamp);
@@ -6,13 +6,17 @@ import java.util.ArrayList;
import java.util.Collections;
import java.util.LinkedList;
import java.util.List;
import java.util.concurrent.ArrayBlockingQueue;
import java.util.concurrent.BlockingQueue;
import java.util.stream.Collectors;
import org.apache.pdfbox.pdmodel.PDDocument;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
import com.knecon.fforesight.service.ocr.processor.service.threads.ImageExtractionThread;
import com.knecon.fforesight.service.ocr.processor.service.threads.ImageProcessingThread;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import com.knecon.fforesight.service.ocr.processor.utils.ListSplittingUtils;
@@ -29,6 +33,8 @@ public class OcrImageFactory {
File documentFile;
Path tmpImageDir;
GhostScriptService ghostScriptService;
BlockingQueue<UnprocessedImage> imageProcessingQueue;
ImageProcessingThread imageProcessingThread;
BlockingQueue<OcrImage> imageOutputQueue;
List<ImageExtractionThread> imageExtractionThreads;
List<Integer> stitchedPageNumbers;
@@ -40,7 +46,7 @@ public class OcrImageFactory {
Path tmpImageDir,
int numberOfThreads,
GhostScriptService ghostScriptService,
BlockingQueue<OcrImage> imageOutputQueue,
BlockingQueue<OcrImage> imageOcrQueue,
OcrProgressLogger logger,
OcrServiceSettings settings,
Statistics stats) {
@@ -49,7 +55,8 @@ public class OcrImageFactory {
this.documentFile = documentFile;
this.tmpImageDir = tmpImageDir;
this.ghostScriptService = ghostScriptService;
this.imageOutputQueue = imageOutputQueue;
this.imageOutputQueue = imageOcrQueue;
this.imageProcessingQueue = new ArrayBlockingQueue<>(imageOcrQueue.remainingCapacity());
this.stitchedPageNumbers = Collections.synchronizedList(new LinkedList<>());
this.stats = stats;
@@ -57,8 +64,10 @@ public class OcrImageFactory {
List<List<Integer>> balancedPageNumbers = ListSplittingUtils.buildBalancedContinuousSublist(document.getNumberOfPages(), numberOfThreads);
for (int i = 0; i < balancedPageNumbers.size(); i++) {
imageExtractionThreads.add(new ImageExtractionThread(i, balancedPageNumbers.get(i), documentFile, logger, stats, settings, imageOutputQueue, stitchedPageNumbers));
imageExtractionThreads.add(new ImageExtractionThread(i, balancedPageNumbers.get(i), documentFile, logger, stats, settings, imageProcessingQueue, stitchedPageNumbers));
}
this.imageProcessingThread = new ImageProcessingThread(imageProcessingQueue, imageOcrQueue, stats, settings, document);
log.info("Started {} image extraction threads, with ({}) pages each",
imageExtractionThreads.size(),
imageExtractionThreads.stream().map(ImageExtractionThread::getPageIndices).map(List::size).map(String::valueOf).collect(Collectors.joining(", ")));
@@ -70,6 +79,8 @@ public class OcrImageFactory {
for (ImageExtractionThread imageExtractionThread : imageExtractionThreads) {
imageExtractionThread.start();
}
imageProcessingThread.start();
}
@@ -79,11 +90,16 @@ public class OcrImageFactory {
for (ImageExtractionThread imageExtractionThread : imageExtractionThreads) {
imageExtractionThread.join();
}
if (stitchedPageNumbers.isEmpty()) {
return;
if (!stitchedPageNumbers.isEmpty()) {
ghostScriptService.renderPagesAsImagesBatchedAndAddToQueue(stitchedPageNumbers, documentFile.toString(), tmpImageDir, document, imageProcessingQueue, stats);
}
ghostScriptService.renderPagesAsImagesBatchedAndAddToQueue(stitchedPageNumbers, documentFile.toString(), tmpImageDir, document, imageOutputQueue, stats);
imageProcessingThread.setAllImagesExtracted(true);
imageProcessingThread.interrupt();
imageProcessingThread.join();
}
}
@@ -2,11 +2,11 @@ package com.knecon.fforesight.service.ocr.processor.service;
import java.awt.Color;
import java.awt.geom.Point2D;
import java.util.Collection;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.cos.COSName;
@@ -20,11 +20,9 @@ import org.apache.pdfbox.pdmodel.graphics.optionalcontent.PDOptionalContentPrope
import org.apache.pdfbox.pdmodel.graphics.state.RenderingMode;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.service.fonts.Type0FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import lombok.AccessLevel;
@@ -44,19 +42,17 @@ public class OcrResultWriter {
@SneakyThrows
public Set<COSDictionary> drawOcrResultsToPdf(PDDocument document, List<OcrResult> ocrResults) {
public Set<COSDictionary> drawOcrResultsToPdf(PDDocument document, Map<Integer, List<OcrResultToWrite>> imagesWithResultsPerPage) {
FontMetricsFactory fontMetricsFactory = new Type0FontMetricsFactory(document);
Set<COSDictionary> dictionariesToUpdate = new HashSet<>();
Map<Integer, List<OcrResult>> resultsPerPage = ocrResults.stream().collect(Collectors.groupingBy(result -> result.image().pageNumber()));
resultsPerPage.keySet().forEach(pageNumber -> drawResultsPerPage(document, pageNumber, resultsPerPage, dictionariesToUpdate, fontMetricsFactory));
imagesWithResultsPerPage.keySet().forEach(pageNumber -> drawResultsPerPage(document, pageNumber, imagesWithResultsPerPage.get(pageNumber), dictionariesToUpdate));
dictionariesToUpdate.add(document.getDocumentInformation().getCOSObject());
return dictionariesToUpdate;
}
@SneakyThrows
private void drawResultsPerPage(PDDocument document, Integer pageNumber, Map<Integer, List<OcrResult>> resultsPerPage, Set<COSDictionary> dictionariesToUpdate, FontMetricsFactory fontMetricsFactory) {
private void drawResultsPerPage(PDDocument document, Integer pageNumber, List<OcrResultToWrite> ocrResultToWrite, Set<COSDictionary> dictionariesToUpdate) {
var pdPage = document.getPage(pageNumber - 1);
@@ -69,7 +65,7 @@ public class OcrResultWriter {
escapeContentStreams(document, pdPage);
List<TextPositionInImage> words = buildTextPositionsOnPage(pageNumber, resultsPerPage, fontMetricsFactory);
List<TextPositionInImage> words = ocrResultToWrite.stream().map(OcrResultToWrite::textPositionInImage).flatMap(Collection::stream).toList();
try (var contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true)) {
// write invisible ocr text inside tagged content
@@ -86,7 +82,6 @@ public class OcrResultWriter {
// write visible ocr text inside optional group
contentStream.beginMarkedContent(COSName.OC, textDebugLayer);
contentStream.saveGraphicsState();
contentStream.setNonStrokingColor(Color.BLUE);
words.forEach(word -> drawVisibleWord(word, contentStream));
contentStream.restoreGraphicsState();
contentStream.endMarkedContent();
@@ -94,7 +89,9 @@ public class OcrResultWriter {
// write word bounding boxes (tesseract output) inside optional group
contentStream.beginMarkedContent(COSName.OC, bBoxDebugLayer);
contentStream.saveGraphicsState();
resultsPerPage.get(pageNumber).stream().map(OcrResult::image).forEach(image -> drawGrid(contentStream, image.position()));
ocrResultToWrite.stream()
.map(OcrResultToWrite::imageBoundingBox)
.forEach(imagePosition -> drawGrid(contentStream, imagePosition));
words.stream().map(TextPositionInImage::getTransformedTextBBox).forEach(word -> drawRectangle(contentStream, word));
contentStream.restoreGraphicsState();
contentStream.endMarkedContent();
@@ -105,15 +102,6 @@ public class OcrResultWriter {
}
private static List<TextPositionInImage> buildTextPositionsOnPage(Integer pageNumber, Map<Integer, List<OcrResult>> resultsPerPage, FontMetricsFactory fontMetricsFactory) {
return resultsPerPage.get(pageNumber)
.stream()
.flatMap(result -> result.getAllWords().stream().filter(word -> !word.isBlank()).map(word -> new TextPositionInImage(word, result.image().ctm(), fontMetricsFactory)))
.toList();
}
@SneakyThrows
private static void escapeContentStreams(PDDocument document, PDPage pdPage) {
// We need to append to the contentstream, otherwise the content could be overlapped by images
@@ -196,6 +184,11 @@ public class OcrResultWriter {
private void drawWord(TextPositionInImage position, PDPageContentStream contentStream, RenderingMode renderingMode) {
try {
contentStream.setNonStrokingColor(switch (position.getFontStyle()) {
case BOLD -> Color.RED;
case ITALIC -> Color.GREEN;
default -> Color.BLUE;
});
contentStream.beginText();
contentStream.setRenderingMode(renderingMode);
contentStream.setFont(position.getFont(), (float) position.getFontSize());
@@ -15,14 +15,18 @@ public class Statistics {
List<Long> tesseractDuration;
AtomicLong pdf2ImgDuration;
AtomicLong writingTextDuration;
AtomicLong imageProcessingDuration;
AtomicLong fontStyleDetectionDuration;
public Statistics(int numberOfExtractThreads, int numberOfOcrThreads) {
this.imageExtraction = Collections.synchronizedList(new ArrayList<>(Collections.nCopies(numberOfExtractThreads, 0L)));
this.tesseractDuration = Collections.synchronizedList(new ArrayList<>(Collections.nCopies(numberOfOcrThreads, 0L)));
this.fontStyleDetectionDuration = new AtomicLong(0);
this.pdf2ImgDuration = new AtomicLong(0);
this.writingTextDuration = new AtomicLong(0);
this.imageProcessingDuration = new AtomicLong(0);
}
@@ -32,6 +36,12 @@ public class Statistics {
}
public void increaseImageProcessing(long duration) {
imageProcessingDuration.addAndGet(duration);
}
public void increaseTesseractDuration(int threadId, long duration) {
tesseractDuration.set(threadId, tesseractDuration.get(threadId) + duration);
@@ -49,19 +59,27 @@ public class Statistics {
writingTextDuration.addAndGet(duration);
}
public void increaseFontStyleDetectionDuration(long duration) {
fontStyleDetectionDuration.addAndGet(duration);
}
@Override
public String toString() {
return String.format("imageExtraction: mean %.2f s, max %.2f s, min %.2f, tesseract: mean %.2f s, max %.2f s, min %.2f, PDF2Img=%.2f s, writingText=%.2f s",
return String.format(
"imageExtraction: mean %.2f s, max %.2f s, min %.2f, tesseract: mean %.2f s, max %.2f s, min %.2f, ImageProcessing=%.2f s, PDF2Img=%.2f s, writingText=%.2f s, FontstyleDetection=%.2f s",
((float) imageExtraction.stream().mapToLong(Long::longValue).average().orElse(0) / 1000),
((float) imageExtraction.stream().mapToLong(Long::longValue).max().orElse(0) / 1000),
((float) imageExtraction.stream().mapToLong(Long::longValue).min().orElse(0) / 1000),
((float) tesseractDuration.stream().mapToLong(Long::longValue).average().orElse(0) / 1000),
((float) tesseractDuration.stream().mapToLong(Long::longValue).max().orElse(0) / 1000),
((float) tesseractDuration.stream().mapToLong(Long::longValue).min().orElse(0) / 1000),
(float) imageProcessingDuration.get() / 1000,
(float) pdf2ImgDuration.get() / 1000,
(float) writingTextDuration.get() / 1000);
(float) writingTextDuration.get() / 1000,
(float) fontStyleDetectionDuration.get() / 1000);
}
}
@@ -36,6 +36,7 @@ public interface FontMetricsFactory {
PDFont getFont();
HeightAndDescent calculateHeightAndDescent(String text);
}
@@ -0,0 +1,5 @@
package com.knecon.fforesight.service.ocr.processor.service.fonts;
public enum FontStyle {
REGULAR, BOLD, ITALIC
}
@@ -1,6 +1,9 @@
package com.knecon.fforesight.service.ocr.processor.service.fonts;
import java.io.ByteArrayInputStream;
import java.util.Collections;
import java.util.List;
import java.util.Set;
import org.apache.fontbox.ttf.GlyphData;
import org.apache.fontbox.ttf.TTFParser;
@@ -12,22 +15,41 @@ import org.apache.pdfbox.pdmodel.font.PDType0Font;
import com.knecon.fforesight.service.ocr.processor.model.HeightAndDescent;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
import software.amazon.awssdk.services.s3.endpoints.internal.Value;
@Slf4j
@RequiredArgsConstructor
public class Type0FontMetricsFactory implements FontMetricsFactory {
private final PDType0Font type0Font;
private final TrueTypeFont trueTypeFont;
// for this specific font back-/forward-slashes have a lot of descent screwing up the font size and therefore bold detection. So if we find such a character we ignore its descent.
private static final Set<Integer> slashGlyphIds = Set.of(18, 63);
public static Type0FontMetricsFactory regular(PDDocument document) {
return createFromResource("fonts/cmu-regular.ttf", document);
}
public static Type0FontMetricsFactory bold(PDDocument document) {
return createFromResource("fonts/cmu-bold.ttf", document);
}
@SneakyThrows
public Type0FontMetricsFactory(PDDocument document) {
private static Type0FontMetricsFactory createFromResource(String resourcePath, PDDocument document) {
try (var in = Thread.currentThread().getContextClassLoader().getResourceAsStream("fonts/cmu-regular.ttf"); var buffer = new RandomAccessReadBuffer(in)) {
this.trueTypeFont = new TTFParser().parse(buffer); // since Type0Font can be descendant from any font, we need to remember the original TrueTypeFont for the glyph information
this.type0Font = PDType0Font.load(document, this.trueTypeFont, false); // use Type0Font for unicode support
try (var in = Thread.currentThread().getContextClassLoader().getResourceAsStream(resourcePath); var buffer = new RandomAccessReadBuffer(in)) {
TrueTypeFont trueTypeFont = new TTFParser().parse(buffer); // since Type0Font can be descendant from any font, we need to remember the original TrueTypeFont for the glyph information
PDType0Font type0Font = PDType0Font.load(document, trueTypeFont, true); // use Type0Font for unicode support
return new Type0FontMetricsFactory(type0Font, trueTypeFont);
}
}
@@ -55,8 +77,9 @@ public class Type0FontMetricsFactory implements FontMetricsFactory {
if (glyph == null || glyph.getBoundingBox() == null) {
continue;
}
descent = Math.min(descent, glyph.getYMinimum());
if (!slashGlyphIds.contains(glyphId)) {
descent = Math.min(descent, glyph.getYMinimum());
}
height = Math.max(height, glyph.getYMaximum());
} catch (Exception e) {
log.warn("descent and height of string {} could not be parsed, using average fallback value!", text);
@@ -0,0 +1,158 @@
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
import java.util.Comparator;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.stream.Stream;
import org.apache.commons.math3.ml.clustering.Cluster;
import org.apache.commons.math3.ml.clustering.DBSCANClusterer;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.FontStyleDetectionModel;
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.TextPositionAndWordImage;
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.WordImage;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
import com.knecon.fforesight.service.ocr.processor.service.fonts.Type0FontMetricsFactory;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class FontStyleDetector {
OcrServiceSettings settings;
StrokeWidthCalculator strokeWidthCalculator;
/**
* Implementation of the MOBDoB algorithm, refer to the paper here:
* <a href="http://mile.ee.iisc.ac.in/publications/softCopy/DocumentAnalysis/Sai_NCVPRIPG2013.pdf">Script Independent Detection of Bold Words in Multi Font-size Documents</a>
* <p>
* As a high level overview: We cluster all text based on its font size. We determine the cluster with the most words. This is assumed to be regular text.
* We then estimate the average stroke width of that cluster by thinning all text to a single pixel and calculating the ratio of remaining pixels.
* (<a href="http://www.leptonica.org/papers/conn.pdf">Leptonica Documentation on thinning</a>)
* For each word we scale this average strokewidth based on its fontsize compared to the most common fontsize.
* Using the scaled strokewidth we do an opening operation.
* (<a href="https://en.wikipedia.org/wiki/Opening_(morphology)">Opening (Morphology)</a>).
* We then threshold the ratio of remaining pixels to determine whether a word is bold or not.
* <p>
* I did take some liberties though. Firstly, the paper uses text height without ascender/descender height for the clustering. I'm using the previously implemented font size.
* But this is based on text width. Thus, I'm also using the height scaling factor to scale the font size by the text height.
* The paper does not describe its clustering algorithm, so I've decided on DBSCAN due to its good runtime and readily available implementation by apache commons math.
* Moreover, the paper states that stroke width scales linearly with text height. I've come to the conclusion this is not the case.
* It seems it scales with the square root of the text height. Or at least this seemed to give the best results.
*/
public Map<Integer, List<OcrResultToWrite>> detectBold(List<OcrResult> ocrResults, PDDocument document) {
FontMetricsFactory fontMetricsFactory = Type0FontMetricsFactory.regular(document);
if (!settings.isBoldDetection()) {
return OcrResultToWrite.buildOcrResultsToWrite(ocrResults, fontMetricsFactory);
}
Map<Integer, List<OcrResultToWrite>> ocrResultToWritePerPage = new HashMap<>();
DBSCANClusterer<TextPositionAndWordImage> clusterer = new DBSCANClusterer<>(0.5, 1);
FontMetricsFactory boldFontMetricsFactory = Type0FontMetricsFactory.bold(document);
for (OcrResult result : ocrResults) {
FontStyleDetectionModel fontStyleDetectionModel = FontStyleDetectionModel.fromOcrResult(result, fontMetricsFactory, settings);
List<Cluster<TextPositionAndWordImage>> clusters = clusterer.cluster(fontStyleDetectionModel.getTextPositionsAndWordImages());
Optional<Cluster<TextPositionAndWordImage>> largestCluster = clusters.stream().max(Comparator.comparingInt(cluster -> cluster.getPoints().size()));
if (largestCluster.isEmpty()) {
insertResultIntoMap(result.image().getPageNumber(), ocrResultToWritePerPage, fontStyleDetectionModel);
continue;
}
List<TextPositionAndWordImage> wordsWithMostCommonTextHeight = largestCluster.get().getPoints();
double standardTextHeight = calculateStandardTextheight(wordsWithMostCommonTextHeight);
double regularStrokeWidth = calculateRegularStrokeWidth(wordsWithMostCommonTextHeight);
for (TextPositionAndWordImage textPositionsAndWordImage : fontStyleDetectionModel.getTextPositionsAndWordImages()) {
decideOnFontStyle(textPositionsAndWordImage, regularStrokeWidth, standardTextHeight, boldFontMetricsFactory);
}
insertResultIntoMap(result.image().getPageNumber(), ocrResultToWritePerPage, fontStyleDetectionModel);
fontStyleDetectionModel.dispose();
}
log.info("Finished bold detection");
return ocrResultToWritePerPage;
}
private static double calculateStandardTextheight(List<TextPositionAndWordImage> wordsWithMostCommonTextHeight) {
return wordsWithMostCommonTextHeight.stream()
.map(TextPositionAndWordImage::getWordImage)
.mapToDouble(WordImage::getTextHeight)
.filter(Double::isFinite)
.average()
.orElseThrow();
}
private double calculateRegularStrokeWidth(List<TextPositionAndWordImage> wordsWithMostCommonTextHeight) {
return wordsWithMostCommonTextHeight.stream()
.mapToDouble(textPositionAndWordImage -> strokeWidthCalculator.calculate(textPositionAndWordImage.getWordImage().getImage()))
.filter(Double::isFinite)
.average()
.orElseThrow();
}
private static void insertResultIntoMap(int pageNumber, Map<Integer, List<OcrResultToWrite>> ocrResultToWritePerPage, FontStyleDetectionModel fontStyleDetectionModel) {
OcrResultToWrite ocrResult = OcrResultToWrite.fromFontStyleDetectionModel(fontStyleDetectionModel);
ocrResultToWritePerPage.compute(pageNumber, (key, existingList) -> {
if (existingList == null) {
return List.of(ocrResult);
} else {
return Stream.concat(existingList.stream(), Stream.of(ocrResult)).toList();
}
});
}
private void decideOnFontStyle(TextPositionAndWordImage textPositionsAndWordImage,
double standardStrokeWidth,
double standardTextHeight,
FontMetricsFactory boldFontMetricsFactory) {
double scaledStrokeWidth = scaleStrokeWidthByFontSize(textPositionsAndWordImage, standardStrokeWidth, standardTextHeight);
if (textPositionsAndWordImage.getWordImage().hasLargerStrokeWidth(scaledStrokeWidth)) {
textPositionsAndWordImage.getTextPositionInImage().setFontMetricsFactory(boldFontMetricsFactory);
textPositionsAndWordImage.getTextPositionInImage().setFontStyle(FontStyle.BOLD);
} else {
textPositionsAndWordImage.getTextPositionInImage().setFontStyle(FontStyle.REGULAR);
}
}
private static double scaleStrokeWidthByFontSize(TextPositionAndWordImage textPositionsAndWordImage, double standardStrokeWidth, double standardFontSize) {
double influenceOfFontSize = 1.0; // the paper states that stroke width scales exactly linearly with font size. This did not seem to be true for me. Maybe some of the preprocessing steps are affecting this.
double fontsizeScalingFactor = Math.sqrt(textPositionsAndWordImage.getWordImage().getTextHeight() / standardFontSize);
return standardStrokeWidth + (influenceOfFontSize * (fontsizeScalingFactor - 1) * standardStrokeWidth);
}
}
@@ -0,0 +1,57 @@
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
import lombok.AccessLevel;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.Sel;
import net.sourceforge.lept4j.util.LeptUtils;
/**
* This code is a good start for detecting italic text, although it has a few issues especially with glyphs which are naturally slanted. E.g. z, 2, 7, /
* If we want this maybe we should exclude these glyphs and then it might have less false positives. But in its current state i don't recommend using it.
*/
@NoArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class ItalicDetector {
static String italicKernel = "ooxxooxxooxxoxxooXxooxxoxxooxxooxxoo";
Sel italicSel = Leptonica1.selCreateFromString(italicKernel, 9, 4, "italicKernel");
Sel brickSel = Leptonica1.selCreateBrick(3, 4, 1, 2, 1);
public boolean isItalic(Pix pix) {
Pix preprocessed = preprocess(pix);
Pix flipped = Leptonica1.pixFlipLR(null, pix);
Pix flippedPreprocessed = preprocess(flipped);
Leptonica1.pixFlipLR(flippedPreprocessed, flippedPreprocessed);
double pixelDensity = ImageProcessingUtils.calculatePixelDensity(preprocessed);
double flippedPixelDensity = ImageProcessingUtils.calculatePixelDensity(flippedPreprocessed);
LeptUtils.disposePix(preprocessed);
LeptUtils.disposePix(flipped);
LeptUtils.disposePix(flippedPreprocessed);
return flippedPixelDensity / pixelDensity < 0.85;
}
private Pix preprocess(Pix pix) {
Pix eroded = Leptonica1.pixErode(null, pix, italicSel.getPointer());
Pix dilated = Leptonica1.pixDilate(null, eroded, brickSel.getPointer());
LeptUtils.disposePix(eroded);
return dilated;
}
public void dispose() {
LeptUtils.dispose(italicSel);
LeptUtils.dispose(brickSel);
}
}
@@ -0,0 +1,58 @@
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
import static net.sourceforge.lept4j.ILeptonica.L_THIN_FG;
import java.nio.IntBuffer;
import org.springframework.stereotype.Service;
import lombok.AccessLevel;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.Sela;
import net.sourceforge.lept4j.util.LeptUtils;
@Service
@NoArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class StrokeWidthCalculator {
Sela thinningSel;
/**
* Uses a series of sels to thin all connected lines to a single pixel. Then the pixel ratio is a good estimation of the stroke width in pixels.
* <a href="http://www.leptonica.org/papers/conn.pdf">Leptonica Documentation on thinning</a>
* Since the baseline is a strokewidth of exactly one, we need to add 1 to the result.
*
* @param input binarized pix with text on it
* @return estimated stroke width in pixels
*/
public double calculate(Pix input) {
init();
Pix thinned = Leptonica1.pixThinConnectedBySet(input, L_THIN_FG, thinningSel, 0);
IntBuffer thinnedPixelCount = IntBuffer.allocate(1);
Leptonica1.pixCountPixels(thinned, thinnedPixelCount, null);
IntBuffer pixelCount = IntBuffer.allocate(1);
Leptonica1.pixCountPixels(input, pixelCount, null);
LeptUtils.disposePix(thinned);
return (double) pixelCount.get() / thinnedPixelCount.get() + 1;
}
private void init() {
if (thinningSel == null) {
thinningSel = Leptonica1.selaMakeThinSets(1, 0);
}
}
}
@@ -0,0 +1,65 @@
package com.knecon.fforesight.service.ocr.processor.service.threads;
import java.util.ArrayList;
import java.util.List;
import java.util.NoSuchElementException;
import java.util.concurrent.BlockingQueue;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.Setter;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
import net.sourceforge.tess4j.TessAPI1;
/*
This just moves the Elements from the GhostScriptOutputListener into the ImageProcessing queue asynchronously
*/
@Slf4j
@RequiredArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class BlockingQueueFiller extends Thread {
final BlockingQueue<RenderedPageImageFile> imageInputQueue;
final BlockingQueue<UnprocessedImage> imageOutputQueue;
@Setter
boolean allImagesQueued;
@SneakyThrows
@Override
public void run() {
// Interrupting signals that the image extraction has finished
try {
while (!allImagesQueued) {
final UnprocessedImage image = imageInputQueue.take();
try {
imageOutputQueue.put(image);
} catch (InterruptedException e) {
imageOutputQueue.put(image);
}
}
} catch (InterruptedException e) {
log.info("All images extracted, emptying processing queue and stopping");
}
// empty the queue
try {
while (true) {
final UnprocessedImage image = imageInputQueue.remove();
imageOutputQueue.put(image);
}
} catch (NoSuchElementException e) {
log.debug("No images left in queue, stopping.");
}
}
}
@@ -0,0 +1,122 @@
package com.knecon.fforesight.service.ocr.processor.service.threads;
import java.io.BufferedReader;
import java.io.File;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Queue;
import java.util.concurrent.BlockingQueue;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RequiredArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class GhostScriptOutputHandler extends Thread {
static Pattern pageFinishedPattern = Pattern.compile("Page (\\d+)");
// If the stdError or stdOut buffer of a thread is not being emptied it might lock the process in case of errors, so we need to empty both streams to prevent a deadlock.
// Since both need to read simultaneously we need to implement the readers as separate threads.
final InputStream is;
final String processName;
final Type type;
final Map<Integer, RenderedPageImageFile> pagesToProcess;
final BlockingQueue<RenderedPageImageFile> renderedPageImageFileOutput;
int currentPageNumber;
public static GhostScriptOutputHandler errorHandler(InputStream is) {
return new GhostScriptOutputHandler(is, "GS", Type.ERROR, null, null);
}
public static GhostScriptOutputHandler stdOut(InputStream is,
Map<Integer, RenderedPageImageFile> pagesToProcess,
BlockingQueue<RenderedPageImageFile> renderedPageImageFileOutput) {
return new GhostScriptOutputHandler(is, "GS", Type.STD_OUT, pagesToProcess, renderedPageImageFileOutput);
}
@SneakyThrows
public void run() {
try (InputStreamReader isr = new InputStreamReader(is); BufferedReader br = new BufferedReader(isr)) {
String line;
while (true) {
line = br.readLine();
if (line == null) {
break;
}
if (type.equals(Type.ERROR)) {
log.error(processName + "_" + type.name() + ">" + line);
} else {
log.debug(processName + "_" + type.name() + ">" + line);
addProcessedImageToQueue(line);
}
}
}
is.close();
if (type.equals(Type.STD_OUT)) {
queueFinishedPage(currentPageNumber);
}
}
private void addProcessedImageToQueue(String line) {
/*
Ghostscript prints the pageNumber it is currently working on, so we remember the current page and queue it as soon as the next comes in.
*/
Matcher pageNumberMatcher = pageFinishedPattern.matcher(line);
if (pageNumberMatcher.find()) {
int pageNumber = Integer.parseInt(pageNumberMatcher.group(1));
if (currentPageNumber == 0) {
currentPageNumber = pageNumber;
return;
}
queueFinishedPage(currentPageNumber);
currentPageNumber = pageNumber;
}
}
private void queueFinishedPage(int pageNumber) {
var imageFile = this.pagesToProcess.get(pageNumber);
if (imageFile == null) {
throw new IllegalArgumentException(String.format("Page number %d does not exist in this thread. It only has pagenumbers %s", pageNumber, pagesToProcess.keySet()));
}
assert new File(imageFile.absoluteFilePath()).isFile();
renderedPageImageFileOutput.add(imageFile);
}
public enum Type {
ERROR,
STD_OUT
}
}
@@ -5,12 +5,11 @@ import java.util.List;
import java.util.concurrent.BlockingQueue;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.io.MemoryUsageSetting;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
import com.knecon.fforesight.service.ocr.processor.service.ImageStreamEngine;
import com.knecon.fforesight.service.ocr.processor.service.OcrProgressLogger;
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
@@ -26,6 +25,7 @@ import lombok.experimental.FieldDefaults;
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class ImageExtractionThread extends Thread {
static double FULL_PAGE_IMAGE_THRESHOLD = 0.99;
static double IMAGE_ALIGNMENT_THRESHOLD = 1;
int id;
@@ -37,9 +37,10 @@ public class ImageExtractionThread extends Thread {
OcrServiceSettings settings;
// output is written to these lists
BlockingQueue<OcrImage> imageOutputQueue;
BlockingQueue<UnprocessedImage> imageProcessingQueue;
List<Integer> stitchedPageNumbers;
@SneakyThrows
@Override
public void run() {
@@ -48,28 +49,28 @@ public class ImageExtractionThread extends Thread {
for (Integer pageIndex : pageIndices) {
try (PDDocument document = Loader.loadPDF(documentFile)) { // load new PDDocument for thread safety, also keeps RAM usage low.
timestamp = System.currentTimeMillis();
List<ExtractedOcrImage> extractedOcrImages = getExtractedOcrImages(pageIndex, document);
List<ExtractedImage> extractedImages = getExtractedImages(pageIndex, document);
stats.increaseImageExtraction(id, System.currentTimeMillis() - timestamp);
if (extractedOcrImages.isEmpty()) {
if (extractedImages.isEmpty()) {
logger.logPageSkipped(pageIndex);
}
if (checkForStitchedImages(extractedOcrImages)) {
if (checkForFullPageOrStitchedImages(extractedImages, document.getPage(pageIndex - 1))) {
stitchedPageNumbers.add(pageIndex);
logger.addImagesToProcess(pageIndex, 0);
continue;
}
for (ExtractedOcrImage image : extractedOcrImages) {
imageOutputQueue.put(image);
logger.addImagesToProcess(image.getPageNumber(), image.getNumberOnPage());
for (ExtractedImage image : extractedImages) {
imageProcessingQueue.put(image);
logger.addImagesToProcess(image.pageNumber(), image.numberOnPage());
}
}
}
}
private List<ExtractedOcrImage> getExtractedOcrImages(Integer pageIndex, PDDocument document) {
private List<ExtractedImage> getExtractedImages(Integer pageIndex, PDDocument document) {
PDPage page = document.getPage(pageIndex - 1);
ImageStreamEngine imageStreamEngine = new ImageStreamEngine(settings);
@@ -79,22 +80,25 @@ public class ImageExtractionThread extends Thread {
@SneakyThrows
private boolean checkForStitchedImages(List<ExtractedOcrImage> imagesOnCurrentPage) {
private boolean checkForFullPageOrStitchedImages(List<ExtractedImage> imagesOnCurrentPage, PDPage page) {
if (imagesOnCurrentPage.size() <= 1) {
if (imagesOnCurrentPage.isEmpty()) {
return false;
}
//checking for intersections or direct alignment of images
ExtractedOcrImage[] imageOnPagesArray = new ExtractedOcrImage[imagesOnCurrentPage.size()];
int index = 0;
for (ExtractedOcrImage imageOnPage : imagesOnCurrentPage) {
imageOnPagesArray[index] = imageOnPage;
index++;
for (ExtractedImage imageOnPage : imagesOnCurrentPage) {
if (imageOnPage.width() > FULL_PAGE_IMAGE_THRESHOLD * page.getCropBox().getWidth() && imageOnPage.height() > FULL_PAGE_IMAGE_THRESHOLD * page.getCropBox()
.getHeight()) {
return true;
}
}
for (int j = 0; j < imageOnPagesArray.length; j++) {
for (int i = j + 1; i < imageOnPagesArray.length; i++) {
if (imageOnPagesArray[j].getImageCoordinatesInInitialUserSpace().aligns(imageOnPagesArray[i].getImageCoordinatesInInitialUserSpace(), IMAGE_ALIGNMENT_THRESHOLD)) {
//checking for intersections or direct alignment of images
for (int j = 0; j < imagesOnCurrentPage.size(); j++) {
for (int i = j + 1; i < imagesOnCurrentPage.size(); i++) {
if (imagesOnCurrentPage.get(j)
.getImageCoordinatesInInitialUserSpace()
.aligns(imagesOnCurrentPage.get(i).getImageCoordinatesInInitialUserSpace(), IMAGE_ALIGNMENT_THRESHOLD)) {
// TODO: see if we can stitch aligning images using BufferedImage and skip the gs conversion entirely
return true;
}
@@ -0,0 +1,250 @@
package com.knecon.fforesight.service.ocr.processor.service.threads;
import static net.sourceforge.tess4j.ITessAPI.TRUE;
import java.nio.FloatBuffer;
import java.nio.IntBuffer;
import java.util.NoSuchElementException;
import java.util.concurrent.BlockingQueue;
import org.apache.pdfbox.pdmodel.PDDocument;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
import com.knecon.fforesight.service.ocr.processor.model.PageInformation;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageOcrImage;
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
import com.sun.jna.ptr.PointerByReference;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.Setter;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
import net.sourceforge.lept4j.L_Kernel;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
import net.sourceforge.tess4j.ITessAPI;
import net.sourceforge.tess4j.TessAPI1;
/*
* This thread does all the image processing. There should only be one, since Leptonica is not thread safe.
*/
@Slf4j
@RequiredArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class ImageProcessingThread extends Thread {
final BlockingQueue<UnprocessedImage> imageInputQueue;
final BlockingQueue<OcrImage> imageOutputQueue;
final ITessAPI.TessBaseAPI detectionScriptHandle = initDetectionScriptHandle();
final L_Kernel gaussianKernel = Leptonica1.makeGaussianKernel(2, 2, 1.2f, 1);
final Statistics stats;
final OcrServiceSettings settings;
final PDDocument document;
@Setter
boolean allImagesExtracted;
@SneakyThrows
@Override
public void run() {
try {
while (!allImagesExtracted) {
final UnprocessedImage image = imageInputQueue.take();
var ocrImage = this.process(image);
try {
imageOutputQueue.put(ocrImage);
} catch (InterruptedException e) {
imageOutputQueue.put(ocrImage);
}
}
} catch (InterruptedException e) {
log.info("All images extracted, emptying processing queue and stopping");
}
try {
while (true) {
final UnprocessedImage image = imageInputQueue.remove();
OcrImage ocrImage = this.process(image);
imageOutputQueue.put(ocrImage);
}
} catch (NoSuchElementException e) {
log.debug("No images left in processing queue, stopping.");
}
TessAPI1.TessBaseAPIEnd(this.detectionScriptHandle);
TessAPI1.TessBaseAPIDelete(this.detectionScriptHandle);
LeptUtils.dispose(gaussianKernel);
}
private OcrImage process(UnprocessedImage unprocessedImage) {
long timestamp = System.currentTimeMillis();
OcrImage ocrImage;
if (unprocessedImage instanceof ExtractedImage extractedImage) {
ocrImage = processExtractedImage(extractedImage);
} else if (unprocessedImage instanceof RenderedPageImageFile renderedPageImageFile) {
ocrImage = processRenderedPageImageFile(renderedPageImageFile);
} else {
throw new UnsupportedOperationException(String.format("Class %s is not supported!", unprocessedImage.getClass()));
}
stats.increaseImageProcessing(System.currentTimeMillis() - timestamp);
return ocrImage;
}
private OcrImage processRenderedPageImageFile(RenderedPageImageFile renderedPageImageFile) {
Pix pix = processPix(renderedPageImageFile.asPix(), settings.getDpi(), settings.getDpi());
int orientDegree = detectOrientation(pix, settings.getDpi(), detectionScriptHandle);
Pix rotatedPix = ImageProcessingUtils.deRotatePix(orientDegree, pix);
OcrImage ocrImage = new RenderedPageOcrImage(pix.h,
pix.w,
PageInformation.fromPDPage(renderedPageImageFile.pageNumber(), document.getPage(renderedPageImageFile.pageNumber() - 1)),
rotatedPix,
orientDegree);
if (pix != rotatedPix) {
LeptUtils.disposePix(pix);
}
return ocrImage;
}
private OcrImage processExtractedImage(ExtractedImage extractedImage) {
float imageDPI = Math.abs(extractedImage.image().getWidth() / (extractedImage.ctm().getScalingFactorX() / 72));
Pix pix = processPix(extractedImage.asPix(), imageDPI, settings.getDpi());
int orientDegree = detectOrientation(pix, settings.getDpi(), detectionScriptHandle);
Pix rotatedPix = ImageProcessingUtils.deRotatePix(orientDegree, pix);
OcrImage ocrImage = new ExtractedOcrImage(extractedImage.pageNumber(),
extractedImage.numberOnPage(),
extractedImage.height(),
extractedImage.width(),
extractedImage.ctm(),
rotatedPix,
pix.h,
pix.w,
orientDegree);
if (pix != rotatedPix) {
LeptUtils.disposePix(pix);
}
return ocrImage;
}
public int detectOrientation(Pix pix, int dpi, ITessAPI.TessBaseAPI detectionScriptHandle) {
TessAPI1.TessBaseAPISetImage2(detectionScriptHandle, pix);
TessAPI1.TessBaseAPISetSourceResolution(detectionScriptHandle, dpi);
IntBuffer orientationDegreeResultBuffer;
FloatBuffer orientationDegreeConfidenceBuffer;
PointerByReference scriptureNameBuffer;
FloatBuffer scriptureConfidenceBuffer;
orientationDegreeResultBuffer = IntBuffer.allocate(1);
orientationDegreeConfidenceBuffer = FloatBuffer.allocate(1);
scriptureNameBuffer = new PointerByReference(); // Is this memory being freed?
scriptureConfidenceBuffer = FloatBuffer.allocate(1);
int orientationDegree = 0;
int result = TessAPI1.TessBaseAPIDetectOrientationScript(detectionScriptHandle,
orientationDegreeResultBuffer,
orientationDegreeConfidenceBuffer,
scriptureNameBuffer,
scriptureConfidenceBuffer);
if (result == TRUE && orientationDegreeConfidenceBuffer.get() > settings.getMinRotationConfidence()) {
orientationDegree = orientationDegreeResultBuffer.get();
}
TessAPI1.TessBaseAPIClear(detectionScriptHandle);
return orientationDegree;
}
@SneakyThrows
private Pix processPix(Pix pix, float imageDpi, int targetDpi) {
Pix grayScale;
Pix scaledUp;
Pix gaussian;
Pix binarized;
//convert to grayscale
if (pix.d == 8) {
grayScale = pix;
} else if (pix.d == 32) {
grayScale = Leptonica1.pixConvertRGBToGrayFast(pix);
} else if (pix.d == 1) {
grayScale = Leptonica1.pixConvert1To8(null, pix, (byte) 0, (byte) 255);
} else {
throw new UnsupportedOperationException(String.format("Unknown pix format with bpp of %d", pix.d));
}
// scale up
float targetFactor = targetDpi / imageDpi;
if (targetFactor > 2.1) {
scaledUp = Leptonica1.pixScaleGray4xLI(grayScale);
} else if (targetFactor > 1.1) {
scaledUp = Leptonica1.pixScaleGray2xLI(grayScale);
} else {
scaledUp = grayScale;
}
// remove noise and prep for Otsu
gaussian = Leptonica1.pixConvolve(scaledUp, gaussianKernel, 8, 1);
// Threshold to binary
if (pix.w < 100 || pix.h < 100) {
binarized = Leptonica1.pixThresholdToBinary(gaussian, 170);
} else {
binarized = Leptonica1.pixOtsuThreshOnBackgroundNorm(gaussian, null, 50, 50, 165, 10, 100, 5, 5, 0.2f, null);
if (binarized == null) { // Sometimes Otsu just fails, then we binarize directly
binarized = Leptonica1.pixThresholdToBinary(gaussian, 170);
}
}
LeptUtils.disposePix(pix);
LeptUtils.disposePix(grayScale);
LeptUtils.disposePix(scaledUp);
LeptUtils.disposePix(gaussian);
return binarized;
}
private static ITessAPI.TessBaseAPI initDetectionScriptHandle() {
ITessAPI.TessBaseAPI handle = TessAPI1.TessBaseAPICreate();
String datapath = System.getenv("TESSDATA_PREFIX");
TessAPI1.TessBaseAPIInit3(handle, datapath, "osd");
TessAPI1.TessBaseAPISetVariable(handle, "debug_file", "/dev/null");
return handle;
}
}
@@ -1,6 +1,10 @@
package com.knecon.fforesight.service.ocr.processor.service.threads;
import static net.sourceforge.tess4j.ITessAPI.TRUE;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPICreate;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIInit1;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPISetPageSegMode;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPISetVariable;
import java.io.File;
import java.nio.FloatBuffer;
@@ -16,6 +20,7 @@ import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
import com.knecon.fforesight.service.ocr.processor.service.OcrProgressLogger;
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
import com.knecon.fforesight.service.ocr.processor.utils.Tesseract2;
import com.sun.jna.StringArray;
import com.sun.jna.ptr.PointerByReference;
import lombok.AccessLevel;
@@ -43,7 +48,6 @@ public class OCRThread extends Thread {
Statistics stats;
OcrServiceSettings settings;
Tesseract2 instance;
ITessAPI.TessBaseAPI detectionScriptHandle;
public OCRThread(int id,
@@ -62,7 +66,6 @@ public class OCRThread extends Thread {
this.stats = stats;
this.settings = settings;
this.instance = createInstance(settings);
this.detectionScriptHandle = initDetectionScriptHandle();
}
@@ -87,10 +90,9 @@ public class OCRThread extends Thread {
this.process(image);
}
} catch (NoSuchElementException e) {
log.debug("Processed all Images, finishing.");
log.debug("Executed tesseract on all Images, finishing.");
}
TessAPI1.TessBaseAPIDelete(this.detectionScriptHandle);
}
@@ -102,15 +104,8 @@ public class OCRThread extends Thread {
int psm = settings.getPsmOverride() < 0 ? image.getOptimalPageSegmentationMode() : settings.getPsmOverride();
int orientDegree = detectOrientation(image);
image.setRotationDegrees(orientDegree);
Pix rotatedPix = image.getRotatedPix();
executeTesseract(psm, image.getDpi(), rotatedPix, tesseractOutputFileName);
synchronized (OCRThread.class) {
image.destroyPix();
LeptUtils.disposePix(rotatedPix);
}
executeTesseract(psm, image.getDpi(), image.getPix(), tesseractOutputFileName);
image.destroyPix();
results.add(OcrResult.create(image, tesseractOutputFileName));
logger.logImageFinished(image, psm);
@@ -118,64 +113,14 @@ public class OCRThread extends Thread {
}
public int detectOrientation(OcrImage image) {
IntBuffer orientationDegreeResultBuffer;
FloatBuffer orientationDegreeConfidenceBuffer;
PointerByReference scriptureNameBuffer;
FloatBuffer scriptureConfidenceBuffer;
TessAPI1.TessBaseAPISetImage2(detectionScriptHandle, image.getPix());
TessAPI1.TessBaseAPISetSourceResolution(detectionScriptHandle, image.getDpi());
synchronized (OCRThread.class) { // must synchronize the mallocs here with the mallocs in leptonica binarization.
orientationDegreeResultBuffer = IntBuffer.allocate(1);
orientationDegreeConfidenceBuffer = FloatBuffer.allocate(1);
scriptureNameBuffer = new PointerByReference();
scriptureConfidenceBuffer = FloatBuffer.allocate(1);
}
int orient_deg = 0;
int result = TessAPI1.TessBaseAPIDetectOrientationScript(detectionScriptHandle,
orientationDegreeResultBuffer,
orientationDegreeConfidenceBuffer,
scriptureNameBuffer,
scriptureConfidenceBuffer);
if (result == TRUE) {
orient_deg = orientationDegreeResultBuffer.get();
}
synchronized (OCRThread.class) {
TessAPI1.TessBaseAPIClear(detectionScriptHandle);
}
return orient_deg;
}
synchronized private static ITessAPI.TessBaseAPI initDetectionScriptHandle() {
ITessAPI.TessBaseAPI handle = TessAPI1.TessBaseAPICreate();
String datapath = System.getenv("TESSDATA_PREFIX");
TessAPI1.TessBaseAPIInit3(handle, datapath, "osd");
return handle;
}
@SneakyThrows
public void executeTesseract(int psm, int dpi, Pix pix, String tesseractOutputFileName) {
if (settings.isDebug()) {
String[] a = tesseractOutputFileName.split("/");
String folder = "/tmp/pixs/" + a[a.length - 3];
new File(folder).mkdirs();
Leptonica1.pixWrite(folder + "/pix_" + a[a.length - 1] + ".png", pix, 3);
}
Leptonica1.pixWrite(tesseractOutputFileName + ".tiff", pix, 5); // write the used image for later bold detection
instance.setVariable("user_defined_dpi", String.valueOf(dpi));
instance.setPageSegMode(psm);
instance.createDocumentsWithResults(pix, null, tesseractOutputFileName, List.of(ITesseract.RenderedFormat.HOCR), ITessAPI.TessPageIteratorLevel.RIL_BLOCK);
}
@@ -1,55 +0,0 @@
package com.knecon.fforesight.service.ocr.processor.service.threads;
import java.io.BufferedReader;
import java.io.InputStream;
import java.io.InputStreamReader;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class ProcessIOLogger extends Thread {
// If the stdError or stdOut buffer of a thread is not being emptied it might lock the process in case of errors, so we need to empty both streams to prevent a deadlock.
// Since both need to read simultaneously we need to implement the readers as separate threads.
InputStream is;
String processName;
Type type;
@SneakyThrows
public void run() {
try (InputStreamReader isr = new InputStreamReader(is); BufferedReader br = new BufferedReader(isr)) {
String line;
while (true) {
line = br.readLine();
if (line == null) {
break;
}
if (type.equals(Type.ERROR)) {
log.error(processName + "_" + type.name() + ">" + line);
} else {
log.debug(processName + "_" + type.name() + ">" + line);
}
}
}
is.close();
}
public enum Type {
ERROR,
STD_OUT
}
}
@@ -14,14 +14,17 @@ public class OcrServiceSettings {
int ocrThreadCount = 4; // Number of OCR threads
int imageExtractThreadCount = 2; // Number of image extraction threads
int gsProcessCount = 2; // Number of Ghostscript processes
int gsProcessCount = 1; // Number of Ghostscript processes
int dpi = 300; // Target DPI for binarized images
int psmOverride = -1; // Overrides the page segmentation mode if > 0
int minImageHeight = 20; // Minimum height for images to be processed
int minImageWidth = 20; // Minimum width for images to be processed
float minRotationConfidence = 2; // Sets a lower bound for the confidence rating for rotated pages.
boolean debug; // If true, overlays OCR images with a grid and draws word bounding boxes
boolean removeWatermark; // If true, watermarks will be removed
String languages = "deu+eng"; // Defines languages loaded into Tesseract as 3-char codes, additional languages must also be installed in the docker environment
COSName ocrMarkedContentTag = COSName.getPDFName("KNECON_OCR");
boolean boldDetection = true; // if true, bold detection will be attempted
double boldThreshold = 0.5; // Words are opened with a brick of average stroke width, if the ratio of remaining pixels is higher the word is determined bold.
}
@@ -2,13 +2,21 @@ package com.knecon.fforesight.service.ocr.processor.utils;
import java.awt.AlphaComposite;
import java.awt.Color;
import java.awt.Graphics;
import java.awt.Graphics2D;
import java.awt.Transparency;
import java.awt.image.BufferedImage;
import java.io.IOException;
import java.nio.IntBuffer;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
import com.sun.jna.ptr.PointerByReference;
import lombok.SneakyThrows;
import lombok.experimental.UtilityClass;
import net.sourceforge.lept4j.L_Kernel;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.lept4j.util.LeptUtils;
@@ -16,61 +24,30 @@ import net.sourceforge.lept4j.util.LeptUtils;
@UtilityClass
public class ImageProcessingUtils {
public static Pix despecklePix(Pix pix) {
public BufferedImage convertToDeviceColorSpace(ExtractedImage extractedImage) {
assert pix.d == 8;
Pix despeckled;
if (pix.w < 100 || pix.h < 100) {
// too small to properly despeckle, just binarize instead.
despeckled = Leptonica1.pixThresholdToBinary(pix, 180);
BufferedImage image;
if (extractedImage.colorSpace() instanceof PDDeviceRGB || extractedImage.colorSpace() instanceof PDDeviceGray) {
image = extractedImage.image();
} else {
despeckled = LeptUtils.despeckle(pix, LeptUtils.SEL_STR3, 3); // sometimes this fails and I can't figure out why. Then we skip the despeckling and just simply convert to binary. Might have something to do with Imagesize, not sure though...
if (despeckled == null) {
despeckled = Leptonica1.pixThresholdToBinary(pix, 180);
}
BufferedImage pdfImage = extractedImage.image();
image = new BufferedImage(pdfImage.getWidth(), pdfImage.getHeight(), BufferedImage.TYPE_BYTE_GRAY);
Graphics g = image.getGraphics();
g.drawImage(pdfImage, 0, 0, null);
g.dispose();
}
if (pix != despeckled) {
LeptUtils.disposePix(pix);
}
return despeckled;
return image;
}
public static Pix scaleToTargetDpi(float imageDpi, int targetDpi, Pix grayScale) {
public Pix deRotatePix(int orientDegree, Pix pix) {
float targetFactor = targetDpi / imageDpi;
if (targetFactor > 3) {
Pix scaledUp;
scaledUp = Leptonica1.pixScaleGray4xLI(grayScale);
LeptUtils.disposePix(grayScale);
return scaledUp;
} else if (targetFactor > 1.9) {
Pix scaledUp;
scaledUp = Leptonica1.pixScaleGray2xLI(grayScale);
LeptUtils.disposePix(grayScale);
return scaledUp;
} else {
return grayScale;
}
}
@SneakyThrows
public static Pix convertToGrayScale(BufferedImage image) {
Pix pix = LeptUtils.convertImageToPix(image);
if (pix.d == 8) {
return pix;
} else if (pix.d == 32) {
Pix grayScale = Leptonica1.pixConvertRGBToGrayFast(pix);
LeptUtils.disposePix(pix);
return grayScale;
} else {
Pix grayScale = Leptonica1.pixConvert1To8(null, pix, (byte) 0, (byte) 255);
LeptUtils.disposePix(pix);
return grayScale;
}
return switch (360 - orientDegree) {
case 90 -> Leptonica1.pixRotateOrth(pix, 1);
case 180 -> Leptonica1.pixRotateOrth(pix, 2);
case 270 -> Leptonica1.pixRotateOrth(pix, 3);
default -> pix;
};
}
@@ -93,4 +70,16 @@ public class ImageProcessingUtils {
}
}
public static double calculatePixelDensity(Pix pix) {
IntBuffer pixelCount = IntBuffer.allocate(1);
int result = Leptonica1.pixCountPixels(pix, pixelCount, null);
if (result == 0) {
return (double) pixelCount.get() / (pix.h * pix.w);
} else {
return -1;
}
}
}
@@ -0,0 +1,73 @@
package com.knecon.fforesight.service.ocr.processor.utils;
import lombok.experimental.UtilityClass;
import net.sourceforge.lept4j.L_Kernel;
import net.sourceforge.lept4j.Leptonica1;
@UtilityClass
public class KernelUtils {
/*
-1, -1, -1
-1, 8, -1
-1, -1, -1
*/
public L_Kernel createFullLaplacianKernel() {
L_Kernel laplacianKernel = Leptonica1.kernelCreate(3, 3);
Leptonica1.kernelSetElement(laplacianKernel, 0, 0, -1);
Leptonica1.kernelSetElement(laplacianKernel, 0, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 0, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 0, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 0, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, 8);
return laplacianKernel;
}
/*
0, 0, -1, 0, 0
0, -1, -1, -1, 0
-1, -1, 12, -1, -1
0, -1, -1, -1, 0
0, 0, -1, 0, 0
*/
public L_Kernel createLaplacianKernel5x5() {
L_Kernel laplacianKernel = Leptonica1.kernelCreate(5, 5);
Leptonica1.kernelSetElement(laplacianKernel, 0, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 3, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 0, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 3, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 4, -1);
Leptonica1.kernelSetElement(laplacianKernel, 3, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 3, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 3, 3, -1);
Leptonica1.kernelSetElement(laplacianKernel, 4, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 2, 12);
return laplacianKernel;
}
/*
0, -1, 0
-1, 4, -1
0, -1, 0
*/
public L_Kernel createLaplacianKernel() {
L_Kernel laplacianKernel = Leptonica1.kernelCreate(3, 3);
Leptonica1.kernelSetElement(laplacianKernel, 0, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 0, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, 4);
return laplacianKernel;
}
}
@@ -1,5 +1,25 @@
package com.knecon.fforesight.service.ocr.processor.utils;
import static net.sourceforge.tess4j.ITessAPI.TRUE;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIDelete;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIEnd;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIGetIterator;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIGetStringVariable;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIMeanTextConf;
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIProcessPage;
import static net.sourceforge.tess4j.TessAPI1.TessDeleteResultRenderer;
import static net.sourceforge.tess4j.TessAPI1.TessHOcrRendererCreate;
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorBegin;
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorBoundingBox;
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorNext;
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorConfidence;
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorDelete;
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorGetPageIterator;
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorGetUTF8Text;
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererBeginDocument;
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererEndDocument;
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererInsert;
import java.awt.Rectangle;
import java.nio.IntBuffer;
import java.util.ArrayList;
@@ -9,20 +29,19 @@ import com.sun.jna.Pointer;
import lombok.extern.slf4j.Slf4j;
import net.sourceforge.lept4j.Pix;
import net.sourceforge.tess4j.ITessAPI;
import net.sourceforge.tess4j.OCRResult;
import net.sourceforge.tess4j.TessAPI1;
import net.sourceforge.tess4j.Tesseract1;
import net.sourceforge.tess4j.Tesseract;
import net.sourceforge.tess4j.TesseractException;
import net.sourceforge.tess4j.Word;
@Slf4j
/**
* Overriden version only so I can use Tesseract1 with Pixs instead of BufferedImages. All Functions are copied and then the BufferedImage -> Pix conversion deleted.
*/
public class Tesseract2 extends Tesseract1 {
*/ public class Tesseract2 extends Tesseract {
private int createDocuments(Pix pix, String filename, TessResultRenderer renderer) {
private int createDocuments(Pix pix, String filename, ITessAPI.TessResultRenderer renderer) {
String title = TessBaseAPIGetStringVariable(getHandle(), DOCUMENT_TITLE);
TessResultRendererBeginDocument(renderer, title);
@@ -62,7 +81,7 @@ public class Tesseract2 extends Tesseract1 {
try {
for (int i = 0; i < pixs.length; i++) {
try {
TessResultRenderer renderer = createRenderers(outputbases[i], formats);
ITessAPI.TessResultRenderer renderer = createRenderers(outputbases[i], formats);
int meanTextConfidence = createDocuments(pixs[i], filenames[i], renderer);
TessDeleteResultRenderer(renderer);
List<Word> words = meanTextConfidence > 0 ? getRecognizedWords(pageIteratorLevel) : new ArrayList<Word>();
@@ -85,8 +104,8 @@ public class Tesseract2 extends Tesseract1 {
List<Word> words = new ArrayList<>();
try {
TessResultIterator ri = TessBaseAPIGetIterator(getHandle());
TessPageIterator pi = TessResultIteratorGetPageIterator(ri);
ITessAPI.TessResultIterator ri = TessBaseAPIGetIterator(getHandle());
ITessAPI.TessPageIterator pi = TessResultIteratorGetPageIterator(ri);
TessPageIteratorBegin(pi);
do {
@@ -119,9 +138,9 @@ public class Tesseract2 extends Tesseract1 {
}
private TessResultRenderer createRenderers(String outputbase, List<RenderedFormat> formats) {
private ITessAPI.TessResultRenderer createRenderers(String outputbase, List<RenderedFormat> formats) {
TessResultRenderer renderer = null;
ITessAPI.TessResultRenderer renderer = null;
for (RenderedFormat format : formats) {
switch (format) {
@@ -138,4 +157,12 @@ public class Tesseract2 extends Tesseract1 {
return renderer;
}
@Override
protected void dispose() {
TessBaseAPIEnd(getHandle());
TessBaseAPIDelete(getHandle());
}
}
@@ -20,7 +20,7 @@ class Type0FontMetricsFactoryTest {
public void testStringWidth() {
try (PDDocument document = Loader.loadPDF(new File(Type0FontMetricsFactoryTest.class.getClassLoader().getResource("InvisibleText.pdf").getPath()))) {
Type0FontMetricsFactory metricsFactory = new Type0FontMetricsFactory(document);
Type0FontMetricsFactory metricsFactory = Type0FontMetricsFactory.regular(document);
FontMetrics fontMetrics = metricsFactory.calculateMetrics("deine mutter", 100, 50);
}
@@ -0,0 +1,36 @@
package com.knecon.fforesight.service.ocr.processor.utils;
import static net.sourceforge.lept4j.ILeptonica.IFF_PNG;
import org.junit.jupiter.api.BeforeEach;
import org.junit.jupiter.api.Disabled;
import org.junit.jupiter.api.Test;
import net.sourceforge.lept4j.Leptonica1;
import net.sourceforge.lept4j.Pix;
@Disabled
class ImageProcessingUtilsTest {
@BeforeEach
public void loadLeptonica() {
System.setProperty("jna.library.path", System.getenv("VCPKG_DYNAMIC_LIB"));
}
@Test
public void testRotation() {
Pix pix = Leptonica1.pixRead("/home/kschuettler/Downloads/painHarold.webp");
Pix pix2 = ImageProcessingUtils.deRotatePix(0, pix);
Leptonica1.pixWrite("/tmp/0.png", pix2, IFF_PNG);
Pix pix3 = ImageProcessingUtils.deRotatePix(90, pix);
Leptonica1.pixWrite("/tmp/90.png", pix3, IFF_PNG);
Pix pix4 = ImageProcessingUtils.deRotatePix(180, pix);
Leptonica1.pixWrite("/tmp/180.png", pix4, IFF_PNG);
Pix pix5 = ImageProcessingUtils.deRotatePix(270, pix);
Leptonica1.pixWrite("/tmp/270.png", pix5, IFF_PNG);
}
}
@@ -1,10 +1,7 @@
package com.knecon.fforesight.service.ocr.processor.utils;
import java.awt.image.BufferedImage;
import java.io.BufferedReader;
import java.io.File;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.util.LinkedList;
import java.util.List;
import java.util.stream.IntStream;
@@ -19,7 +16,7 @@ import org.springframework.core.io.ClassPathResource;
import org.springframework.util.FileSystemUtils;
import com.knecon.fforesight.service.ocr.processor.service.OsUtils;
import com.knecon.fforesight.service.ocr.processor.service.threads.ProcessIOLogger;
import com.knecon.fforesight.service.ocr.processor.service.threads.GhostScriptOutputHandler;
import lombok.SneakyThrows;
@@ -50,29 +47,6 @@ public class Pdf2ImgTest {
}
@Test
@SneakyThrows
public void testGhostScript() {
String outputDir = "/tmp/ghostscript_out/";
new File(outputDir).mkdirs();
ClassPathResource resource = new ClassPathResource("files/Cyberport__SD-Faktura-Kopie_(ZRG2)_-_31.08.2020.pdf");
String[] cmdArgs = new String[]{"gs", "-dNOPAUSE", "-sDEVICE=tiff24nc", "-r" + DPI, "-sOutputFile=" + outputDir + "page%04d", resource.getFile().toString(), "-c", "quit"};
Process p = Runtime.getRuntime().exec(cmdArgs);
ProcessIOLogger logger = new ProcessIOLogger(p.getInputStream(), "GS", ProcessIOLogger.Type.STD_OUT);
logger.start();
ProcessIOLogger errorLogger = new ProcessIOLogger(p.getErrorStream(), "GS", ProcessIOLogger.Type.STD_OUT);
errorLogger.start();
int exitcode = p.waitFor();
logger.join();
errorLogger.join();
System.out.println("Ghostscript finished with exit code " + exitcode);
FileSystemUtils.deleteRecursively(new File(outputDir));
}
@Test
@SneakyThrows
public void testGhostScriptParallel() {
@@ -3,7 +3,7 @@ import org.springframework.boot.gradle.tasks.bundling.BootBuildImage
plugins {
application
id("com.iqser.red.service.java-conventions")
id("org.springframework.boot") version "3.1.3"
id("org.springframework.boot") version "3.1.5"
id("io.spring.dependency-management") version "1.1.3"
id("org.sonarqube") version "4.3.0.3225"
id("io.freefair.lombok") version "8.2.2"
@@ -11,19 +11,26 @@ plugins {
configurations {
all {
exclude(group = "org.springframework.boot", module = "spring-boot-starter-logging")
exclude(group = "commons-logging", module = "commons-logging")
exclude(group = "org.springframework.boot", module = "spring-boot-starter-log4j2")
exclude(group = "com.iqser.red.commons", module = "logging-commons")
}
}
val springBootStarterVersion = "3.1.5"
dependencies {
implementation(project(":ocr-service-processor"))
implementation(project(":ocr-service-api"))
implementation("com.knecon.fforesight:tracing-commons:0.3.0")
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:4.0.4")
implementation("org.springframework.boot:spring-boot-starter-amqp:3.1.4")
implementation("org.springframework.boot:spring-boot-starter-amqp:${springBootStarterVersion}")
testImplementation("org.springframework.boot:spring-boot-starter-test:3.1.4")
implementation("net.logstash.logback:logstash-logback-encoder:7.4")
implementation("ch.qos.logback:logback-classic")
testImplementation("org.springframework.boot:spring-boot-starter-test:${springBootStarterVersion}")
testImplementation("com.iqser.red.commons:test-commons:2.1.0")
testImplementation("org.springframework.amqp:spring-rabbit-test:3.0.2")
}
@@ -5,10 +5,16 @@ persistence-service.url: "http://persistence-service-v1:8080"
tenant-user-management-service.url: "http://tenant-user-management-service:8080/internal"
fforesight.tenants.remote: true
logging.type: ${LOGGING_TYPE:CONSOLE}
kubernetes.namespace: ${NAMESPACE:default}
project.version: 1.0-SNAPSHOT
server:
port: 8080
spring:
application:
name: ocr-service
main:
allow-circular-references: true # FIXME
profiles:
@@ -33,6 +39,7 @@ fforesight:
ignored-endpoints: [ '/actuator/health', '/actuator/health/**' ]
enabled: true
logging.pattern.level: "%5p [${spring.application.name},%X{traceId:-},%X{spanId:-}]"
management:
endpoint:
@@ -41,9 +48,12 @@ management:
health.enabled: true
endpoints.web.exposure.include: prometheus, health, metrics
metrics.export.prometheus.enabled: ${monitoring.enabled:false}
storage:
backend: 's3'
tracing:
enabled: ${TRACING_ENABLED:false}
sampling:
probability: ${TRACING_PROBABILITY:1.0}
otlp:
tracing:
endpoint: ${OTLP_ENDPOINT:http://otel-collector-opentelemetry-collector.otel-collector:4318/v1/traces}
pdftron.license: ${PDFTRON_LICENSE}
@@ -0,0 +1,17 @@
<configuration>
<springProperty scope="configuration" name="logType" source="logging.type"/>
<springProperty scope="context" name="application.name" source="spring.application.name"/>
<springProperty scope="context" name="version" source="project.version"/>
<include resource="org/springframework/boot/logging/logback/defaults.xml"/>
<include resource="org/springframework/boot/logging/logback/console-appender.xml"/>
<appender name="JSON" class="ch.qos.logback.core.ConsoleAppender">
<encoder class="net.logstash.logback.encoder.LogstashEncoder"/>
</appender>
<root level="INFO">
<appender-ref ref="${logType}"/>
</root>
</configuration>
@@ -64,7 +64,7 @@ public class OcrServiceIntegrationTest extends AbstractTest {
@SneakyThrows
public void testOcr() {
String text = testOCR("files/2009-1048395_50pages_tables.pdf");
String text = testOCR("files/402Study.pdf");
}
@@ -162,13 +162,7 @@ public class OcrServiceIntegrationTest extends AbstractTest {
@SneakyThrows
public void testOcrForSpecificFile() {
testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/syn-dm-testfiles/F.2. A16003E - Acute Inhalation Study.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/syn-dm-testfiles/A23220A - 404 - Skin Irritation in vivo.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/OcrMassTest/G.1.2 - 1768300_MMNA_A13617AV_report.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/OcrMassTest/SOLICITA_VICTRATO-GOLD-II_Item 17_Toxicidade Inalatoria Aguda.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/OcrMassTest/SOLICITA_VICTRATO-GOLD-II_Item 20_Sensibilizacao_02.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/OcrMassTest/ITEM 23_A15149W - Dermal absorption of formulated product.pdf"));
// testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/OcrMassTest/SOLICITA_VICTRATO-GOLD-II_Item 16_Toxicidade Cutanea Aguda.pdf"));
testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/syn-dm-testfiles2/A16361B - Acute Dermal Toxicity Study in Rats.pdf"));
}