Compare commits
41
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8924e905ad | ||
|
|
fb1fe35bc1 | ||
|
|
912f00aa84 | ||
|
|
bab16ad9b2 | ||
|
|
be4656189b | ||
|
|
8944b57344 | ||
|
|
67540950b8 | ||
|
|
6f29270e66 | ||
|
|
b961c9e324 | ||
|
|
4b6411161e | ||
|
|
14982eae7c | ||
|
|
99fc16130b | ||
|
|
80d38fb785 | ||
|
|
c06974ce69 | ||
|
|
591c7d7fab | ||
|
|
0300a087d4 | ||
|
|
98752ff1d1 | ||
|
|
ae09a59a7c | ||
|
|
65d818200f | ||
|
|
6fe95c6940 | ||
|
|
202132e14c | ||
|
|
0264e28cc2 | ||
|
|
a50f54676e | ||
|
|
1926707ae1 | ||
|
|
d3190844a3 | ||
|
|
c7ccbae6ff | ||
|
|
880bebcafc | ||
|
|
955ff6281d | ||
|
|
efd3a1d952 | ||
|
|
bb5b4a2fd8 | ||
|
|
6f99664906 | ||
|
|
574f7ac25e | ||
|
|
12217f2459 | ||
|
|
19747cbca5 | ||
|
|
2632d2023d | ||
|
|
4c225c2219 | ||
|
|
3d09f46844 | ||
|
|
77355b5367 | ||
|
|
57e194fcd0 | ||
|
|
c556687499 | ||
|
|
759bae6499 |
+1
-1
@@ -10,7 +10,7 @@ deploy:
|
||||
script:
|
||||
- echo "Building with gradle version ${BUILDVERSION}"
|
||||
- gradle -Pversion=${BUILDVERSION} publish
|
||||
- gradle bootBuildImage --cleanCache --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
|
||||
- gradle bootBuildImage --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
|
||||
- echo "BUILDVERSION=$BUILDVERSION" >> version.env
|
||||
artifacts:
|
||||
reports:
|
||||
|
||||
@@ -15,11 +15,17 @@ The service uses PDFTron to attempt the removal of invisible elements and waterm
|
||||
Extracts all images from the PDF using PDFBox
|
||||
3. Striped Image Detection and Stitching
|
||||
Detects if images are striped and stitches them together using Ghostscript.
|
||||
4. Binarization
|
||||
Binarizes the resulting images using Leptonica and the Otsu thresholding algorithm.
|
||||
4. Image Processing
|
||||
- Convert to grayscale
|
||||
- Upscale to target DPI
|
||||
- Filter using Gauss kernel
|
||||
- Binarizes the resulting images using Leptonica and the Otsu thresholding algorithm.
|
||||
- Despeckle using various morphological operations
|
||||
5. OCR Processing
|
||||
Runs Tesseract on the images to extract text.
|
||||
6. Text Integration
|
||||
6. Font style detection
|
||||
Detection of bold text using stroke width estimation
|
||||
7. Text Integration
|
||||
Draws the resulting text onto the original PDF using PDFBox.
|
||||
|
||||
Steps 2.-5. happen in parallel and communicate via a blocking queue to limit RAM usage.
|
||||
|
||||
@@ -25,6 +25,8 @@ tasks.named<Test>("test") {
|
||||
reports {
|
||||
junitXml.outputLocation.set(layout.buildDirectory.dir("reports/junit"))
|
||||
}
|
||||
minHeapSize = "512m"
|
||||
maxHeapSize = "8192m"
|
||||
}
|
||||
|
||||
tasks.test {
|
||||
|
||||
@@ -14,13 +14,14 @@ dependencies {
|
||||
api("net.sourceforge.tess4j:tess4j:5.8.0")
|
||||
api("com.iqser.red.commons:metric-commons:2.1.0")
|
||||
api("com.iqser.red.commons:storage-commons:2.45.0")
|
||||
api("com.knecon.fforesight:tenant-commons:0.13.0")
|
||||
api("com.knecon.fforesight:tenant-commons:0.19.0")
|
||||
api("com.pdftron:PDFNet:10.5.0")
|
||||
api("org.apache.pdfbox:pdfbox:3.0.0")
|
||||
api("org.apache.pdfbox:jbig2-imageio:3.0.4")
|
||||
api("com.github.jai-imageio:jai-imageio-core:1.4.0")
|
||||
api("com.github.jai-imageio:jai-imageio-jpeg2000:1.4.0")
|
||||
api("io.github.karols:hocr4j:0.1.2")
|
||||
api("org.apache.commons:commons-math3:3.6.1")
|
||||
api("io.github.karols:hocr4j:0.2.0")
|
||||
api("com.amazonaws:aws-java-sdk-kms:1.12.440")
|
||||
api("com.google.guava:guava:31.1-jre")
|
||||
api("com.iqser.red.commons:pdftron-logic-commons:2.20.0")
|
||||
|
||||
+36
@@ -0,0 +1,36 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.awt.image.BufferedImage;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDColorSpace;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
public record ExtractedImage(
|
||||
int pageNumber, QuadPoint position, int height, int width, BufferedImage image, Matrix ctm, int numberOnPage, PDColorSpace colorSpace) implements UnprocessedImage {
|
||||
|
||||
@SneakyThrows
|
||||
public Pix asPix() {
|
||||
|
||||
BufferedImage image = ImageProcessingUtils.convertToDeviceColorSpace(this);
|
||||
ImageProcessingUtils.setAlphaChannelToWhite(image);
|
||||
return LeptUtils.convertImageToPix(image);
|
||||
}
|
||||
|
||||
|
||||
public QuadPoint getImageCoordinatesInInitialUserSpace() {
|
||||
|
||||
return QuadPoint.fromRectangle2D(new Rectangle2D.Double(0, 0, 1, 1)).getTransformed(ctm.createAffineTransform());
|
||||
}
|
||||
|
||||
}
|
||||
+15
-104
@@ -1,16 +1,16 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import java.awt.AlphaComposite;
|
||||
import java.awt.Color;
|
||||
import java.awt.Graphics2D;
|
||||
import java.awt.Transparency;
|
||||
import java.awt.Graphics;
|
||||
import java.awt.geom.AffineTransform;
|
||||
import java.awt.image.BufferedImage;
|
||||
import java.io.IOException;
|
||||
import java.nio.IntBuffer;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
@@ -18,107 +18,25 @@ import lombok.Setter;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
import net.sourceforge.tess4j.ITessAPI;
|
||||
|
||||
@Slf4j
|
||||
@Getter
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class ExtractedOcrImage implements OcrImage {
|
||||
|
||||
final int pageNumber;
|
||||
final Pix pix;
|
||||
final int originalHeight;
|
||||
final int originalWidth;
|
||||
final int height;
|
||||
final int width;
|
||||
final Matrix ctm;
|
||||
final int numberOnPage;
|
||||
|
||||
@Setter
|
||||
int pageNumber;
|
||||
int numberOnPage;
|
||||
int originalHeight;
|
||||
int originalWidth;
|
||||
Matrix ctm;
|
||||
Pix pix;
|
||||
int height;
|
||||
int width;
|
||||
int rotationDegrees;
|
||||
|
||||
@SneakyThrows
|
||||
public ExtractedOcrImage(int pageNumber, int numberOnPage, BufferedImage bufferedImage, Matrix ctm, int targetDpi, boolean isGray) {
|
||||
|
||||
this.pageNumber = pageNumber;
|
||||
this.numberOnPage = numberOnPage;
|
||||
this.ctm = ctm;
|
||||
this.originalHeight = bufferedImage.getHeight();
|
||||
this.originalWidth = bufferedImage.getWidth();
|
||||
float imageDPI = Math.abs(bufferedImage.getWidth() / (ctm.getScalingFactorX() / 72));
|
||||
this.pix = binarize(bufferedImage, imageDPI, targetDpi, isGray);
|
||||
this.height = pix.h;
|
||||
this.width = pix.w;
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private Pix binarize(BufferedImage image, float imageDpi, int targetDpi, boolean isGray) {
|
||||
|
||||
setAlphaChannelToWhite(image);
|
||||
Pix grayScale = convertToGrayScale(image, isGray);
|
||||
Pix scaledUp = scaleToTargetDpi(imageDpi, targetDpi, grayScale);
|
||||
Pix despeckled = LeptUtils.despeckle(scaledUp, LeptUtils.SEL_STR3, 3);
|
||||
LeptUtils.disposePix(scaledUp);
|
||||
return despeckled;
|
||||
}
|
||||
|
||||
|
||||
private static Pix scaleToTargetDpi(float imageDpi, int targetDpi, Pix grayScale) {
|
||||
|
||||
Pix scaledUp;
|
||||
float targetFactor = targetDpi / imageDpi;
|
||||
|
||||
if (targetFactor > 3) {
|
||||
scaledUp = Leptonica1.pixScaleGray4xLI(grayScale);
|
||||
LeptUtils.disposePix(grayScale);
|
||||
} else if (targetFactor > 1.9) {
|
||||
scaledUp = Leptonica1.pixScaleGray2xLI(grayScale);
|
||||
LeptUtils.disposePix(grayScale);
|
||||
} else {
|
||||
scaledUp = grayScale;
|
||||
}
|
||||
return scaledUp;
|
||||
}
|
||||
|
||||
|
||||
private static Pix convertToGrayScale(BufferedImage image, boolean isGray) throws IOException {
|
||||
|
||||
Pix pix = LeptUtils.convertImageToPix(image);
|
||||
Pix grayScale;
|
||||
if (isGray) {
|
||||
grayScale = pix;
|
||||
} else {
|
||||
grayScale = Leptonica1.pixConvertRGBToGrayFast(pix);
|
||||
LeptUtils.disposePix(pix);
|
||||
}
|
||||
return grayScale;
|
||||
}
|
||||
|
||||
|
||||
private static void setAlphaChannelToWhite(BufferedImage image) {
|
||||
|
||||
if (image.getTransparency() == Transparency.TRANSLUCENT) {
|
||||
// NOTE: For BITMASK images, the color model is likely IndexColorModel,
|
||||
// and this model will contain the "real" color of the transparent parts
|
||||
// which is likely a better fit than unconditionally setting it to white.
|
||||
|
||||
// Fill background with white
|
||||
Graphics2D graphics = image.createGraphics();
|
||||
try {
|
||||
graphics.setComposite(AlphaComposite.DstOver); // Set composite rules to paint "behind"
|
||||
graphics.setPaint(Color.WHITE);
|
||||
graphics.fillRect(0, 0, image.getWidth(), image.getHeight());
|
||||
} finally {
|
||||
graphics.dispose();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public AffineTransform getImageCTM() {
|
||||
@@ -140,11 +58,4 @@ public class ExtractedOcrImage implements OcrImage {
|
||||
return affineTransform;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int getOptimalPageSegmentationMode() {
|
||||
|
||||
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+21
-38
@@ -2,12 +2,16 @@ package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import java.awt.geom.AffineTransform;
|
||||
import java.awt.geom.Point2D;
|
||||
import java.awt.image.BufferedImage;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.PdfDpiCalculator;
|
||||
|
||||
import lombok.SneakyThrows;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
import net.sourceforge.tess4j.ITessAPI;
|
||||
|
||||
public interface OcrImage {
|
||||
|
||||
@@ -27,9 +31,19 @@ public interface OcrImage {
|
||||
int getNumberOnPage();
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the height of the original image (not necessarily in pdf coordinates).
|
||||
*
|
||||
* @return the height of the image
|
||||
*/
|
||||
int getHeight();
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the width of the original image (not necessarily in pdf coordinates).
|
||||
*
|
||||
* @return the width of the image
|
||||
*/
|
||||
int getWidth();
|
||||
|
||||
|
||||
@@ -40,7 +54,7 @@ public interface OcrImage {
|
||||
*/
|
||||
default QuadPoint getImageBounds() {
|
||||
|
||||
// cannot be solved with a nice rotation matrix, since the after rotating the text coordinates in the image will always start at (0,0) and will therefore always start at (0,0) in the PDF.
|
||||
// cannot be solved with a nice rotation matrix. After rotating the text coordinates in the image will always start at (0,0) and will therefore always start at (0,0) in the PDF.
|
||||
// So in order to mimic this behavior we need to start with (0,0) coordinates always.
|
||||
if (getRotationDegrees() == 90 || getRotationDegrees() == 270) {
|
||||
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, getWidth()), new Point2D.Double(getHeight(), getWidth()), new Point2D.Double(getHeight(), 0));
|
||||
@@ -74,17 +88,13 @@ public interface OcrImage {
|
||||
*
|
||||
* @return The optimal page segmentation mode.
|
||||
*/
|
||||
int getOptimalPageSegmentationMode(); // TODO: evaluate if PSM can be dynamically chosen to increase performance
|
||||
default int getOptimalPageSegmentationMode() {
|
||||
|
||||
|
||||
/**
|
||||
* Sets the rotation degree of the OCR image. The rotation degree specifies the amount of rotation applied to the image.
|
||||
* Currently only quadrant rotations are supported.
|
||||
* Rotated partial images work, due to the CTM present in the pdf working with any rotation.
|
||||
*
|
||||
* @param rotationDegree The rotation degree of the OCR image.
|
||||
*/
|
||||
void setRotationDegrees(int rotationDegree);
|
||||
if (getWidth() < 200 || getHeight() < 200) {
|
||||
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
|
||||
}
|
||||
return ITessAPI.TessPageSegMode.PSM_AUTO;
|
||||
} // TODO: evaluate if PSM can be dynamically chosen to increase performance
|
||||
|
||||
|
||||
/**
|
||||
@@ -95,22 +105,6 @@ public interface OcrImage {
|
||||
Pix getPix();
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the rotated image of the OCR image.
|
||||
*
|
||||
* @return The rotated BufferedImage object of the OCR image.
|
||||
*/
|
||||
default Pix getRotatedPix() {
|
||||
|
||||
return switch (360 - getRotationDegrees()) {
|
||||
case 90 -> Leptonica1.pixRotateOrth(getPix(), 1);
|
||||
case 180 -> Leptonica1.pixRotateOrth(getPix(), 2);
|
||||
case 270 -> Leptonica1.pixRotateOrth(getPix(), 3);
|
||||
default -> getPix();
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
default int getDpi() {
|
||||
|
||||
return PdfDpiCalculator.calculateDpi(getImageBounds(), getImageCTM(), getWidth());
|
||||
@@ -125,17 +119,6 @@ public interface OcrImage {
|
||||
AffineTransform getImageCTM();
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the size (width * height) of the image.
|
||||
*
|
||||
* @return The size of the image.
|
||||
*/
|
||||
default int getImageSize() {
|
||||
|
||||
return getHeight() * getWidth();
|
||||
}
|
||||
|
||||
|
||||
default void destroyPix() {
|
||||
|
||||
LeptUtils.disposePix(getPix());
|
||||
|
||||
+3
-13
@@ -7,27 +7,17 @@ import com.knecon.fforesight.service.ocr.processor.service.HOcrPageParser;
|
||||
|
||||
import io.github.karols.hocr4j.Word;
|
||||
|
||||
public record OcrResult(Image image, String hOcrPageAbsolutePath) {
|
||||
public record OcrResult(OcrImage image, String tesseractOutputFilePath) {
|
||||
|
||||
public static OcrResult create(OcrImage image, String tesseractResult) {
|
||||
|
||||
return new OcrResult(Image.fromOcrImage(image), tesseractResult);
|
||||
return new OcrResult(image, tesseractResult);
|
||||
}
|
||||
|
||||
|
||||
public List<Word> getAllWords() {
|
||||
|
||||
return HOcrPageParser.extractHocrPage(hOcrPageAbsolutePath).getAllWords();
|
||||
}
|
||||
|
||||
|
||||
public record Image(Integer pageNumber, AffineTransform ctm, QuadPoint position) {
|
||||
|
||||
public static Image fromOcrImage(OcrImage image) {
|
||||
|
||||
return new Image(image.getPageNumber(), image.getImageCTM(), image.getImageCoordinatesInInitialUserSpace());
|
||||
}
|
||||
|
||||
return HOcrPageParser.extractHocrPage(tesseractOutputFilePath).getAllWords();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.FontStyleDetectionModel;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
|
||||
|
||||
public record OcrResultToWrite(List<TextPositionInImage> textPositionInImage, QuadPoint imageBoundingBox) {
|
||||
|
||||
public static OcrResultToWrite fromFontStyleDetectionModel(FontStyleDetectionModel fontStyleDetectionModel) {
|
||||
|
||||
return new OcrResultToWrite(fontStyleDetectionModel.getTextPositionInImages(), fontStyleDetectionModel.getImageBounds());
|
||||
}
|
||||
|
||||
|
||||
public static Map<Integer, List<OcrResultToWrite>> buildOcrResultsToWrite(List<OcrResult> ocrResults, FontMetricsFactory fontMetricsFactory) {
|
||||
|
||||
return ocrResults.stream()
|
||||
.collect(Collectors.groupingBy(ocrResult -> ocrResult.image().getPageNumber()))
|
||||
.entrySet()
|
||||
.stream()
|
||||
.collect(Collectors.toMap(Map.Entry::getKey,
|
||||
entry -> entry.getValue()
|
||||
.stream()
|
||||
.map(ocrResult -> new OcrResultToWrite(ocrResult.getAllWords()
|
||||
.stream()
|
||||
.filter(word -> !word.isBlank())
|
||||
.map(word -> new TextPositionInImage(word, ocrResult.image().getImageCTM(), fontMetricsFactory, FontStyle.REGULAR))
|
||||
.toList(), ocrResult.image().getImageCoordinatesInInitialUserSpace()))
|
||||
.toList()));
|
||||
}
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
|
||||
public record PageInformation(int height, int width, int number, int rotationDegrees) {
|
||||
|
||||
public static PageInformation fromPDPage(int pageNum, PDPage page) {
|
||||
|
||||
return new PageInformation((int) page.getMediaBox().getHeight(), (int) page.getMediaBox().getWidth(), pageNum, page.getRotation());
|
||||
}
|
||||
|
||||
}
|
||||
+6
@@ -97,4 +97,10 @@ public record QuadPoint(Point2D a, Point2D b, Point2D c, Point2D d) {
|
||||
d().getY());
|
||||
}
|
||||
|
||||
|
||||
public double size() {
|
||||
|
||||
return a().distance(b()) * a().distance(d());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+10
-1
@@ -1,5 +1,14 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
public record RenderedPageImageFile(int pageNumber, String absoluteFilePath) {
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
|
||||
public record RenderedPageImageFile(int pageNumber, String absoluteFilePath) implements UnprocessedImage {
|
||||
|
||||
@Override
|
||||
public Pix asPix() {
|
||||
|
||||
return Leptonica1.pixRead(absoluteFilePath);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+8
-47
@@ -8,6 +8,7 @@ import org.apache.pdfbox.pdmodel.PDPage;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.Setter;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
@@ -16,36 +17,17 @@ import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.tess4j.ITessAPI;
|
||||
|
||||
@Getter
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class RenderedPageOcrImage implements OcrImage {
|
||||
|
||||
final String absoluteImagePath;
|
||||
final int height;
|
||||
final int width;
|
||||
final PageInformation pageInformation;
|
||||
final Pix pix;
|
||||
@Setter
|
||||
int height;
|
||||
int width;
|
||||
PageInformation pageInformation;
|
||||
Pix pix;
|
||||
int rotationDegrees;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public RenderedPageOcrImage(RenderedPageImageFile renderedPageImageFile, PDDocument document) {
|
||||
|
||||
this.pageInformation = PageInformation.fromPDPage(renderedPageImageFile.pageNumber(), document.getPage(renderedPageImageFile.pageNumber() - 1));
|
||||
this.absoluteImagePath = renderedPageImageFile.absoluteFilePath();
|
||||
this.pix = Leptonica1.pixRead(absoluteImagePath);
|
||||
this.height = getPix().h;
|
||||
this.width = getPix().w;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int getOptimalPageSegmentationMode() {
|
||||
|
||||
return ITessAPI.TessPageSegMode.PSM_SINGLE_BLOCK;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public AffineTransform getImageCTM() {
|
||||
|
||||
@@ -78,17 +60,6 @@ public class RenderedPageOcrImage implements OcrImage {
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public QuadPoint getImageBounds() {
|
||||
|
||||
if (rotationDegrees == 90 || rotationDegrees == 270) {
|
||||
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, width), new Point2D.Double(height, width), new Point2D.Double(height, 0));
|
||||
} else {
|
||||
return new QuadPoint(new Point2D.Double(0, 0), new Point2D.Double(0, height), new Point2D.Double(width, height), new Point2D.Double(width, 0));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int getPageNumber() {
|
||||
|
||||
@@ -107,7 +78,7 @@ public class RenderedPageOcrImage implements OcrImage {
|
||||
|
||||
// PDFBox always returns page height and width based on rotation
|
||||
double pageWidth;
|
||||
if (pageInformation.rotationDegrees == 90 || pageInformation.rotationDegrees == 270) {
|
||||
if (pageInformation.rotationDegrees() == 90 || pageInformation.rotationDegrees() == 270) {
|
||||
pageWidth = pageInformation.height();
|
||||
} else {
|
||||
pageWidth = pageInformation.width();
|
||||
@@ -116,14 +87,4 @@ public class RenderedPageOcrImage implements OcrImage {
|
||||
return pageWidth / width;
|
||||
}
|
||||
|
||||
|
||||
private record PageInformation(int height, int width, int number, int rotationDegrees) {
|
||||
|
||||
public static PageInformation fromPDPage(int pageNum, PDPage page) {
|
||||
|
||||
return new PageInformation((int) page.getCropBox().getHeight(), (int) page.getCropBox().getWidth(), pageNum, page.getRotation());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+19
-6
@@ -7,29 +7,35 @@ import org.apache.pdfbox.pdmodel.font.PDFont;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
|
||||
|
||||
import io.github.karols.hocr4j.Bounds;
|
||||
import io.github.karols.hocr4j.Word;
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Getter
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE, makeFinal = true)
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class TextPositionInImage {
|
||||
|
||||
QuadPoint position;
|
||||
String text;
|
||||
AffineTransform imageCTM;
|
||||
final QuadPoint position;
|
||||
final String text;
|
||||
final AffineTransform imageCTM;
|
||||
|
||||
@Setter
|
||||
FontMetricsFactory fontMetricsFactory;
|
||||
@Setter
|
||||
FontStyle fontStyle;
|
||||
|
||||
|
||||
public TextPositionInImage(Word word, AffineTransform imageCTM, FontMetricsFactory fontMetricsFactory) {
|
||||
public TextPositionInImage(Word word, AffineTransform imageCTM, FontMetricsFactory fontMetricsFactory, FontStyle fontStyle) {
|
||||
|
||||
this.position = QuadPoint.fromBounds(word.getBounds());
|
||||
this.text = word.getText();
|
||||
this.imageCTM = imageCTM;
|
||||
this.fontMetricsFactory = fontMetricsFactory;
|
||||
this.fontStyle = fontStyle;
|
||||
}
|
||||
|
||||
|
||||
@@ -90,6 +96,13 @@ public class TextPositionInImage {
|
||||
}
|
||||
|
||||
|
||||
public double getTextHeight() {
|
||||
|
||||
var metrics = fontMetricsFactory.calculateMetrics(text, getTransformedWidth(), getTransformedHeight());
|
||||
return fontMetricsFactory.calculateFontSize(text, getTransformedWidth()) * metrics.getHeightScaling();
|
||||
}
|
||||
|
||||
|
||||
public double getHeight() {
|
||||
|
||||
return position.a().distance(position.b());
|
||||
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model;
|
||||
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
|
||||
public interface UnprocessedImage {
|
||||
|
||||
Pix asPix();
|
||||
|
||||
}
|
||||
+58
@@ -0,0 +1,58 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
@Getter
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public final class FontStyleDetectionModel {
|
||||
|
||||
QuadPoint imageBounds;
|
||||
Pix image;
|
||||
List<TextPositionAndWordImage> textPositionsAndWordImages;
|
||||
|
||||
|
||||
public static FontStyleDetectionModel fromOcrResult(OcrResult ocrResult, FontMetricsFactory fontMetricsFactory, OcrServiceSettings settings) {
|
||||
|
||||
var image = Leptonica1.pixRead(ocrResult.tesseractOutputFilePath() + ".tiff");
|
||||
var wordPixes = ocrResult.getAllWords().stream().filter(word -> !word.isBlank()).map(word -> TextPositionAndWordImage.create(ocrResult.image().getImageCTM(), word, image, settings, fontMetricsFactory)).toList();
|
||||
|
||||
return new FontStyleDetectionModel(ocrResult.image().getImageCoordinatesInInitialUserSpace(), image, wordPixes);
|
||||
}
|
||||
|
||||
|
||||
public List<TextPositionInImage> getTextPositionInImages() {
|
||||
|
||||
return textPositionsAndWordImages.stream().map(TextPositionAndWordImage::getTextPositionInImage).toList();
|
||||
}
|
||||
|
||||
|
||||
public List<WordImage> getWordImages() {
|
||||
|
||||
return textPositionsAndWordImages.stream().map(TextPositionAndWordImage::getWordImage).toList();
|
||||
}
|
||||
|
||||
|
||||
public void dispose() {
|
||||
|
||||
LeptUtils.disposePix(image);
|
||||
getWordImages().forEach(WordImage::dispose);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
|
||||
|
||||
import java.awt.geom.AffineTransform;
|
||||
import java.util.Objects;
|
||||
|
||||
import org.apache.commons.math3.ml.clustering.Clusterable;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
import io.github.karols.hocr4j.Word;
|
||||
import lombok.Getter;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
|
||||
@Getter
|
||||
public final class TextPositionAndWordImage implements Clusterable {
|
||||
|
||||
private final TextPositionInImage textPositionInImage;
|
||||
private final WordImage wordImage;
|
||||
|
||||
|
||||
public TextPositionAndWordImage(TextPositionInImage textPositionInImage, WordImage wordImage) {
|
||||
|
||||
this.textPositionInImage = textPositionInImage;
|
||||
this.wordImage = wordImage;
|
||||
}
|
||||
|
||||
|
||||
public static TextPositionAndWordImage create(AffineTransform imageCTM, Word word, Pix image, OcrServiceSettings settings, FontMetricsFactory fontMetricsFactory) {
|
||||
|
||||
TextPositionInImage textPositionInImage = new TextPositionInImage(word, imageCTM, fontMetricsFactory, FontStyle.REGULAR);
|
||||
WordImage wordImage = new WordImage(textPositionInImage.getTextHeight(), word, image, settings);
|
||||
return new TextPositionAndWordImage(textPositionInImage, wordImage);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public double[] getPoint() {
|
||||
|
||||
return wordImage.getPoint();
|
||||
}
|
||||
|
||||
|
||||
public double getTextHeight() {
|
||||
|
||||
return wordImage.getTextHeight();
|
||||
}
|
||||
|
||||
}
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.model.scriptdetection;
|
||||
|
||||
import org.apache.commons.math3.ml.clustering.Clusterable;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.service.scriptdetection.StrokeWidthCalculator;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
|
||||
|
||||
import io.github.karols.hocr4j.Word;
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import net.sourceforge.lept4j.Box;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
@Getter
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class WordImage implements Clusterable {
|
||||
|
||||
Pix image;
|
||||
String text;
|
||||
double textHeight;
|
||||
OcrServiceSettings settings;
|
||||
|
||||
|
||||
public WordImage(double textHeight, Word word, Pix originalImage, OcrServiceSettings settings) {
|
||||
|
||||
Box box = new Box(word.getBounds().getLeft(), word.getBounds().getTop(), word.getBounds().getWidth(), word.getBounds().getHeight(), 1);
|
||||
this.image = Leptonica1.pixClipRectangle(originalImage, box, null);
|
||||
box.clear();
|
||||
this.text = word.getText();
|
||||
this.textHeight = textHeight;
|
||||
this.settings = settings;
|
||||
}
|
||||
|
||||
|
||||
public boolean hasLargerStrokeWidth(double strokeWidth) {
|
||||
|
||||
int roundedStrokeWidth = (int) Math.round(strokeWidth);
|
||||
double roundingError = (roundedStrokeWidth - strokeWidth) / strokeWidth;
|
||||
|
||||
// add 1 to open a bit bigger than the estimated regular stroke width
|
||||
Pix openedPix = Leptonica1.pixOpenBrick(null, image, roundedStrokeWidth + 1, roundedStrokeWidth + 1);
|
||||
|
||||
double openedPixelDensity = ImageProcessingUtils.calculatePixelDensity(openedPix);
|
||||
|
||||
double pixelDensity = ImageProcessingUtils.calculatePixelDensity(image);
|
||||
|
||||
LeptUtils.disposePix(openedPix);
|
||||
|
||||
return (openedPixelDensity * (1 + roundingError)) / pixelDensity > (settings.getBoldThreshold());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public double[] getPoint() {
|
||||
|
||||
return new double[]{textHeight};
|
||||
}
|
||||
|
||||
|
||||
public void dispose() {
|
||||
|
||||
LeptUtils.disposePix(image);
|
||||
}
|
||||
|
||||
}
|
||||
+40
-28
@@ -3,19 +3,21 @@ package com.knecon.fforesight.service.ocr.processor.service;
|
||||
import java.io.InputStream;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
import java.util.concurrent.LinkedBlockingDeque;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageOcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.ProcessIOLogger;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.BlockingQueueFiller;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.GhostScriptOutputHandler;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ListSplittingUtils;
|
||||
|
||||
@@ -42,17 +44,19 @@ public class GhostScriptService {
|
||||
String documentAbsolutePath,
|
||||
Path tmpImageDir,
|
||||
PDDocument document,
|
||||
BlockingQueue<OcrImage> imageOutputQueue,
|
||||
BlockingQueue<UnprocessedImage> imageProcessingQueue,
|
||||
Statistics stats) {
|
||||
|
||||
BlockingQueue<RenderedPageImageFile> imageFileCollectorQueue = new LinkedBlockingDeque<>();
|
||||
BlockingQueueFiller asyncTransferThread = new BlockingQueueFiller(imageFileCollectorQueue, imageProcessingQueue);
|
||||
asyncTransferThread.start();
|
||||
int numOfProcesses = Math.min(settings.getGsProcessCount(), stitchedPageNumbers.size());
|
||||
|
||||
List<List<ProcessInfo>> processInfoBatches = buildSubListForEachProcess(stitchedPageNumbers,
|
||||
numOfProcesses,
|
||||
2 * settings.getOcrThreadCount()); // use 2 times the thread count as batch size, such that GS generates the rendered pages as needed by the OCR Threads
|
||||
256 * numOfProcesses); // GS has a limit on how many pageIndices per call are possible, so we limit it to 256 pages per process
|
||||
for (int batchIdx = 0; batchIdx < processInfoBatches.size(); batchIdx++) {
|
||||
long timestamp = System.currentTimeMillis();
|
||||
List<RenderedPageImageFile> renderedPageImageFiles = Collections.synchronizedList(new LinkedList<>());
|
||||
List<ProcessInfo> processInfos = processInfoBatches.get(batchIdx);
|
||||
|
||||
log.info("Batch {}: Running {} gs processes with ({}) pages each",
|
||||
@@ -63,9 +67,9 @@ public class GhostScriptService {
|
||||
int finalBatchIdx = batchIdx;
|
||||
List<Process> processes = processInfos.stream()
|
||||
.parallel()
|
||||
.map(info -> buildCmdArgs(info.processIdx(), finalBatchIdx, info.stitchedPageNumbers(), tmpImageDir, documentAbsolutePath, renderedPageImageFiles))
|
||||
.peek(s -> log.debug(String.join(" ", s)))
|
||||
.map(this::executeProcess)
|
||||
.map(info -> buildCmdArgs(info.processIdx(), finalBatchIdx, info.stitchedPageNumbers(), tmpImageDir, documentAbsolutePath))
|
||||
.peek(s -> log.debug(String.join(" ", s.cmdArgs())))
|
||||
.map(processInfo -> executeProcess(processInfo, imageFileCollectorQueue))
|
||||
.toList();
|
||||
|
||||
List<Integer> processExitCodes = new LinkedList<>();
|
||||
@@ -73,14 +77,9 @@ public class GhostScriptService {
|
||||
processExitCodes.add(process.waitFor());
|
||||
}
|
||||
stats.increasePDF2ImgDuration(System.currentTimeMillis() - timestamp);
|
||||
|
||||
log.info("Batch {}: Ghostscript processes finished with exit codes " + processExitCodes, batchIdx);
|
||||
for (RenderedPageImageFile renderedPageImageFile : renderedPageImageFiles) {
|
||||
OcrImage image = new RenderedPageOcrImage(renderedPageImageFile, document);
|
||||
imageOutputQueue.put(image);
|
||||
}
|
||||
|
||||
}
|
||||
asyncTransferThread.setAllImagesQueued(true);
|
||||
}
|
||||
|
||||
|
||||
@@ -107,20 +106,28 @@ public class GhostScriptService {
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private String[] buildCmdArgs(Integer processIdx,
|
||||
Integer batchIdx,
|
||||
List<Integer> stitchedImagePageIndices,
|
||||
Path outputDir,
|
||||
String documentAbsolutePath,
|
||||
List<RenderedPageImageFile> fullPageImages) {
|
||||
private ProcessCmdsAndRenderedImageFiles buildCmdArgs(Integer processIdx,
|
||||
Integer batchIdx,
|
||||
List<Integer> stitchedImagePageIndices,
|
||||
Path outputDir,
|
||||
String documentAbsolutePath) {
|
||||
|
||||
String imagePathFormat = outputDir.resolve("output_" + processIdx + "_" + batchIdx + ".%04d" + FORMAT).toFile().toString();
|
||||
|
||||
Map<Integer, RenderedPageImageFile> fullPageImages = new HashMap<>();
|
||||
for (int i = 0; i < stitchedImagePageIndices.size(); i++) {
|
||||
Integer pageNumber = stitchedImagePageIndices.get(i);
|
||||
fullPageImages.add(new RenderedPageImageFile(pageNumber, String.format(imagePathFormat, i + 1)));
|
||||
fullPageImages.put(pageNumber, new RenderedPageImageFile(pageNumber, String.format(imagePathFormat, i + 1)));
|
||||
}
|
||||
|
||||
String[] cmdArgs = buildCmdArgs(stitchedImagePageIndices, documentAbsolutePath, imagePathFormat);
|
||||
|
||||
return new ProcessCmdsAndRenderedImageFiles(cmdArgs, fullPageImages);
|
||||
}
|
||||
|
||||
|
||||
private String[] buildCmdArgs(List<Integer> stitchedImagePageIndices, String documentAbsolutePath, String imagePathFormat) {
|
||||
|
||||
StringBuilder sPageList = new StringBuilder();
|
||||
int i = 1;
|
||||
for (Integer integer : stitchedImagePageIndices) {
|
||||
@@ -131,18 +138,19 @@ public class GhostScriptService {
|
||||
i++;
|
||||
}
|
||||
|
||||
return new String[]{"gs", "-dNOPAUSE", "-sDEVICE=" + DEVICE, "-r" + settings.getDpi(), "-sPageList=" + sPageList, "-sOutputFile=" + imagePathFormat, documentAbsolutePath, "-c", "quit"};
|
||||
String[] cmdArgs = new String[]{"gs", "-dNOPAUSE", "-sDEVICE=" + DEVICE, "-r" + settings.getDpi(), "-sPageList=" + sPageList, "-sOutputFile=" + imagePathFormat, documentAbsolutePath, "-c", "quit"};
|
||||
return cmdArgs;
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private Process executeProcess(String[] cmdArgs) {
|
||||
private Process executeProcess(ProcessCmdsAndRenderedImageFiles processInfo, BlockingQueue<RenderedPageImageFile> imageFileCollectorQueue) {
|
||||
|
||||
Process p = Runtime.getRuntime().exec(cmdArgs);
|
||||
Process p = Runtime.getRuntime().exec(processInfo.cmdArgs());
|
||||
InputStream stdOut = p.getInputStream();
|
||||
ProcessIOLogger stdOutLogger = new ProcessIOLogger(stdOut, "GS", ProcessIOLogger.Type.STD_OUT);
|
||||
GhostScriptOutputHandler stdOutLogger = GhostScriptOutputHandler.stdOut(stdOut, processInfo.renderedPageImageFiles(), imageFileCollectorQueue);
|
||||
InputStream stdError = p.getErrorStream();
|
||||
ProcessIOLogger stdErrorLogger = new ProcessIOLogger(stdError, "GS", ProcessIOLogger.Type.ERROR);
|
||||
GhostScriptOutputHandler stdErrorLogger = GhostScriptOutputHandler.errorHandler(stdError);
|
||||
|
||||
stdOutLogger.start();
|
||||
stdErrorLogger.start();
|
||||
@@ -150,6 +158,10 @@ public class GhostScriptService {
|
||||
}
|
||||
|
||||
|
||||
private record ProcessCmdsAndRenderedImageFiles(String[] cmdArgs, Map<Integer, RenderedPageImageFile> renderedPageImageFiles) {
|
||||
|
||||
}
|
||||
|
||||
private record ProcessInfo(Integer processIdx, List<Integer> stitchedPageNumbers) {
|
||||
|
||||
}
|
||||
|
||||
+13
-24
@@ -1,7 +1,6 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service;
|
||||
|
||||
import java.awt.Graphics;
|
||||
import java.awt.image.BufferedImage;
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.io.IOException;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
@@ -18,13 +17,12 @@ import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
|
||||
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
|
||||
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
import lombok.Getter;
|
||||
@@ -33,8 +31,7 @@ import lombok.SneakyThrows;
|
||||
@Getter
|
||||
public class ImageStreamEngine extends PDFStreamEngine {
|
||||
|
||||
private ExtractedOcrImage currentImageOnPage;
|
||||
private List<ExtractedOcrImage> imagesOnCurrentPage;
|
||||
private List<ExtractedImage> imagesOnCurrentPage;
|
||||
private OcrServiceSettings settings;
|
||||
private int pageNum;
|
||||
|
||||
@@ -69,23 +66,15 @@ public class ImageStreamEngine extends PDFStreamEngine {
|
||||
}
|
||||
|
||||
Matrix imageCTM = getGraphicsState().getCurrentTransformationMatrix();
|
||||
if (imageXObject.getColorSpace() instanceof PDDeviceRGB) {
|
||||
BufferedImage image = imageXObject.getImage();
|
||||
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi(), false);
|
||||
} else if (imageXObject.getColorSpace() instanceof PDDeviceGray) {
|
||||
BufferedImage image = imageXObject.getImage();
|
||||
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi(), true);
|
||||
} else {
|
||||
BufferedImage pdfImage = imageXObject.getImage();
|
||||
BufferedImage image = new BufferedImage(pdfImage.getWidth(), pdfImage.getHeight(),
|
||||
BufferedImage.TYPE_BYTE_GRAY);
|
||||
Graphics g = image.getGraphics();
|
||||
g.drawImage(pdfImage, 0, 0, null);
|
||||
g.dispose();
|
||||
this.currentImageOnPage = new ExtractedOcrImage(pageNum, imagesOnCurrentPage.size(), image, imageCTM, settings.getDpi(), true);
|
||||
}
|
||||
this.imagesOnCurrentPage.add(this.currentImageOnPage);
|
||||
//imagesOnPages.add(this.currentImageOnPage);
|
||||
this.imagesOnCurrentPage.add(new ExtractedImage(pageNum,
|
||||
QuadPoint.fromRectangle2D(new Rectangle2D.Double(0, 0, imageXObject.getWidth(), imageXObject.getHeight())),
|
||||
imageXObject.getHeight(),
|
||||
imageXObject.getWidth(),
|
||||
imageXObject.getImage(),
|
||||
imageCTM,
|
||||
imagesOnCurrentPage.size(),
|
||||
imageXObject.getColorSpace()));
|
||||
|
||||
} else if (xobject instanceof PDFormXObject) {
|
||||
PDFormXObject form = (PDFormXObject) xobject;
|
||||
showForm(form);
|
||||
|
||||
+13
-4
@@ -9,6 +9,7 @@ import java.io.OutputStream;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ArrayBlockingQueue;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
import java.util.stream.IntStream;
|
||||
@@ -20,8 +21,10 @@ import org.springframework.util.FileSystemUtils;
|
||||
|
||||
import com.iqser.red.pdftronlogic.commons.InvisibleElementRemovalService;
|
||||
import com.iqser.red.pdftronlogic.commons.WatermarkRemovalService;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.scriptdetection.FontStyleDetector;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.OCRThread;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
@@ -44,6 +47,7 @@ public class OCRService {
|
||||
InvisibleElementRemovalService invisibleElementRemovalService;
|
||||
OcrResultWriter ocrResultWriter;
|
||||
GhostScriptService ghostScriptService;
|
||||
FontStyleDetector boldDetector;
|
||||
|
||||
|
||||
/**
|
||||
@@ -107,7 +111,7 @@ public class OCRService {
|
||||
int numberOfOcrThreads = Math.min(settings.getOcrThreadCount(), document.getNumberOfPages());
|
||||
stats = new Statistics(numberOfExtractThreads, numberOfOcrThreads);
|
||||
|
||||
BlockingQueue<OcrImage> ocrImageQueue = new ArrayBlockingQueue<>(numberOfOcrThreads);
|
||||
BlockingQueue<OcrImage> ocrImageQueue = new ArrayBlockingQueue<>((int) (1.5 * numberOfOcrThreads));
|
||||
|
||||
OcrImageFactory ocrImageFactory = new OcrImageFactory(document,
|
||||
documentFile,
|
||||
@@ -128,16 +132,21 @@ public class OCRService {
|
||||
.toList();
|
||||
log.info("Started {} OCR consumer threads, listening for images on the queue", ocrThreads.size());
|
||||
ocrImageFactory.join();
|
||||
log.info("Extracted all images, interrupting ocr threads");
|
||||
log.info("Processed all images, interrupting ocr threads");
|
||||
|
||||
ocrThreads.forEach(Thread::interrupt);
|
||||
for (OCRThread ocrThread : ocrThreads) {
|
||||
ocrThread.join();
|
||||
}
|
||||
|
||||
log.info("OCR processing has finished, writing results");
|
||||
log.info("Tesseract OCR has finished for file {} and dossier {}", fileId, dossierId);
|
||||
|
||||
timestamp = System.currentTimeMillis();
|
||||
var dictionariesToUpdate = ocrResultWriter.drawOcrResultsToPdf(document, ocrResults);
|
||||
Map<Integer, List<OcrResultToWrite>> imageWithTextPositionsPerPage = boldDetector.detectBold(ocrResults, document);
|
||||
stats.increaseFontStyleDetectionDuration(System.currentTimeMillis() - timestamp);
|
||||
|
||||
timestamp = System.currentTimeMillis();
|
||||
var dictionariesToUpdate = ocrResultWriter.drawOcrResultsToPdf(document, imageWithTextPositionsPerPage);
|
||||
log.info("Saving document");
|
||||
document.saveIncremental(out, dictionariesToUpdate);
|
||||
stats.increaseWritingTextDuration(System.currentTimeMillis() - timestamp);
|
||||
|
||||
+22
-6
@@ -6,13 +6,17 @@ import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ArrayBlockingQueue;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.ImageExtractionThread;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.ImageProcessingThread;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ListSplittingUtils;
|
||||
|
||||
@@ -29,6 +33,8 @@ public class OcrImageFactory {
|
||||
File documentFile;
|
||||
Path tmpImageDir;
|
||||
GhostScriptService ghostScriptService;
|
||||
BlockingQueue<UnprocessedImage> imageProcessingQueue;
|
||||
ImageProcessingThread imageProcessingThread;
|
||||
BlockingQueue<OcrImage> imageOutputQueue;
|
||||
List<ImageExtractionThread> imageExtractionThreads;
|
||||
List<Integer> stitchedPageNumbers;
|
||||
@@ -40,7 +46,7 @@ public class OcrImageFactory {
|
||||
Path tmpImageDir,
|
||||
int numberOfThreads,
|
||||
GhostScriptService ghostScriptService,
|
||||
BlockingQueue<OcrImage> imageOutputQueue,
|
||||
BlockingQueue<OcrImage> imageOcrQueue,
|
||||
OcrProgressLogger logger,
|
||||
OcrServiceSettings settings,
|
||||
Statistics stats) {
|
||||
@@ -49,7 +55,8 @@ public class OcrImageFactory {
|
||||
this.documentFile = documentFile;
|
||||
this.tmpImageDir = tmpImageDir;
|
||||
this.ghostScriptService = ghostScriptService;
|
||||
this.imageOutputQueue = imageOutputQueue;
|
||||
this.imageOutputQueue = imageOcrQueue;
|
||||
this.imageProcessingQueue = new ArrayBlockingQueue<>(imageOcrQueue.remainingCapacity());
|
||||
this.stitchedPageNumbers = Collections.synchronizedList(new LinkedList<>());
|
||||
this.stats = stats;
|
||||
|
||||
@@ -57,8 +64,10 @@ public class OcrImageFactory {
|
||||
|
||||
List<List<Integer>> balancedPageNumbers = ListSplittingUtils.buildBalancedContinuousSublist(document.getNumberOfPages(), numberOfThreads);
|
||||
for (int i = 0; i < balancedPageNumbers.size(); i++) {
|
||||
imageExtractionThreads.add(new ImageExtractionThread(i, balancedPageNumbers.get(i), documentFile, logger, stats, settings, imageOutputQueue, stitchedPageNumbers));
|
||||
imageExtractionThreads.add(new ImageExtractionThread(i, balancedPageNumbers.get(i), documentFile, logger, stats, settings, imageProcessingQueue, stitchedPageNumbers));
|
||||
}
|
||||
this.imageProcessingThread = new ImageProcessingThread(imageProcessingQueue, imageOcrQueue, stats, settings, document);
|
||||
|
||||
log.info("Started {} image extraction threads, with ({}) pages each",
|
||||
imageExtractionThreads.size(),
|
||||
imageExtractionThreads.stream().map(ImageExtractionThread::getPageIndices).map(List::size).map(String::valueOf).collect(Collectors.joining(", ")));
|
||||
@@ -70,6 +79,8 @@ public class OcrImageFactory {
|
||||
for (ImageExtractionThread imageExtractionThread : imageExtractionThreads) {
|
||||
imageExtractionThread.start();
|
||||
}
|
||||
imageProcessingThread.start();
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -79,11 +90,16 @@ public class OcrImageFactory {
|
||||
for (ImageExtractionThread imageExtractionThread : imageExtractionThreads) {
|
||||
imageExtractionThread.join();
|
||||
}
|
||||
if (stitchedPageNumbers.isEmpty()) {
|
||||
return;
|
||||
|
||||
if (!stitchedPageNumbers.isEmpty()) {
|
||||
ghostScriptService.renderPagesAsImagesBatchedAndAddToQueue(stitchedPageNumbers, documentFile.toString(), tmpImageDir, document, imageProcessingQueue, stats);
|
||||
}
|
||||
|
||||
ghostScriptService.renderPagesAsImagesBatchedAndAddToQueue(stitchedPageNumbers, documentFile.toString(), tmpImageDir, document, imageOutputQueue, stats);
|
||||
imageProcessingThread.setAllImagesExtracted(true);
|
||||
imageProcessingThread.interrupt();
|
||||
|
||||
imageProcessingThread.join();
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+14
-21
@@ -2,11 +2,11 @@ package com.knecon.fforesight.service.ocr.processor.service;
|
||||
|
||||
import java.awt.Color;
|
||||
import java.awt.geom.Point2D;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
@@ -20,11 +20,9 @@ import org.apache.pdfbox.pdmodel.graphics.optionalcontent.PDOptionalContentPrope
|
||||
import org.apache.pdfbox.pdmodel.graphics.state.RenderingMode;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.QuadPoint;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.TextPositionInImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.Type0FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
@@ -44,19 +42,17 @@ public class OcrResultWriter {
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public Set<COSDictionary> drawOcrResultsToPdf(PDDocument document, List<OcrResult> ocrResults) {
|
||||
public Set<COSDictionary> drawOcrResultsToPdf(PDDocument document, Map<Integer, List<OcrResultToWrite>> imagesWithResultsPerPage) {
|
||||
|
||||
FontMetricsFactory fontMetricsFactory = new Type0FontMetricsFactory(document);
|
||||
Set<COSDictionary> dictionariesToUpdate = new HashSet<>();
|
||||
Map<Integer, List<OcrResult>> resultsPerPage = ocrResults.stream().collect(Collectors.groupingBy(result -> result.image().pageNumber()));
|
||||
resultsPerPage.keySet().forEach(pageNumber -> drawResultsPerPage(document, pageNumber, resultsPerPage, dictionariesToUpdate, fontMetricsFactory));
|
||||
imagesWithResultsPerPage.keySet().forEach(pageNumber -> drawResultsPerPage(document, pageNumber, imagesWithResultsPerPage.get(pageNumber), dictionariesToUpdate));
|
||||
dictionariesToUpdate.add(document.getDocumentInformation().getCOSObject());
|
||||
return dictionariesToUpdate;
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private void drawResultsPerPage(PDDocument document, Integer pageNumber, Map<Integer, List<OcrResult>> resultsPerPage, Set<COSDictionary> dictionariesToUpdate, FontMetricsFactory fontMetricsFactory) {
|
||||
private void drawResultsPerPage(PDDocument document, Integer pageNumber, List<OcrResultToWrite> ocrResultToWrite, Set<COSDictionary> dictionariesToUpdate) {
|
||||
|
||||
var pdPage = document.getPage(pageNumber - 1);
|
||||
|
||||
@@ -69,7 +65,7 @@ public class OcrResultWriter {
|
||||
|
||||
escapeContentStreams(document, pdPage);
|
||||
|
||||
List<TextPositionInImage> words = buildTextPositionsOnPage(pageNumber, resultsPerPage, fontMetricsFactory);
|
||||
List<TextPositionInImage> words = ocrResultToWrite.stream().map(OcrResultToWrite::textPositionInImage).flatMap(Collection::stream).toList();
|
||||
try (var contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true)) {
|
||||
|
||||
// write invisible ocr text inside tagged content
|
||||
@@ -86,7 +82,6 @@ public class OcrResultWriter {
|
||||
// write visible ocr text inside optional group
|
||||
contentStream.beginMarkedContent(COSName.OC, textDebugLayer);
|
||||
contentStream.saveGraphicsState();
|
||||
contentStream.setNonStrokingColor(Color.BLUE);
|
||||
words.forEach(word -> drawVisibleWord(word, contentStream));
|
||||
contentStream.restoreGraphicsState();
|
||||
contentStream.endMarkedContent();
|
||||
@@ -94,7 +89,9 @@ public class OcrResultWriter {
|
||||
// write word bounding boxes (tesseract output) inside optional group
|
||||
contentStream.beginMarkedContent(COSName.OC, bBoxDebugLayer);
|
||||
contentStream.saveGraphicsState();
|
||||
resultsPerPage.get(pageNumber).stream().map(OcrResult::image).forEach(image -> drawGrid(contentStream, image.position()));
|
||||
ocrResultToWrite.stream()
|
||||
.map(OcrResultToWrite::imageBoundingBox)
|
||||
.forEach(imagePosition -> drawGrid(contentStream, imagePosition));
|
||||
words.stream().map(TextPositionInImage::getTransformedTextBBox).forEach(word -> drawRectangle(contentStream, word));
|
||||
contentStream.restoreGraphicsState();
|
||||
contentStream.endMarkedContent();
|
||||
@@ -105,15 +102,6 @@ public class OcrResultWriter {
|
||||
}
|
||||
|
||||
|
||||
private static List<TextPositionInImage> buildTextPositionsOnPage(Integer pageNumber, Map<Integer, List<OcrResult>> resultsPerPage, FontMetricsFactory fontMetricsFactory) {
|
||||
|
||||
return resultsPerPage.get(pageNumber)
|
||||
.stream()
|
||||
.flatMap(result -> result.getAllWords().stream().filter(word -> !word.isBlank()).map(word -> new TextPositionInImage(word, result.image().ctm(), fontMetricsFactory)))
|
||||
.toList();
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private static void escapeContentStreams(PDDocument document, PDPage pdPage) {
|
||||
// We need to append to the contentstream, otherwise the content could be overlapped by images
|
||||
@@ -196,6 +184,11 @@ public class OcrResultWriter {
|
||||
private void drawWord(TextPositionInImage position, PDPageContentStream contentStream, RenderingMode renderingMode) {
|
||||
|
||||
try {
|
||||
contentStream.setNonStrokingColor(switch (position.getFontStyle()) {
|
||||
case BOLD -> Color.RED;
|
||||
case ITALIC -> Color.GREEN;
|
||||
default -> Color.BLUE;
|
||||
});
|
||||
contentStream.beginText();
|
||||
contentStream.setRenderingMode(renderingMode);
|
||||
contentStream.setFont(position.getFont(), (float) position.getFontSize());
|
||||
|
||||
+20
-2
@@ -15,14 +15,18 @@ public class Statistics {
|
||||
List<Long> tesseractDuration;
|
||||
AtomicLong pdf2ImgDuration;
|
||||
AtomicLong writingTextDuration;
|
||||
AtomicLong imageProcessingDuration;
|
||||
AtomicLong fontStyleDetectionDuration;
|
||||
|
||||
|
||||
public Statistics(int numberOfExtractThreads, int numberOfOcrThreads) {
|
||||
|
||||
this.imageExtraction = Collections.synchronizedList(new ArrayList<>(Collections.nCopies(numberOfExtractThreads, 0L)));
|
||||
this.tesseractDuration = Collections.synchronizedList(new ArrayList<>(Collections.nCopies(numberOfOcrThreads, 0L)));
|
||||
this.fontStyleDetectionDuration = new AtomicLong(0);
|
||||
this.pdf2ImgDuration = new AtomicLong(0);
|
||||
this.writingTextDuration = new AtomicLong(0);
|
||||
this.imageProcessingDuration = new AtomicLong(0);
|
||||
}
|
||||
|
||||
|
||||
@@ -32,6 +36,12 @@ public class Statistics {
|
||||
}
|
||||
|
||||
|
||||
public void increaseImageProcessing(long duration) {
|
||||
|
||||
imageProcessingDuration.addAndGet(duration);
|
||||
}
|
||||
|
||||
|
||||
public void increaseTesseractDuration(int threadId, long duration) {
|
||||
|
||||
tesseractDuration.set(threadId, tesseractDuration.get(threadId) + duration);
|
||||
@@ -49,19 +59,27 @@ public class Statistics {
|
||||
writingTextDuration.addAndGet(duration);
|
||||
}
|
||||
|
||||
public void increaseFontStyleDetectionDuration(long duration) {
|
||||
|
||||
fontStyleDetectionDuration.addAndGet(duration);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return String.format("imageExtraction: mean %.2f s, max %.2f s, min %.2f, tesseract: mean %.2f s, max %.2f s, min %.2f, PDF2Img=%.2f s, writingText=%.2f s",
|
||||
return String.format(
|
||||
"imageExtraction: mean %.2f s, max %.2f s, min %.2f, tesseract: mean %.2f s, max %.2f s, min %.2f, ImageProcessing=%.2f s, PDF2Img=%.2f s, writingText=%.2f s, FontstyleDetection=%.2f s",
|
||||
((float) imageExtraction.stream().mapToLong(Long::longValue).average().orElse(0) / 1000),
|
||||
((float) imageExtraction.stream().mapToLong(Long::longValue).max().orElse(0) / 1000),
|
||||
((float) imageExtraction.stream().mapToLong(Long::longValue).min().orElse(0) / 1000),
|
||||
((float) tesseractDuration.stream().mapToLong(Long::longValue).average().orElse(0) / 1000),
|
||||
((float) tesseractDuration.stream().mapToLong(Long::longValue).max().orElse(0) / 1000),
|
||||
((float) tesseractDuration.stream().mapToLong(Long::longValue).min().orElse(0) / 1000),
|
||||
(float) imageProcessingDuration.get() / 1000,
|
||||
(float) pdf2ImgDuration.get() / 1000,
|
||||
(float) writingTextDuration.get() / 1000);
|
||||
(float) writingTextDuration.get() / 1000,
|
||||
(float) fontStyleDetectionDuration.get() / 1000);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -36,6 +36,7 @@ public interface FontMetricsFactory {
|
||||
|
||||
PDFont getFont();
|
||||
|
||||
|
||||
HeightAndDescent calculateHeightAndDescent(String text);
|
||||
|
||||
}
|
||||
|
||||
+5
@@ -0,0 +1,5 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.fonts;
|
||||
|
||||
public enum FontStyle {
|
||||
REGULAR, BOLD, ITALIC
|
||||
}
|
||||
+29
-6
@@ -1,6 +1,9 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.fonts;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import org.apache.fontbox.ttf.GlyphData;
|
||||
import org.apache.fontbox.ttf.TTFParser;
|
||||
@@ -12,22 +15,41 @@ import org.apache.pdfbox.pdmodel.font.PDType0Font;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.HeightAndDescent;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import software.amazon.awssdk.services.s3.endpoints.internal.Value;
|
||||
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class Type0FontMetricsFactory implements FontMetricsFactory {
|
||||
|
||||
private final PDType0Font type0Font;
|
||||
private final TrueTypeFont trueTypeFont;
|
||||
|
||||
// for this specific font back-/forward-slashes have a lot of descent screwing up the font size and therefore bold detection. So if we find such a character we ignore its descent.
|
||||
private static final Set<Integer> slashGlyphIds = Set.of(18, 63);
|
||||
|
||||
|
||||
public static Type0FontMetricsFactory regular(PDDocument document) {
|
||||
|
||||
return createFromResource("fonts/cmu-regular.ttf", document);
|
||||
}
|
||||
|
||||
|
||||
public static Type0FontMetricsFactory bold(PDDocument document) {
|
||||
|
||||
return createFromResource("fonts/cmu-bold.ttf", document);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public Type0FontMetricsFactory(PDDocument document) {
|
||||
private static Type0FontMetricsFactory createFromResource(String resourcePath, PDDocument document) {
|
||||
|
||||
try (var in = Thread.currentThread().getContextClassLoader().getResourceAsStream("fonts/cmu-regular.ttf"); var buffer = new RandomAccessReadBuffer(in)) {
|
||||
this.trueTypeFont = new TTFParser().parse(buffer); // since Type0Font can be descendant from any font, we need to remember the original TrueTypeFont for the glyph information
|
||||
this.type0Font = PDType0Font.load(document, this.trueTypeFont, false); // use Type0Font for unicode support
|
||||
try (var in = Thread.currentThread().getContextClassLoader().getResourceAsStream(resourcePath); var buffer = new RandomAccessReadBuffer(in)) {
|
||||
TrueTypeFont trueTypeFont = new TTFParser().parse(buffer); // since Type0Font can be descendant from any font, we need to remember the original TrueTypeFont for the glyph information
|
||||
PDType0Font type0Font = PDType0Font.load(document, trueTypeFont, true); // use Type0Font for unicode support
|
||||
return new Type0FontMetricsFactory(type0Font, trueTypeFont);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -55,8 +77,9 @@ public class Type0FontMetricsFactory implements FontMetricsFactory {
|
||||
if (glyph == null || glyph.getBoundingBox() == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
descent = Math.min(descent, glyph.getYMinimum());
|
||||
if (!slashGlyphIds.contains(glyphId)) {
|
||||
descent = Math.min(descent, glyph.getYMinimum());
|
||||
}
|
||||
height = Math.max(height, glyph.getYMaximum());
|
||||
} catch (Exception e) {
|
||||
log.warn("descent and height of string {} could not be parsed, using average fallback value!", text);
|
||||
|
||||
+158
@@ -0,0 +1,158 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
|
||||
|
||||
import java.util.Comparator;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.apache.commons.math3.ml.clustering.Cluster;
|
||||
import org.apache.commons.math3.ml.clustering.DBSCANClusterer;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrResultToWrite;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.FontStyleDetectionModel;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.TextPositionAndWordImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.scriptdetection.WordImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.FontStyle;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.fonts.Type0FontMetricsFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class FontStyleDetector {
|
||||
|
||||
OcrServiceSettings settings;
|
||||
StrokeWidthCalculator strokeWidthCalculator;
|
||||
|
||||
|
||||
/**
|
||||
* Implementation of the MOBDoB algorithm, refer to the paper here:
|
||||
* <a href="http://mile.ee.iisc.ac.in/publications/softCopy/DocumentAnalysis/Sai_NCVPRIPG2013.pdf">Script Independent Detection of Bold Words in Multi Font-size Documents</a>
|
||||
* <p>
|
||||
* As a high level overview: We cluster all text based on its font size. We determine the cluster with the most words. This is assumed to be regular text.
|
||||
* We then estimate the average stroke width of that cluster by thinning all text to a single pixel and calculating the ratio of remaining pixels.
|
||||
* (<a href="http://www.leptonica.org/papers/conn.pdf">Leptonica Documentation on thinning</a>)
|
||||
* For each word we scale this average strokewidth based on its fontsize compared to the most common fontsize.
|
||||
* Using the scaled strokewidth we do an opening operation.
|
||||
* (<a href="https://en.wikipedia.org/wiki/Opening_(morphology)">Opening (Morphology)</a>).
|
||||
* We then threshold the ratio of remaining pixels to determine whether a word is bold or not.
|
||||
* <p>
|
||||
* I did take some liberties though. Firstly, the paper uses text height without ascender/descender height for the clustering. I'm using the previously implemented font size.
|
||||
* But this is based on text width. Thus, I'm also using the height scaling factor to scale the font size by the text height.
|
||||
* The paper does not describe its clustering algorithm, so I've decided on DBSCAN due to its good runtime and readily available implementation by apache commons math.
|
||||
* Moreover, the paper states that stroke width scales linearly with text height. I've come to the conclusion this is not the case.
|
||||
* It seems it scales with the square root of the text height. Or at least this seemed to give the best results.
|
||||
*/
|
||||
public Map<Integer, List<OcrResultToWrite>> detectBold(List<OcrResult> ocrResults, PDDocument document) {
|
||||
|
||||
FontMetricsFactory fontMetricsFactory = Type0FontMetricsFactory.regular(document);
|
||||
if (!settings.isBoldDetection()) {
|
||||
return OcrResultToWrite.buildOcrResultsToWrite(ocrResults, fontMetricsFactory);
|
||||
}
|
||||
|
||||
Map<Integer, List<OcrResultToWrite>> ocrResultToWritePerPage = new HashMap<>();
|
||||
|
||||
DBSCANClusterer<TextPositionAndWordImage> clusterer = new DBSCANClusterer<>(0.5, 1);
|
||||
|
||||
FontMetricsFactory boldFontMetricsFactory = Type0FontMetricsFactory.bold(document);
|
||||
|
||||
for (OcrResult result : ocrResults) {
|
||||
FontStyleDetectionModel fontStyleDetectionModel = FontStyleDetectionModel.fromOcrResult(result, fontMetricsFactory, settings);
|
||||
|
||||
List<Cluster<TextPositionAndWordImage>> clusters = clusterer.cluster(fontStyleDetectionModel.getTextPositionsAndWordImages());
|
||||
Optional<Cluster<TextPositionAndWordImage>> largestCluster = clusters.stream().max(Comparator.comparingInt(cluster -> cluster.getPoints().size()));
|
||||
|
||||
if (largestCluster.isEmpty()) {
|
||||
insertResultIntoMap(result.image().getPageNumber(), ocrResultToWritePerPage, fontStyleDetectionModel);
|
||||
continue;
|
||||
}
|
||||
|
||||
List<TextPositionAndWordImage> wordsWithMostCommonTextHeight = largestCluster.get().getPoints();
|
||||
|
||||
double standardTextHeight = calculateStandardTextheight(wordsWithMostCommonTextHeight);
|
||||
double regularStrokeWidth = calculateRegularStrokeWidth(wordsWithMostCommonTextHeight);
|
||||
|
||||
for (TextPositionAndWordImage textPositionsAndWordImage : fontStyleDetectionModel.getTextPositionsAndWordImages()) {
|
||||
decideOnFontStyle(textPositionsAndWordImage, regularStrokeWidth, standardTextHeight, boldFontMetricsFactory);
|
||||
}
|
||||
|
||||
insertResultIntoMap(result.image().getPageNumber(), ocrResultToWritePerPage, fontStyleDetectionModel);
|
||||
fontStyleDetectionModel.dispose();
|
||||
}
|
||||
|
||||
log.info("Finished bold detection");
|
||||
return ocrResultToWritePerPage;
|
||||
}
|
||||
|
||||
|
||||
private static double calculateStandardTextheight(List<TextPositionAndWordImage> wordsWithMostCommonTextHeight) {
|
||||
|
||||
return wordsWithMostCommonTextHeight.stream()
|
||||
.map(TextPositionAndWordImage::getWordImage)
|
||||
.mapToDouble(WordImage::getTextHeight)
|
||||
.filter(Double::isFinite)
|
||||
.average()
|
||||
.orElseThrow();
|
||||
}
|
||||
|
||||
|
||||
private double calculateRegularStrokeWidth(List<TextPositionAndWordImage> wordsWithMostCommonTextHeight) {
|
||||
|
||||
return wordsWithMostCommonTextHeight.stream()
|
||||
.mapToDouble(textPositionAndWordImage -> strokeWidthCalculator.calculate(textPositionAndWordImage.getWordImage().getImage()))
|
||||
.filter(Double::isFinite)
|
||||
.average()
|
||||
.orElseThrow();
|
||||
}
|
||||
|
||||
|
||||
private static void insertResultIntoMap(int pageNumber, Map<Integer, List<OcrResultToWrite>> ocrResultToWritePerPage, FontStyleDetectionModel fontStyleDetectionModel) {
|
||||
|
||||
OcrResultToWrite ocrResult = OcrResultToWrite.fromFontStyleDetectionModel(fontStyleDetectionModel);
|
||||
|
||||
ocrResultToWritePerPage.compute(pageNumber, (key, existingList) -> {
|
||||
if (existingList == null) {
|
||||
return List.of(ocrResult);
|
||||
} else {
|
||||
return Stream.concat(existingList.stream(), Stream.of(ocrResult)).toList();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
private void decideOnFontStyle(TextPositionAndWordImage textPositionsAndWordImage,
|
||||
double standardStrokeWidth,
|
||||
double standardTextHeight,
|
||||
FontMetricsFactory boldFontMetricsFactory) {
|
||||
|
||||
double scaledStrokeWidth = scaleStrokeWidthByFontSize(textPositionsAndWordImage, standardStrokeWidth, standardTextHeight);
|
||||
|
||||
if (textPositionsAndWordImage.getWordImage().hasLargerStrokeWidth(scaledStrokeWidth)) {
|
||||
textPositionsAndWordImage.getTextPositionInImage().setFontMetricsFactory(boldFontMetricsFactory);
|
||||
textPositionsAndWordImage.getTextPositionInImage().setFontStyle(FontStyle.BOLD);
|
||||
} else {
|
||||
textPositionsAndWordImage.getTextPositionInImage().setFontStyle(FontStyle.REGULAR);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private static double scaleStrokeWidthByFontSize(TextPositionAndWordImage textPositionsAndWordImage, double standardStrokeWidth, double standardFontSize) {
|
||||
|
||||
double influenceOfFontSize = 1.0; // the paper states that stroke width scales exactly linearly with font size. This did not seem to be true for me. Maybe some of the preprocessing steps are affecting this.
|
||||
double fontsizeScalingFactor = Math.sqrt(textPositionsAndWordImage.getWordImage().getTextHeight() / standardFontSize);
|
||||
return standardStrokeWidth + (influenceOfFontSize * (fontsizeScalingFactor - 1) * standardStrokeWidth);
|
||||
}
|
||||
|
||||
}
|
||||
+57
@@ -0,0 +1,57 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.Sel;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
/**
|
||||
* This code is a good start for detecting italic text, although it has a few issues especially with glyphs which are naturally slanted. E.g. z, 2, 7, /
|
||||
* If we want this maybe we should exclude these glyphs and then it might have less false positives. But in its current state i don't recommend using it.
|
||||
*/
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class ItalicDetector {
|
||||
|
||||
|
||||
static String italicKernel = "ooxxooxxooxxoxxooXxooxxoxxooxxooxxoo";
|
||||
Sel italicSel = Leptonica1.selCreateFromString(italicKernel, 9, 4, "italicKernel");
|
||||
Sel brickSel = Leptonica1.selCreateBrick(3, 4, 1, 2, 1);
|
||||
|
||||
|
||||
public boolean isItalic(Pix pix) {
|
||||
|
||||
Pix preprocessed = preprocess(pix);
|
||||
Pix flipped = Leptonica1.pixFlipLR(null, pix);
|
||||
Pix flippedPreprocessed = preprocess(flipped);
|
||||
Leptonica1.pixFlipLR(flippedPreprocessed, flippedPreprocessed);
|
||||
double pixelDensity = ImageProcessingUtils.calculatePixelDensity(preprocessed);
|
||||
double flippedPixelDensity = ImageProcessingUtils.calculatePixelDensity(flippedPreprocessed);
|
||||
LeptUtils.disposePix(preprocessed);
|
||||
LeptUtils.disposePix(flipped);
|
||||
LeptUtils.disposePix(flippedPreprocessed);
|
||||
return flippedPixelDensity / pixelDensity < 0.85;
|
||||
}
|
||||
|
||||
|
||||
private Pix preprocess(Pix pix) {
|
||||
|
||||
Pix eroded = Leptonica1.pixErode(null, pix, italicSel.getPointer());
|
||||
Pix dilated = Leptonica1.pixDilate(null, eroded, brickSel.getPointer());
|
||||
LeptUtils.disposePix(eroded);
|
||||
return dilated;
|
||||
}
|
||||
|
||||
|
||||
public void dispose() {
|
||||
|
||||
LeptUtils.dispose(italicSel);
|
||||
LeptUtils.dispose(brickSel);
|
||||
}
|
||||
|
||||
}
|
||||
+58
@@ -0,0 +1,58 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.scriptdetection;
|
||||
|
||||
import static net.sourceforge.lept4j.ILeptonica.L_THIN_FG;
|
||||
|
||||
import java.nio.IntBuffer;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.Sela;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
@Service
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class StrokeWidthCalculator {
|
||||
|
||||
Sela thinningSel;
|
||||
|
||||
|
||||
/**
|
||||
* Uses a series of sels to thin all connected lines to a single pixel. Then the pixel ratio is a good estimation of the stroke width in pixels.
|
||||
* <a href="http://www.leptonica.org/papers/conn.pdf">Leptonica Documentation on thinning</a>
|
||||
* Since the baseline is a strokewidth of exactly one, we need to add 1 to the result.
|
||||
*
|
||||
* @param input binarized pix with text on it
|
||||
* @return estimated stroke width in pixels
|
||||
*/
|
||||
public double calculate(Pix input) {
|
||||
|
||||
init();
|
||||
|
||||
Pix thinned = Leptonica1.pixThinConnectedBySet(input, L_THIN_FG, thinningSel, 0);
|
||||
|
||||
IntBuffer thinnedPixelCount = IntBuffer.allocate(1);
|
||||
Leptonica1.pixCountPixels(thinned, thinnedPixelCount, null);
|
||||
|
||||
IntBuffer pixelCount = IntBuffer.allocate(1);
|
||||
Leptonica1.pixCountPixels(input, pixelCount, null);
|
||||
|
||||
LeptUtils.disposePix(thinned);
|
||||
|
||||
return (double) pixelCount.get() / thinnedPixelCount.get() + 1;
|
||||
}
|
||||
|
||||
|
||||
private void init() {
|
||||
|
||||
if (thinningSel == null) {
|
||||
thinningSel = Leptonica1.selaMakeThinSets(1, 0);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
+65
@@ -0,0 +1,65 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.threads;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.NoSuchElementException;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.Setter;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import net.sourceforge.tess4j.TessAPI1;
|
||||
|
||||
/*
|
||||
This just moves the Elements from the GhostScriptOutputListener into the ImageProcessing queue asynchronously
|
||||
*/
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class BlockingQueueFiller extends Thread {
|
||||
|
||||
final BlockingQueue<RenderedPageImageFile> imageInputQueue;
|
||||
final BlockingQueue<UnprocessedImage> imageOutputQueue;
|
||||
|
||||
@Setter
|
||||
boolean allImagesQueued;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
@Override
|
||||
public void run() {
|
||||
|
||||
// Interrupting signals that the image extraction has finished
|
||||
try {
|
||||
while (!allImagesQueued) {
|
||||
final UnprocessedImage image = imageInputQueue.take();
|
||||
try {
|
||||
imageOutputQueue.put(image);
|
||||
} catch (InterruptedException e) {
|
||||
imageOutputQueue.put(image);
|
||||
}
|
||||
}
|
||||
} catch (InterruptedException e) {
|
||||
log.info("All images extracted, emptying processing queue and stopping");
|
||||
}
|
||||
|
||||
// empty the queue
|
||||
try {
|
||||
while (true) {
|
||||
final UnprocessedImage image = imageInputQueue.remove();
|
||||
imageOutputQueue.put(image);
|
||||
}
|
||||
} catch (NoSuchElementException e) {
|
||||
log.debug("No images left in queue, stopping.");
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+122
@@ -0,0 +1,122 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.threads;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.File;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Queue;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class GhostScriptOutputHandler extends Thread {
|
||||
|
||||
static Pattern pageFinishedPattern = Pattern.compile("Page (\\d+)");
|
||||
|
||||
// If the stdError or stdOut buffer of a thread is not being emptied it might lock the process in case of errors, so we need to empty both streams to prevent a deadlock.
|
||||
// Since both need to read simultaneously we need to implement the readers as separate threads.
|
||||
|
||||
final InputStream is;
|
||||
final String processName;
|
||||
final Type type;
|
||||
|
||||
final Map<Integer, RenderedPageImageFile> pagesToProcess;
|
||||
final BlockingQueue<RenderedPageImageFile> renderedPageImageFileOutput;
|
||||
|
||||
int currentPageNumber;
|
||||
|
||||
|
||||
public static GhostScriptOutputHandler errorHandler(InputStream is) {
|
||||
|
||||
return new GhostScriptOutputHandler(is, "GS", Type.ERROR, null, null);
|
||||
}
|
||||
|
||||
|
||||
public static GhostScriptOutputHandler stdOut(InputStream is,
|
||||
Map<Integer, RenderedPageImageFile> pagesToProcess,
|
||||
BlockingQueue<RenderedPageImageFile> renderedPageImageFileOutput) {
|
||||
|
||||
return new GhostScriptOutputHandler(is, "GS", Type.STD_OUT, pagesToProcess, renderedPageImageFileOutput);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void run() {
|
||||
|
||||
try (InputStreamReader isr = new InputStreamReader(is); BufferedReader br = new BufferedReader(isr)) {
|
||||
|
||||
String line;
|
||||
while (true) {
|
||||
line = br.readLine();
|
||||
|
||||
if (line == null) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (type.equals(Type.ERROR)) {
|
||||
log.error(processName + "_" + type.name() + ">" + line);
|
||||
} else {
|
||||
log.debug(processName + "_" + type.name() + ">" + line);
|
||||
addProcessedImageToQueue(line);
|
||||
}
|
||||
}
|
||||
}
|
||||
is.close();
|
||||
if (type.equals(Type.STD_OUT)) {
|
||||
queueFinishedPage(currentPageNumber);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
private void addProcessedImageToQueue(String line) {
|
||||
|
||||
/*
|
||||
Ghostscript prints the pageNumber it is currently working on, so we remember the current page and queue it as soon as the next comes in.
|
||||
*/
|
||||
Matcher pageNumberMatcher = pageFinishedPattern.matcher(line);
|
||||
if (pageNumberMatcher.find()) {
|
||||
int pageNumber = Integer.parseInt(pageNumberMatcher.group(1));
|
||||
|
||||
if (currentPageNumber == 0) {
|
||||
currentPageNumber = pageNumber;
|
||||
return;
|
||||
}
|
||||
|
||||
queueFinishedPage(currentPageNumber);
|
||||
currentPageNumber = pageNumber;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void queueFinishedPage(int pageNumber) {
|
||||
|
||||
var imageFile = this.pagesToProcess.get(pageNumber);
|
||||
if (imageFile == null) {
|
||||
throw new IllegalArgumentException(String.format("Page number %d does not exist in this thread. It only has pagenumbers %s", pageNumber, pagesToProcess.keySet()));
|
||||
}
|
||||
assert new File(imageFile.absoluteFilePath()).isFile();
|
||||
renderedPageImageFileOutput.add(imageFile);
|
||||
}
|
||||
|
||||
|
||||
public enum Type {
|
||||
ERROR,
|
||||
STD_OUT
|
||||
}
|
||||
|
||||
}
|
||||
+26
-22
@@ -5,12 +5,11 @@ import java.util.List;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.io.MemoryUsageSetting;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.ImageStreamEngine;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.OcrProgressLogger;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
|
||||
@@ -26,6 +25,7 @@ import lombok.experimental.FieldDefaults;
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class ImageExtractionThread extends Thread {
|
||||
|
||||
static double FULL_PAGE_IMAGE_THRESHOLD = 0.99;
|
||||
static double IMAGE_ALIGNMENT_THRESHOLD = 1;
|
||||
|
||||
int id;
|
||||
@@ -37,9 +37,10 @@ public class ImageExtractionThread extends Thread {
|
||||
OcrServiceSettings settings;
|
||||
|
||||
// output is written to these lists
|
||||
BlockingQueue<OcrImage> imageOutputQueue;
|
||||
BlockingQueue<UnprocessedImage> imageProcessingQueue;
|
||||
List<Integer> stitchedPageNumbers;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
@Override
|
||||
public void run() {
|
||||
@@ -48,28 +49,28 @@ public class ImageExtractionThread extends Thread {
|
||||
for (Integer pageIndex : pageIndices) {
|
||||
try (PDDocument document = Loader.loadPDF(documentFile)) { // load new PDDocument for thread safety, also keeps RAM usage low.
|
||||
timestamp = System.currentTimeMillis();
|
||||
List<ExtractedOcrImage> extractedOcrImages = getExtractedOcrImages(pageIndex, document);
|
||||
List<ExtractedImage> extractedImages = getExtractedImages(pageIndex, document);
|
||||
stats.increaseImageExtraction(id, System.currentTimeMillis() - timestamp);
|
||||
if (extractedOcrImages.isEmpty()) {
|
||||
if (extractedImages.isEmpty()) {
|
||||
logger.logPageSkipped(pageIndex);
|
||||
}
|
||||
|
||||
if (checkForStitchedImages(extractedOcrImages)) {
|
||||
if (checkForFullPageOrStitchedImages(extractedImages, document.getPage(pageIndex - 1))) {
|
||||
stitchedPageNumbers.add(pageIndex);
|
||||
logger.addImagesToProcess(pageIndex, 0);
|
||||
continue;
|
||||
}
|
||||
|
||||
for (ExtractedOcrImage image : extractedOcrImages) {
|
||||
imageOutputQueue.put(image);
|
||||
logger.addImagesToProcess(image.getPageNumber(), image.getNumberOnPage());
|
||||
for (ExtractedImage image : extractedImages) {
|
||||
imageProcessingQueue.put(image);
|
||||
logger.addImagesToProcess(image.pageNumber(), image.numberOnPage());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private List<ExtractedOcrImage> getExtractedOcrImages(Integer pageIndex, PDDocument document) {
|
||||
private List<ExtractedImage> getExtractedImages(Integer pageIndex, PDDocument document) {
|
||||
|
||||
PDPage page = document.getPage(pageIndex - 1);
|
||||
ImageStreamEngine imageStreamEngine = new ImageStreamEngine(settings);
|
||||
@@ -79,22 +80,25 @@ public class ImageExtractionThread extends Thread {
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private boolean checkForStitchedImages(List<ExtractedOcrImage> imagesOnCurrentPage) {
|
||||
private boolean checkForFullPageOrStitchedImages(List<ExtractedImage> imagesOnCurrentPage, PDPage page) {
|
||||
|
||||
if (imagesOnCurrentPage.size() <= 1) {
|
||||
if (imagesOnCurrentPage.isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
//checking for intersections or direct alignment of images
|
||||
ExtractedOcrImage[] imageOnPagesArray = new ExtractedOcrImage[imagesOnCurrentPage.size()];
|
||||
int index = 0;
|
||||
for (ExtractedOcrImage imageOnPage : imagesOnCurrentPage) {
|
||||
imageOnPagesArray[index] = imageOnPage;
|
||||
index++;
|
||||
for (ExtractedImage imageOnPage : imagesOnCurrentPage) {
|
||||
if (imageOnPage.width() > FULL_PAGE_IMAGE_THRESHOLD * page.getCropBox().getWidth() && imageOnPage.height() > FULL_PAGE_IMAGE_THRESHOLD * page.getCropBox()
|
||||
.getHeight()) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
for (int j = 0; j < imageOnPagesArray.length; j++) {
|
||||
for (int i = j + 1; i < imageOnPagesArray.length; i++) {
|
||||
if (imageOnPagesArray[j].getImageCoordinatesInInitialUserSpace().aligns(imageOnPagesArray[i].getImageCoordinatesInInitialUserSpace(), IMAGE_ALIGNMENT_THRESHOLD)) {
|
||||
|
||||
//checking for intersections or direct alignment of images
|
||||
for (int j = 0; j < imagesOnCurrentPage.size(); j++) {
|
||||
for (int i = j + 1; i < imagesOnCurrentPage.size(); i++) {
|
||||
if (imagesOnCurrentPage.get(j)
|
||||
.getImageCoordinatesInInitialUserSpace()
|
||||
.aligns(imagesOnCurrentPage.get(i).getImageCoordinatesInInitialUserSpace(), IMAGE_ALIGNMENT_THRESHOLD)) {
|
||||
// TODO: see if we can stitch aligning images using BufferedImage and skip the gs conversion entirely
|
||||
return true;
|
||||
}
|
||||
|
||||
+250
@@ -0,0 +1,250 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.threads;
|
||||
|
||||
import static net.sourceforge.tess4j.ITessAPI.TRUE;
|
||||
|
||||
import java.nio.FloatBuffer;
|
||||
import java.nio.IntBuffer;
|
||||
import java.util.NoSuchElementException;
|
||||
import java.util.concurrent.BlockingQueue;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedOcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.OcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.PageInformation;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageImageFile;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.RenderedPageOcrImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.model.UnprocessedImage;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
|
||||
import com.knecon.fforesight.service.ocr.processor.settings.OcrServiceSettings;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.ImageProcessingUtils;
|
||||
import com.sun.jna.ptr.PointerByReference;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.Setter;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import net.sourceforge.lept4j.L_Kernel;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
import net.sourceforge.tess4j.ITessAPI;
|
||||
import net.sourceforge.tess4j.TessAPI1;
|
||||
|
||||
/*
|
||||
* This thread does all the image processing. There should only be one, since Leptonica is not thread safe.
|
||||
*/
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class ImageProcessingThread extends Thread {
|
||||
|
||||
final BlockingQueue<UnprocessedImage> imageInputQueue;
|
||||
final BlockingQueue<OcrImage> imageOutputQueue;
|
||||
final ITessAPI.TessBaseAPI detectionScriptHandle = initDetectionScriptHandle();
|
||||
final L_Kernel gaussianKernel = Leptonica1.makeGaussianKernel(2, 2, 1.2f, 1);
|
||||
final Statistics stats;
|
||||
final OcrServiceSettings settings;
|
||||
final PDDocument document;
|
||||
|
||||
@Setter
|
||||
boolean allImagesExtracted;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
@Override
|
||||
public void run() {
|
||||
|
||||
try {
|
||||
while (!allImagesExtracted) {
|
||||
final UnprocessedImage image = imageInputQueue.take();
|
||||
var ocrImage = this.process(image);
|
||||
try {
|
||||
imageOutputQueue.put(ocrImage);
|
||||
} catch (InterruptedException e) {
|
||||
imageOutputQueue.put(ocrImage);
|
||||
}
|
||||
}
|
||||
} catch (InterruptedException e) {
|
||||
log.info("All images extracted, emptying processing queue and stopping");
|
||||
}
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
final UnprocessedImage image = imageInputQueue.remove();
|
||||
OcrImage ocrImage = this.process(image);
|
||||
imageOutputQueue.put(ocrImage);
|
||||
}
|
||||
} catch (NoSuchElementException e) {
|
||||
log.debug("No images left in processing queue, stopping.");
|
||||
}
|
||||
|
||||
TessAPI1.TessBaseAPIEnd(this.detectionScriptHandle);
|
||||
TessAPI1.TessBaseAPIDelete(this.detectionScriptHandle);
|
||||
LeptUtils.dispose(gaussianKernel);
|
||||
}
|
||||
|
||||
|
||||
private OcrImage process(UnprocessedImage unprocessedImage) {
|
||||
|
||||
long timestamp = System.currentTimeMillis();
|
||||
|
||||
OcrImage ocrImage;
|
||||
if (unprocessedImage instanceof ExtractedImage extractedImage) {
|
||||
ocrImage = processExtractedImage(extractedImage);
|
||||
} else if (unprocessedImage instanceof RenderedPageImageFile renderedPageImageFile) {
|
||||
ocrImage = processRenderedPageImageFile(renderedPageImageFile);
|
||||
} else {
|
||||
throw new UnsupportedOperationException(String.format("Class %s is not supported!", unprocessedImage.getClass()));
|
||||
}
|
||||
|
||||
stats.increaseImageProcessing(System.currentTimeMillis() - timestamp);
|
||||
|
||||
return ocrImage;
|
||||
}
|
||||
|
||||
|
||||
private OcrImage processRenderedPageImageFile(RenderedPageImageFile renderedPageImageFile) {
|
||||
|
||||
Pix pix = processPix(renderedPageImageFile.asPix(), settings.getDpi(), settings.getDpi());
|
||||
|
||||
int orientDegree = detectOrientation(pix, settings.getDpi(), detectionScriptHandle);
|
||||
Pix rotatedPix = ImageProcessingUtils.deRotatePix(orientDegree, pix);
|
||||
|
||||
OcrImage ocrImage = new RenderedPageOcrImage(pix.h,
|
||||
pix.w,
|
||||
PageInformation.fromPDPage(renderedPageImageFile.pageNumber(), document.getPage(renderedPageImageFile.pageNumber() - 1)),
|
||||
rotatedPix,
|
||||
orientDegree);
|
||||
|
||||
if (pix != rotatedPix) {
|
||||
LeptUtils.disposePix(pix);
|
||||
}
|
||||
|
||||
return ocrImage;
|
||||
}
|
||||
|
||||
|
||||
private OcrImage processExtractedImage(ExtractedImage extractedImage) {
|
||||
|
||||
float imageDPI = Math.abs(extractedImage.image().getWidth() / (extractedImage.ctm().getScalingFactorX() / 72));
|
||||
|
||||
Pix pix = processPix(extractedImage.asPix(), imageDPI, settings.getDpi());
|
||||
|
||||
int orientDegree = detectOrientation(pix, settings.getDpi(), detectionScriptHandle);
|
||||
Pix rotatedPix = ImageProcessingUtils.deRotatePix(orientDegree, pix);
|
||||
|
||||
OcrImage ocrImage = new ExtractedOcrImage(extractedImage.pageNumber(),
|
||||
extractedImage.numberOnPage(),
|
||||
extractedImage.height(),
|
||||
extractedImage.width(),
|
||||
extractedImage.ctm(),
|
||||
rotatedPix,
|
||||
pix.h,
|
||||
pix.w,
|
||||
orientDegree);
|
||||
|
||||
if (pix != rotatedPix) {
|
||||
LeptUtils.disposePix(pix);
|
||||
}
|
||||
return ocrImage;
|
||||
}
|
||||
|
||||
|
||||
public int detectOrientation(Pix pix, int dpi, ITessAPI.TessBaseAPI detectionScriptHandle) {
|
||||
|
||||
TessAPI1.TessBaseAPISetImage2(detectionScriptHandle, pix);
|
||||
TessAPI1.TessBaseAPISetSourceResolution(detectionScriptHandle, dpi);
|
||||
|
||||
IntBuffer orientationDegreeResultBuffer;
|
||||
FloatBuffer orientationDegreeConfidenceBuffer;
|
||||
PointerByReference scriptureNameBuffer;
|
||||
FloatBuffer scriptureConfidenceBuffer;
|
||||
|
||||
orientationDegreeResultBuffer = IntBuffer.allocate(1);
|
||||
orientationDegreeConfidenceBuffer = FloatBuffer.allocate(1);
|
||||
scriptureNameBuffer = new PointerByReference(); // Is this memory being freed?
|
||||
scriptureConfidenceBuffer = FloatBuffer.allocate(1);
|
||||
|
||||
int orientationDegree = 0;
|
||||
int result = TessAPI1.TessBaseAPIDetectOrientationScript(detectionScriptHandle,
|
||||
orientationDegreeResultBuffer,
|
||||
orientationDegreeConfidenceBuffer,
|
||||
scriptureNameBuffer,
|
||||
scriptureConfidenceBuffer);
|
||||
if (result == TRUE && orientationDegreeConfidenceBuffer.get() > settings.getMinRotationConfidence()) {
|
||||
orientationDegree = orientationDegreeResultBuffer.get();
|
||||
}
|
||||
|
||||
TessAPI1.TessBaseAPIClear(detectionScriptHandle);
|
||||
|
||||
return orientationDegree;
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private Pix processPix(Pix pix, float imageDpi, int targetDpi) {
|
||||
|
||||
Pix grayScale;
|
||||
Pix scaledUp;
|
||||
Pix gaussian;
|
||||
Pix binarized;
|
||||
|
||||
//convert to grayscale
|
||||
if (pix.d == 8) {
|
||||
grayScale = pix;
|
||||
} else if (pix.d == 32) {
|
||||
grayScale = Leptonica1.pixConvertRGBToGrayFast(pix);
|
||||
} else if (pix.d == 1) {
|
||||
grayScale = Leptonica1.pixConvert1To8(null, pix, (byte) 0, (byte) 255);
|
||||
} else {
|
||||
throw new UnsupportedOperationException(String.format("Unknown pix format with bpp of %d", pix.d));
|
||||
}
|
||||
|
||||
// scale up
|
||||
float targetFactor = targetDpi / imageDpi;
|
||||
if (targetFactor > 2.1) {
|
||||
scaledUp = Leptonica1.pixScaleGray4xLI(grayScale);
|
||||
} else if (targetFactor > 1.1) {
|
||||
scaledUp = Leptonica1.pixScaleGray2xLI(grayScale);
|
||||
} else {
|
||||
scaledUp = grayScale;
|
||||
}
|
||||
|
||||
// remove noise and prep for Otsu
|
||||
gaussian = Leptonica1.pixConvolve(scaledUp, gaussianKernel, 8, 1);
|
||||
|
||||
// Threshold to binary
|
||||
if (pix.w < 100 || pix.h < 100) {
|
||||
binarized = Leptonica1.pixThresholdToBinary(gaussian, 170);
|
||||
} else {
|
||||
binarized = Leptonica1.pixOtsuThreshOnBackgroundNorm(gaussian, null, 50, 50, 165, 10, 100, 5, 5, 0.2f, null);
|
||||
|
||||
if (binarized == null) { // Sometimes Otsu just fails, then we binarize directly
|
||||
binarized = Leptonica1.pixThresholdToBinary(gaussian, 170);
|
||||
}
|
||||
}
|
||||
|
||||
LeptUtils.disposePix(pix);
|
||||
LeptUtils.disposePix(grayScale);
|
||||
LeptUtils.disposePix(scaledUp);
|
||||
LeptUtils.disposePix(gaussian);
|
||||
|
||||
return binarized;
|
||||
}
|
||||
|
||||
|
||||
|
||||
private static ITessAPI.TessBaseAPI initDetectionScriptHandle() {
|
||||
|
||||
ITessAPI.TessBaseAPI handle = TessAPI1.TessBaseAPICreate();
|
||||
String datapath = System.getenv("TESSDATA_PREFIX");
|
||||
TessAPI1.TessBaseAPIInit3(handle, datapath, "osd");
|
||||
TessAPI1.TessBaseAPISetVariable(handle, "debug_file", "/dev/null");
|
||||
return handle;
|
||||
}
|
||||
|
||||
}
|
||||
+10
-46
@@ -1,6 +1,10 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.threads;
|
||||
|
||||
import static net.sourceforge.tess4j.ITessAPI.TRUE;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPICreate;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIInit1;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPISetPageSegMode;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPISetVariable;
|
||||
|
||||
import java.io.File;
|
||||
import java.nio.FloatBuffer;
|
||||
@@ -16,6 +20,7 @@ import com.knecon.fforesight.service.ocr.processor.model.OcrResult;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.OcrProgressLogger;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.Statistics;
|
||||
import com.knecon.fforesight.service.ocr.processor.utils.Tesseract2;
|
||||
import com.sun.jna.StringArray;
|
||||
import com.sun.jna.ptr.PointerByReference;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
@@ -43,7 +48,6 @@ public class OCRThread extends Thread {
|
||||
Statistics stats;
|
||||
OcrServiceSettings settings;
|
||||
Tesseract2 instance;
|
||||
ITessAPI.TessBaseAPI detectionScriptHandle;
|
||||
|
||||
|
||||
public OCRThread(int id,
|
||||
@@ -62,7 +66,6 @@ public class OCRThread extends Thread {
|
||||
this.stats = stats;
|
||||
this.settings = settings;
|
||||
this.instance = createInstance(settings);
|
||||
this.detectionScriptHandle = initDetectionScriptHandle();
|
||||
}
|
||||
|
||||
|
||||
@@ -87,8 +90,9 @@ public class OCRThread extends Thread {
|
||||
this.process(image);
|
||||
}
|
||||
} catch (NoSuchElementException e) {
|
||||
log.debug("Processed all Images, finishing.");
|
||||
log.debug("Executed tesseract on all Images, finishing.");
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -100,12 +104,8 @@ public class OCRThread extends Thread {
|
||||
|
||||
int psm = settings.getPsmOverride() < 0 ? image.getOptimalPageSegmentationMode() : settings.getPsmOverride();
|
||||
|
||||
int orientDegree = detectOrientation(image);
|
||||
image.setRotationDegrees(orientDegree);
|
||||
Pix rotatedPix = image.getRotatedPix();
|
||||
executeTesseract(psm, image.getDpi(), rotatedPix, tesseractOutputFileName);
|
||||
executeTesseract(psm, image.getDpi(), image.getPix(), tesseractOutputFileName);
|
||||
image.destroyPix();
|
||||
LeptUtils.disposePix(rotatedPix);
|
||||
|
||||
results.add(OcrResult.create(image, tesseractOutputFileName));
|
||||
logger.logImageFinished(image, psm);
|
||||
@@ -113,50 +113,14 @@ public class OCRThread extends Thread {
|
||||
}
|
||||
|
||||
|
||||
public int detectOrientation(OcrImage image) {
|
||||
|
||||
TessAPI1.TessBaseAPISetImage2(detectionScriptHandle, image.getPix());
|
||||
TessAPI1.TessBaseAPISetSourceResolution(detectionScriptHandle, image.getDpi());
|
||||
|
||||
IntBuffer orient_degB = IntBuffer.allocate(1);
|
||||
FloatBuffer orient_confB = FloatBuffer.allocate(1);
|
||||
PointerByReference script_nameB = new PointerByReference();
|
||||
FloatBuffer script_confB = FloatBuffer.allocate(1);
|
||||
|
||||
int orient_deg = 0;
|
||||
int result = TessAPI1.TessBaseAPIDetectOrientationScript(detectionScriptHandle, orient_degB, orient_confB, script_nameB, script_confB);
|
||||
if (result == TRUE) {
|
||||
orient_deg = orient_degB.get();
|
||||
}
|
||||
TessAPI1.TessBaseAPIClear(detectionScriptHandle);
|
||||
|
||||
return orient_deg;
|
||||
}
|
||||
|
||||
|
||||
private static ITessAPI.TessBaseAPI initDetectionScriptHandle() {
|
||||
|
||||
ITessAPI.TessBaseAPI handle = TessAPI1.TessBaseAPICreate();
|
||||
String datapath = System.getenv("TESSDATA_PREFIX");
|
||||
TessAPI1.TessBaseAPIInit3(handle, datapath, "osd");
|
||||
|
||||
return handle;
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void executeTesseract(int psm, int dpi, Pix pix, String tesseractOutputFileName) {
|
||||
|
||||
if (settings.isDebug()) {
|
||||
String[] a = tesseractOutputFileName.split("/");
|
||||
String folder = "/tmp/pixs/" + a[a.length - 3];
|
||||
new File(folder).mkdirs();
|
||||
Leptonica1.pixWrite(folder + "/pix_" + a[a.length - 1] + ".png", pix, 3);
|
||||
}
|
||||
|
||||
Leptonica1.pixWrite(tesseractOutputFileName + ".tiff", pix, 5); // write the used image for later bold detection
|
||||
instance.setVariable("user_defined_dpi", String.valueOf(dpi));
|
||||
instance.setPageSegMode(psm);
|
||||
instance.createDocumentsWithResults(pix, null, tesseractOutputFileName, List.of(ITesseract.RenderedFormat.HOCR), ITessAPI.TessPageIteratorLevel.RIL_BLOCK);
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
-55
@@ -1,55 +0,0 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.service.threads;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class ProcessIOLogger extends Thread {
|
||||
|
||||
// If the stdError or stdOut buffer of a thread is not being emptied it might lock the process in case of errors, so we need to empty both streams to prevent a deadlock.
|
||||
// Since both need to read simultaneously we need to implement the readers as separate threads.
|
||||
|
||||
InputStream is;
|
||||
String processName;
|
||||
Type type;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void run() {
|
||||
|
||||
try (InputStreamReader isr = new InputStreamReader(is); BufferedReader br = new BufferedReader(isr)) {
|
||||
|
||||
String line;
|
||||
while (true) {
|
||||
line = br.readLine();
|
||||
|
||||
if (line == null) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (type.equals(Type.ERROR)) {
|
||||
log.error(processName + "_" + type.name() + ">" + line);
|
||||
} else {
|
||||
log.debug(processName + "_" + type.name() + ">" + line);
|
||||
}
|
||||
}
|
||||
}
|
||||
is.close();
|
||||
}
|
||||
|
||||
|
||||
public enum Type {
|
||||
ERROR,
|
||||
STD_OUT
|
||||
}
|
||||
|
||||
}
|
||||
+4
-1
@@ -14,14 +14,17 @@ public class OcrServiceSettings {
|
||||
|
||||
int ocrThreadCount = 4; // Number of OCR threads
|
||||
int imageExtractThreadCount = 2; // Number of image extraction threads
|
||||
int gsProcessCount = 2; // Number of Ghostscript processes
|
||||
int gsProcessCount = 1; // Number of Ghostscript processes
|
||||
int dpi = 300; // Target DPI for binarized images
|
||||
int psmOverride = -1; // Overrides the page segmentation mode if > 0
|
||||
int minImageHeight = 20; // Minimum height for images to be processed
|
||||
int minImageWidth = 20; // Minimum width for images to be processed
|
||||
float minRotationConfidence = 2; // Sets a lower bound for the confidence rating for rotated pages.
|
||||
boolean debug; // If true, overlays OCR images with a grid and draws word bounding boxes
|
||||
boolean removeWatermark; // If true, watermarks will be removed
|
||||
String languages = "deu+eng"; // Defines languages loaded into Tesseract as 3-char codes, additional languages must also be installed in the docker environment
|
||||
COSName ocrMarkedContentTag = COSName.getPDFName("KNECON_OCR");
|
||||
boolean boldDetection = true; // if true, bold detection will be attempted
|
||||
double boldThreshold = 0.5; // Words are opened with a brick of average stroke width, if the ratio of remaining pixels is higher the word is determined bold.
|
||||
|
||||
}
|
||||
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.utils;
|
||||
|
||||
import java.awt.AlphaComposite;
|
||||
import java.awt.Color;
|
||||
import java.awt.Graphics;
|
||||
import java.awt.Graphics2D;
|
||||
import java.awt.Transparency;
|
||||
import java.awt.image.BufferedImage;
|
||||
import java.nio.IntBuffer;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceGray;
|
||||
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.model.ExtractedImage;
|
||||
import com.sun.jna.ptr.PointerByReference;
|
||||
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.UtilityClass;
|
||||
import net.sourceforge.lept4j.L_Kernel;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.lept4j.util.LeptUtils;
|
||||
|
||||
@UtilityClass
|
||||
public class ImageProcessingUtils {
|
||||
|
||||
public BufferedImage convertToDeviceColorSpace(ExtractedImage extractedImage) {
|
||||
|
||||
BufferedImage image;
|
||||
if (extractedImage.colorSpace() instanceof PDDeviceRGB || extractedImage.colorSpace() instanceof PDDeviceGray) {
|
||||
image = extractedImage.image();
|
||||
} else {
|
||||
BufferedImage pdfImage = extractedImage.image();
|
||||
image = new BufferedImage(pdfImage.getWidth(), pdfImage.getHeight(), BufferedImage.TYPE_BYTE_GRAY);
|
||||
Graphics g = image.getGraphics();
|
||||
g.drawImage(pdfImage, 0, 0, null);
|
||||
g.dispose();
|
||||
}
|
||||
return image;
|
||||
}
|
||||
|
||||
|
||||
public Pix deRotatePix(int orientDegree, Pix pix) {
|
||||
|
||||
return switch (360 - orientDegree) {
|
||||
case 90 -> Leptonica1.pixRotateOrth(pix, 1);
|
||||
case 180 -> Leptonica1.pixRotateOrth(pix, 2);
|
||||
case 270 -> Leptonica1.pixRotateOrth(pix, 3);
|
||||
default -> pix;
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
public static void setAlphaChannelToWhite(BufferedImage image) {
|
||||
|
||||
if (image.getTransparency() == Transparency.TRANSLUCENT) {
|
||||
// NOTE: For BITMASK images, the color model is likely IndexColorModel,
|
||||
// and this model will contain the "real" color of the transparent parts
|
||||
// which is likely a better fit than unconditionally setting it to white.
|
||||
|
||||
// Fill background with white
|
||||
Graphics2D graphics = image.createGraphics();
|
||||
try {
|
||||
graphics.setComposite(AlphaComposite.DstOver); // Set composite rules to paint "behind"
|
||||
graphics.setPaint(Color.WHITE);
|
||||
graphics.fillRect(0, 0, image.getWidth(), image.getHeight());
|
||||
} finally {
|
||||
graphics.dispose();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public static double calculatePixelDensity(Pix pix) {
|
||||
|
||||
IntBuffer pixelCount = IntBuffer.allocate(1);
|
||||
int result = Leptonica1.pixCountPixels(pix, pixelCount, null);
|
||||
if (result == 0) {
|
||||
return (double) pixelCount.get() / (pix.h * pix.w);
|
||||
} else {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
+73
@@ -0,0 +1,73 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.utils;
|
||||
|
||||
import lombok.experimental.UtilityClass;
|
||||
import net.sourceforge.lept4j.L_Kernel;
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
|
||||
@UtilityClass
|
||||
public class KernelUtils {
|
||||
|
||||
/*
|
||||
-1, -1, -1
|
||||
-1, 8, -1
|
||||
-1, -1, -1
|
||||
*/
|
||||
public L_Kernel createFullLaplacianKernel() {
|
||||
|
||||
L_Kernel laplacianKernel = Leptonica1.kernelCreate(3, 3);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 0, 0, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 0, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 0, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 0, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 0, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, 8);
|
||||
return laplacianKernel;
|
||||
}
|
||||
|
||||
/*
|
||||
0, 0, -1, 0, 0
|
||||
0, -1, -1, -1, 0
|
||||
-1, -1, 12, -1, -1
|
||||
0, -1, -1, -1, 0
|
||||
0, 0, -1, 0, 0
|
||||
*/
|
||||
public L_Kernel createLaplacianKernel5x5() {
|
||||
|
||||
L_Kernel laplacianKernel = Leptonica1.kernelCreate(5, 5);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 0, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 3, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 0, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 3, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 4, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 3, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 3, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 3, 3, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 4, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 2, 12);
|
||||
return laplacianKernel;
|
||||
}
|
||||
|
||||
/*
|
||||
0, -1, 0
|
||||
-1, 4, -1
|
||||
0, -1, 0
|
||||
*/
|
||||
public L_Kernel createLaplacianKernel() {
|
||||
|
||||
L_Kernel laplacianKernel = Leptonica1.kernelCreate(3, 3);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 0, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 0, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 2, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 2, 1, -1);
|
||||
Leptonica1.kernelSetElement(laplacianKernel, 1, 1, 4);
|
||||
return laplacianKernel;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
+39
-9
@@ -1,5 +1,25 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.utils;
|
||||
|
||||
import static net.sourceforge.tess4j.ITessAPI.TRUE;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIDelete;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIEnd;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIGetIterator;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIGetStringVariable;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIMeanTextConf;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessBaseAPIProcessPage;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessDeleteResultRenderer;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessHOcrRendererCreate;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorBegin;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorBoundingBox;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessPageIteratorNext;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorConfidence;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorDelete;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorGetPageIterator;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultIteratorGetUTF8Text;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererBeginDocument;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererEndDocument;
|
||||
import static net.sourceforge.tess4j.TessAPI1.TessResultRendererInsert;
|
||||
|
||||
import java.awt.Rectangle;
|
||||
import java.nio.IntBuffer;
|
||||
import java.util.ArrayList;
|
||||
@@ -9,17 +29,19 @@ import com.sun.jna.Pointer;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
import net.sourceforge.tess4j.ITessAPI;
|
||||
import net.sourceforge.tess4j.OCRResult;
|
||||
import net.sourceforge.tess4j.TessAPI1;
|
||||
import net.sourceforge.tess4j.Tesseract1;
|
||||
import net.sourceforge.tess4j.Tesseract;
|
||||
import net.sourceforge.tess4j.TesseractException;
|
||||
import net.sourceforge.tess4j.Word;
|
||||
|
||||
@Slf4j
|
||||
public class Tesseract2 extends Tesseract1 {
|
||||
/**
|
||||
* Overriden version only so I can use Tesseract1 with Pixs instead of BufferedImages. All Functions are copied and then the BufferedImage -> Pix conversion deleted.
|
||||
*/ public class Tesseract2 extends Tesseract {
|
||||
|
||||
|
||||
private int createDocuments(Pix pix, String filename, TessResultRenderer renderer) {
|
||||
private int createDocuments(Pix pix, String filename, ITessAPI.TessResultRenderer renderer) {
|
||||
|
||||
String title = TessBaseAPIGetStringVariable(getHandle(), DOCUMENT_TITLE);
|
||||
TessResultRendererBeginDocument(renderer, title);
|
||||
@@ -59,7 +81,7 @@ public class Tesseract2 extends Tesseract1 {
|
||||
try {
|
||||
for (int i = 0; i < pixs.length; i++) {
|
||||
try {
|
||||
TessResultRenderer renderer = createRenderers(outputbases[i], formats);
|
||||
ITessAPI.TessResultRenderer renderer = createRenderers(outputbases[i], formats);
|
||||
int meanTextConfidence = createDocuments(pixs[i], filenames[i], renderer);
|
||||
TessDeleteResultRenderer(renderer);
|
||||
List<Word> words = meanTextConfidence > 0 ? getRecognizedWords(pageIteratorLevel) : new ArrayList<Word>();
|
||||
@@ -82,8 +104,8 @@ public class Tesseract2 extends Tesseract1 {
|
||||
List<Word> words = new ArrayList<>();
|
||||
|
||||
try {
|
||||
TessResultIterator ri = TessBaseAPIGetIterator(getHandle());
|
||||
TessPageIterator pi = TessResultIteratorGetPageIterator(ri);
|
||||
ITessAPI.TessResultIterator ri = TessBaseAPIGetIterator(getHandle());
|
||||
ITessAPI.TessPageIterator pi = TessResultIteratorGetPageIterator(ri);
|
||||
TessPageIteratorBegin(pi);
|
||||
|
||||
do {
|
||||
@@ -116,9 +138,9 @@ public class Tesseract2 extends Tesseract1 {
|
||||
}
|
||||
|
||||
|
||||
private TessResultRenderer createRenderers(String outputbase, List<RenderedFormat> formats) {
|
||||
private ITessAPI.TessResultRenderer createRenderers(String outputbase, List<RenderedFormat> formats) {
|
||||
|
||||
TessResultRenderer renderer = null;
|
||||
ITessAPI.TessResultRenderer renderer = null;
|
||||
|
||||
for (RenderedFormat format : formats) {
|
||||
switch (format) {
|
||||
@@ -135,4 +157,12 @@ public class Tesseract2 extends Tesseract1 {
|
||||
return renderer;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
protected void dispose() {
|
||||
|
||||
TessBaseAPIEnd(getHandle());
|
||||
TessBaseAPIDelete(getHandle());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+1
-1
@@ -20,7 +20,7 @@ class Type0FontMetricsFactoryTest {
|
||||
public void testStringWidth() {
|
||||
|
||||
try (PDDocument document = Loader.loadPDF(new File(Type0FontMetricsFactoryTest.class.getClassLoader().getResource("InvisibleText.pdf").getPath()))) {
|
||||
Type0FontMetricsFactory metricsFactory = new Type0FontMetricsFactory(document);
|
||||
Type0FontMetricsFactory metricsFactory = Type0FontMetricsFactory.regular(document);
|
||||
FontMetrics fontMetrics = metricsFactory.calculateMetrics("deine mutter", 100, 50);
|
||||
}
|
||||
|
||||
|
||||
+36
@@ -0,0 +1,36 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.utils;
|
||||
|
||||
import static net.sourceforge.lept4j.ILeptonica.IFF_PNG;
|
||||
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Disabled;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import net.sourceforge.lept4j.Leptonica1;
|
||||
import net.sourceforge.lept4j.Pix;
|
||||
|
||||
@Disabled
|
||||
class ImageProcessingUtilsTest {
|
||||
|
||||
@BeforeEach
|
||||
public void loadLeptonica() {
|
||||
|
||||
System.setProperty("jna.library.path", System.getenv("VCPKG_DYNAMIC_LIB"));
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testRotation() {
|
||||
|
||||
Pix pix = Leptonica1.pixRead("/home/kschuettler/Downloads/painHarold.webp");
|
||||
Pix pix2 = ImageProcessingUtils.deRotatePix(0, pix);
|
||||
Leptonica1.pixWrite("/tmp/0.png", pix2, IFF_PNG);
|
||||
Pix pix3 = ImageProcessingUtils.deRotatePix(90, pix);
|
||||
Leptonica1.pixWrite("/tmp/90.png", pix3, IFF_PNG);
|
||||
Pix pix4 = ImageProcessingUtils.deRotatePix(180, pix);
|
||||
Leptonica1.pixWrite("/tmp/180.png", pix4, IFF_PNG);
|
||||
Pix pix5 = ImageProcessingUtils.deRotatePix(270, pix);
|
||||
Leptonica1.pixWrite("/tmp/270.png", pix5, IFF_PNG);
|
||||
}
|
||||
|
||||
}
|
||||
+1
-27
@@ -1,10 +1,7 @@
|
||||
package com.knecon.fforesight.service.ocr.processor.utils;
|
||||
|
||||
import java.awt.image.BufferedImage;
|
||||
import java.io.BufferedReader;
|
||||
import java.io.File;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.stream.IntStream;
|
||||
@@ -19,7 +16,7 @@ import org.springframework.core.io.ClassPathResource;
|
||||
import org.springframework.util.FileSystemUtils;
|
||||
|
||||
import com.knecon.fforesight.service.ocr.processor.service.OsUtils;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.ProcessIOLogger;
|
||||
import com.knecon.fforesight.service.ocr.processor.service.threads.GhostScriptOutputHandler;
|
||||
|
||||
import lombok.SneakyThrows;
|
||||
|
||||
@@ -50,29 +47,6 @@ public class Pdf2ImgTest {
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
@SneakyThrows
|
||||
public void testGhostScript() {
|
||||
|
||||
String outputDir = "/tmp/ghostscript_out/";
|
||||
new File(outputDir).mkdirs();
|
||||
ClassPathResource resource = new ClassPathResource("files/Cyberport__SD-Faktura-Kopie_(ZRG2)_-_31.08.2020.pdf");
|
||||
|
||||
String[] cmdArgs = new String[]{"gs", "-dNOPAUSE", "-sDEVICE=tiff24nc", "-r" + DPI, "-sOutputFile=" + outputDir + "page%04d", resource.getFile().toString(), "-c", "quit"};
|
||||
Process p = Runtime.getRuntime().exec(cmdArgs);
|
||||
ProcessIOLogger logger = new ProcessIOLogger(p.getInputStream(), "GS", ProcessIOLogger.Type.STD_OUT);
|
||||
logger.start();
|
||||
ProcessIOLogger errorLogger = new ProcessIOLogger(p.getErrorStream(), "GS", ProcessIOLogger.Type.STD_OUT);
|
||||
errorLogger.start();
|
||||
int exitcode = p.waitFor();
|
||||
logger.join();
|
||||
errorLogger.join();
|
||||
System.out.println("Ghostscript finished with exit code " + exitcode);
|
||||
FileSystemUtils.deleteRecursively(new File(outputDir));
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
@SneakyThrows
|
||||
public void testGhostScriptParallel() {
|
||||
|
||||
@@ -3,7 +3,7 @@ import org.springframework.boot.gradle.tasks.bundling.BootBuildImage
|
||||
plugins {
|
||||
application
|
||||
id("com.iqser.red.service.java-conventions")
|
||||
id("org.springframework.boot") version "3.1.3"
|
||||
id("org.springframework.boot") version "3.1.5"
|
||||
id("io.spring.dependency-management") version "1.1.3"
|
||||
id("org.sonarqube") version "4.3.0.3225"
|
||||
id("io.freefair.lombok") version "8.2.2"
|
||||
@@ -11,19 +11,26 @@ plugins {
|
||||
|
||||
configurations {
|
||||
all {
|
||||
exclude(group = "org.springframework.boot", module = "spring-boot-starter-logging")
|
||||
exclude(group = "commons-logging", module = "commons-logging")
|
||||
exclude(group = "org.springframework.boot", module = "spring-boot-starter-log4j2")
|
||||
exclude(group = "com.iqser.red.commons", module = "logging-commons")
|
||||
}
|
||||
}
|
||||
|
||||
val springBootStarterVersion = "3.1.5"
|
||||
|
||||
dependencies {
|
||||
implementation(project(":ocr-service-processor"))
|
||||
implementation(project(":ocr-service-api"))
|
||||
|
||||
implementation("com.knecon.fforesight:tracing-commons:0.3.0")
|
||||
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:4.0.4")
|
||||
implementation("org.springframework.boot:spring-boot-starter-amqp:3.1.4")
|
||||
implementation("org.springframework.boot:spring-boot-starter-amqp:${springBootStarterVersion}")
|
||||
|
||||
testImplementation("org.springframework.boot:spring-boot-starter-test:3.1.4")
|
||||
implementation("net.logstash.logback:logstash-logback-encoder:7.4")
|
||||
implementation("ch.qos.logback:logback-classic")
|
||||
|
||||
testImplementation("org.springframework.boot:spring-boot-starter-test:${springBootStarterVersion}")
|
||||
testImplementation("com.iqser.red.commons:test-commons:2.1.0")
|
||||
testImplementation("org.springframework.amqp:spring-rabbit-test:3.0.2")
|
||||
}
|
||||
|
||||
@@ -5,10 +5,16 @@ persistence-service.url: "http://persistence-service-v1:8080"
|
||||
tenant-user-management-service.url: "http://tenant-user-management-service:8080/internal"
|
||||
fforesight.tenants.remote: true
|
||||
|
||||
logging.type: ${LOGGING_TYPE:CONSOLE}
|
||||
kubernetes.namespace: ${NAMESPACE:default}
|
||||
project.version: 1.0-SNAPSHOT
|
||||
|
||||
server:
|
||||
port: 8080
|
||||
|
||||
spring:
|
||||
application:
|
||||
name: ocr-service
|
||||
main:
|
||||
allow-circular-references: true # FIXME
|
||||
profiles:
|
||||
@@ -33,6 +39,7 @@ fforesight:
|
||||
ignored-endpoints: [ '/actuator/health', '/actuator/health/**' ]
|
||||
enabled: true
|
||||
|
||||
logging.pattern.level: "%5p [${spring.application.name},%X{traceId:-},%X{spanId:-}]"
|
||||
|
||||
management:
|
||||
endpoint:
|
||||
@@ -41,9 +48,12 @@ management:
|
||||
health.enabled: true
|
||||
endpoints.web.exposure.include: prometheus, health, metrics
|
||||
metrics.export.prometheus.enabled: ${monitoring.enabled:false}
|
||||
|
||||
|
||||
storage:
|
||||
backend: 's3'
|
||||
tracing:
|
||||
enabled: ${TRACING_ENABLED:false}
|
||||
sampling:
|
||||
probability: ${TRACING_PROBABILITY:1.0}
|
||||
otlp:
|
||||
tracing:
|
||||
endpoint: ${OTLP_ENDPOINT:http://otel-collector-opentelemetry-collector.otel-collector:4318/v1/traces}
|
||||
|
||||
pdftron.license: ${PDFTRON_LICENSE}
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
<configuration>
|
||||
|
||||
<springProperty scope="configuration" name="logType" source="logging.type"/>
|
||||
<springProperty scope="context" name="application.name" source="spring.application.name"/>
|
||||
<springProperty scope="context" name="version" source="project.version"/>
|
||||
<include resource="org/springframework/boot/logging/logback/defaults.xml"/>
|
||||
<include resource="org/springframework/boot/logging/logback/console-appender.xml"/>
|
||||
|
||||
<appender name="JSON" class="ch.qos.logback.core.ConsoleAppender">
|
||||
<encoder class="net.logstash.logback.encoder.LogstashEncoder"/>
|
||||
</appender>
|
||||
|
||||
<root level="INFO">
|
||||
<appender-ref ref="${logType}"/>
|
||||
</root>
|
||||
|
||||
</configuration>
|
||||
+57
-1
@@ -4,10 +4,15 @@ import static com.iqser.red.pdftronlogic.commons.PdfTextExtraction.extractAllTex
|
||||
import static com.knecon.fforesight.service.ocr.processor.service.OsUtils.getTemporaryDirectory;
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileInputStream;
|
||||
import java.io.FileOutputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Disabled;
|
||||
@@ -59,7 +64,7 @@ public class OcrServiceIntegrationTest extends AbstractTest {
|
||||
@SneakyThrows
|
||||
public void testOcr() {
|
||||
|
||||
String text = testOCR("files/2009-1048395_50pages_tables.pdf");
|
||||
String text = testOCR("files/402Study.pdf");
|
||||
}
|
||||
|
||||
|
||||
@@ -127,4 +132,55 @@ public class OcrServiceIntegrationTest extends AbstractTest {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
@SneakyThrows
|
||||
public void testOcrForAllDMFiles() {
|
||||
|
||||
String dir = "/home/kschuettler/Dokumente/TestFiles/syn-dm-testfiles/";
|
||||
List<File> foundFiles = Files.walk(Path.of(dir))
|
||||
.sorted(Comparator.comparingLong(this::getFileSize))
|
||||
.map(Path::toFile)
|
||||
.filter(file -> file.getName().endsWith(".pdf"))
|
||||
.peek(System.out::println)
|
||||
.toList();
|
||||
int fileCount = foundFiles.size();
|
||||
AtomicInteger processedCount = new AtomicInteger();
|
||||
System.out.printf("Found %s files, starting OCR for each.%n%n", fileCount);
|
||||
foundFiles.stream().peek(file -> System.out.printf("%s/%s: %s%n", processedCount.getAndIncrement(), fileCount, file)).forEach(this::testOCRForFile);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public long getFileSize(Path path) {
|
||||
|
||||
return Files.size(path);
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
@SneakyThrows
|
||||
public void testOcrForSpecificFile() {
|
||||
|
||||
testOCRForFile(new File("/home/kschuettler/Dokumente/TestFiles/syn-dm-testfiles2/A16361B - Acute Dermal Toxicity Study in Rats.pdf"));
|
||||
}
|
||||
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private void testOCRForFile(File file) {
|
||||
|
||||
var originId = FileStorageService.getStorageId(TEST_DOSSIER_ID, "file", FileType.ORIGIN);
|
||||
try (var fileStream = new FileInputStream(file)) {
|
||||
storageService.storeObject(TenantContext.getTenantId(), originId, fileStream);
|
||||
}
|
||||
|
||||
Path tmpFileName = Path.of(getTemporaryDirectory()).resolve(Path.of(file.getAbsolutePath()).getFileName());
|
||||
try (var out = new FileOutputStream(tmpFileName.toFile())) {
|
||||
ocrService.runOcrOnDocument(TEST_DOSSIER_ID, "file", out);
|
||||
System.out.println("File:" + tmpFileName);
|
||||
}
|
||||
System.out.println("\n\n");
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user