Compare commits

...
Author SHA1 Message Date
shamel-hussain 75d7657305 RED-5718: Run the ocr service as specific user 2023-02-21 16:02:19 +01:00
Kilian Schuettler a6d99f5916 Pull request #8: RED-6126: In the OCRService, OCR Text is not applied to Document
Merge in RED/ocr-service from RED-6126 to master

* commit '0bc4fea2a52c92efaaaf8cf93c2ae02766168a80':
  RED-6126: In the OCRService, OCR Text is not applied to Document *removed unnecessary getXObject() call, since it fails for inline_images
2023-02-14 09:57:28 +01:00
Kilian Schuettler 0bc4fea2a5 RED-6126: In the OCRService, OCR Text is not applied to Document
*removed unnecessary getXObject() call, since it fails for inline_images
2023-02-13 17:55:02 +01:00
Kilian Schuettler 001719a34c Pull request #7: RED-6126 performance test
Merge in RED/ocr-service from RED-6126-performance-test to master

* commit '37f1e03ebcd5356e0f0b403a5c0cdd20fc133997':
  RED-6126: performance-test *refactor to improve cleanness *closed inputStream
  RED-6126: performance-test *fixed NullPointerException *fixed StackOverFlowError by ignoring very small images and moving to while loop instead of recursion
  RED-6126: performance-test *fixed time calculation
  RED-6126: performance-test *improved error logging
  RED-6126: performance-test *re-enabled overlap detection *re-creating helper document for every page instead of reusing and adding/removing pages
  RED-6126: Performance Tests *moved to streams for pdf file transfer *disabled overlap detection
2023-02-10 15:00:55 +01:00
Kilian Schuettler 37f1e03ebc RED-6126: performance-test
*refactor to improve cleanness
*closed inputStream
2023-02-10 14:49:10 +01:00
Kilian Schuettler b3fa14b342 RED-6126: performance-test
*fixed NullPointerException
*fixed StackOverFlowError by ignoring very small images and moving to while loop instead of recursion
2023-02-10 12:27:16 +01:00
Kilian Schuettler 7065d098f3 RED-6126: performance-test
*fixed time calculation
2023-02-09 16:31:42 +01:00
Kilian Schuettler 8db0b712f7 RED-6126: performance-test
*improved error logging
2023-02-09 13:57:21 +01:00
8 changed files with 155 additions and 71 deletions
@@ -1,4 +1,4 @@
FROM red/base-image:2.0.0
FROM red/base-image:baseImageUser5
COPY "libs/pdftron/OCRModuleLinux.tar.gz" .
RUN tar xvzf OCRModuleLinux.tar.gz
@@ -27,9 +27,12 @@ public class ImagePositionRetrievalService {
private static final double TOLERANCE = 1e-1;
// any image with smaller height and width than this gets thrown out, see everyPointInDashedLineIsImage.pdf
private static final int PIXEL_THRESHOLD = 10;
/**
* Iterates over all elements in a PDF Document and retrieves the bounding box for each image,
* Iterates over all elements in a PDF Document and retrieves the bounding box for each image, that is larger than the pixel threshold of 10 in either dimension.
* Then it adjusts the bounding boxes for the page rotation.
* If the mirrorY flag is set, the Y Coordinates are mirrored and moved up by the page height. This is required for PDFTrons OCRModule.
*
@@ -63,7 +66,12 @@ public class ImagePositionRetrievalService {
Element element;
while ((element = reader.next()) != null) {
switch (element.getType()) {
case Element.e_image, Element.e_inline_image -> imagePositions.addRect(toRotationAdjustedRect(element.getBBox(), currentPage, mirrorY));
case Element.e_image, Element.e_inline_image -> {
// see everyPointInDashedLineIsImage.pdf TestFile
if (element.getImageHeight() > PIXEL_THRESHOLD || element.getImageWidth() > PIXEL_THRESHOLD) {
imagePositions.addRect(toRotationAdjustedRect(element.getBBox(), currentPage, mirrorY));
}
}
case Element.e_form -> {
reader.formBegin();
findImagePositionsOnPage(reader, imagePositions, currentPage, mirrorY);
@@ -77,39 +85,49 @@ public class ImagePositionRetrievalService {
@SneakyThrows
public RectCollection mergeOverlappingRects(RectCollection imagePositions) {
if (imagePositions.getNumRects() == 1) {
if (imagePositions.getNumRects() < 2) {
return imagePositions;
}
List<Rectangle2D> rectangleList = toSortedRectangleList(imagePositions);
rectangleList = mergeRectangleListRecursive(rectangleList, 0);
mergeRectangleList(rectangleList);
return toRectCollection(rectangleList);
}
// Sometimes images are split up into stripes, here we merge the positions of aligned and intersecting rectangles into one larger rectangle
private List<Rectangle2D> mergeRectangleListRecursive(List<Rectangle2D> rectangleList, int currentIdx) {
private void mergeRectangleList(List<Rectangle2D> rectangleList) {
if (rectangleList.size() < currentIdx + 2) {
return rectangleList;
for (int idx = 0; rectangleList.size() >= idx + 2; ) {
var rect1 = rectangleList.get(idx);
var rect2 = rectangleList.get(idx + 1);
if (intersects(rect1, rect2) && isAlignedXOrY(rect1, rect2)) {
rectangleList.remove(idx + 1);
rectangleList.remove(idx);
rectangleList.add(idx, rect1.createUnion(rect2));
} else {
++idx;
}
}
}
var rect1 = rectangleList.get(currentIdx);
var rect2 = rectangleList.get(currentIdx + 1);
private boolean intersects(Rectangle2D rect1, Rectangle2D rect2) {
return rect1.intersects(rect2.getMinX() - TOLERANCE, rect2.getMinY() - TOLERANCE, rect2.getWidth() + (2 * TOLERANCE), rect2.getHeight() + (2 * TOLERANCE));
}
private boolean isAlignedXOrY(Rectangle2D rect1, Rectangle2D rect2) {
boolean isAlignedX = Math.abs(rect1.getMinX() - rect2.getMinX()) < TOLERANCE && Math.abs(rect1.getMaxX() - rect2.getMaxX()) < TOLERANCE;
boolean isAlignedY = Math.abs(rect1.getMinY() - rect2.getMinY()) < TOLERANCE && Math.abs(rect1.getMaxY() - rect2.getMaxY()) < TOLERANCE;
boolean intersects = rect1.intersects(rect2.getMinX() - TOLERANCE, rect2.getMinY() - TOLERANCE, rect2.getWidth() + (2 * TOLERANCE), rect2.getHeight() + (2 * TOLERANCE));
if (intersects && (isAlignedX || isAlignedY)) {
rectangleList.remove(currentIdx + 1);
rectangleList.remove(currentIdx);
rectangleList.add(currentIdx, rect1.createUnion(rect2));
return mergeRectangleListRecursive(rectangleList, currentIdx);
} else {
return mergeRectangleListRecursive(rectangleList, currentIdx + 1);
}
return isAlignedX || isAlignedY;
}
@@ -76,6 +76,8 @@ public class InvisibleElementRemovalService {
Page page = iterator.next();
visitedXObjIds.add(page.getSDFObj().getObjNum());
InvisibleElementRemovalContext context = InvisibleElementRemovalContext.builder()
.reader(reader)
.clippingPathStack(new ClippingPathStack(page.getMediaBox()))
@@ -221,8 +223,14 @@ public class InvisibleElementRemovalService {
private void processPath(Element pathElement, ElementWriter writer, InvisibleElementRemovalContext context) throws PDFNetException {
PathData pathData = pathElement.getPathData();
GeneralPath linePath = convertToGeneralPath(pathElement.getPathData());
if (pathData.getOperators().length == 0 && pathData.getPoints().length == 0) {
writer.writeGStateChanges(pathElement);
return;
}
GeneralPath linePath = convertToGeneralPath(pathData);
//transform path to initial user space
var ctm = pathElement.getCTM();
@@ -1,7 +1,10 @@
package com.iqser.red.service.ocr.v1.server.service;
import static java.lang.String.format;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.OutputStream;
import java.util.Map;
@@ -59,16 +62,22 @@ public class OCRService {
* @param fileId The file id
* @param out OutputStream to write the file to
*/
@SneakyThrows
@Timed("redactmanager_runOcrOnDocument")
public void runOcrOnDocument(String dossierId, String fileId, OutputStream out) {
public void runOcrOnDocument(String dossierId, String fileId, OutputStream out) throws IOException {
try (ByteArrayOutputStream transferOutputStream = new ByteArrayOutputStream()) {
try (InputStream fileStream = fileStorageService.getOriginalFileAsStream(dossierId, fileId)) {
long start = System.currentTimeMillis();
log.debug("Start invisible element removal for file with dossierId {} and fileId {}", dossierId, fileId);
invisibleElementRemovalService.removeInvisibleElements(fileStream, transferOutputStream, false);
long end = System.currentTimeMillis();
log.debug("Invisible element removal successful for file with dossierId {} and fileId {}, took {}s", dossierId, fileId, format("%.1f", (end - start) / 1000.0));
}
try (InputStream transferInputStream = new ByteArrayInputStream(transferOutputStream.toByteArray())) {
long start = System.currentTimeMillis();
runOcr(transferInputStream, out, fileId);
long end = System.currentTimeMillis();
log.info("ocr successful for file with dossierId {} and fileId {}, took {}s", dossierId, fileId, format("%.1f", (end - start) / 1000.0));
}
}
}
@@ -81,36 +90,26 @@ public class OCRService {
Map<Integer, RectCollection> pageIdToRectCollection = imagePositionRetrievalService.getImagePositionPerPage(pdfDoc, true);
// Optimization:
// When a page does not have a TextZone, PDFTron whites out the page. But, PDFTron scans it anyway, resulting in a longer runtime.
// So, we need to remove pages without images.
// Furthermore, creating a new document is *much* faster than reusing the same document and adding/removing pages one by one.
// Therefore, we create a new Document with a single page for every page that contains text.
int numProcessedPages = 0;
for (Integer pageId : pageIdToRectCollection.keySet()) {
try {
// optimization: creating a new document is faster than reusing the same and adding/removing pages one by one
OCROptions options = new OCROptions();
PDFDoc ocrPageDoc = new PDFDoc();
// optimization: only scanning pages that contain images
Page pdfPage = pdfDoc.getPage(pageId);
// optimization: this line ensures the ocr text is placed correctly by PDFTron
pdfPage.setMediaBox(pdfPage.getCropBox());
ocrPageDoc.pagePushBack(pdfPage);
options.addTextZonesForPage(pageIdToRectCollection.get(pageId), 1);
options.addLang(ENGLISH);
options.addDPI(settings.getOcrDPI());
OCRModule.processPDF(ocrPageDoc, options);
PDFDoc singlePagePdfDoc = extractSinglePagePdfDoc(pdfDoc, pageId);
processOcr(pageIdToRectCollection, pageId, singlePagePdfDoc);
++numProcessedPages;
StringBuilder zonesString = new StringBuilder();
for (int j = 0; j < pageIdToRectCollection.get(pageId).getNumRects(); ++j) {
var r = pageIdToRectCollection.get(pageId).getRectAt(j);
zonesString.append(String.format("[lower left (%.1f|%.1f) upper right (%.1f|%.1f)]", r.getX1(), r.getY1(), r.getX2(), r.getY2()));
}
log.info("{}/{} Page {} done, OCR regions {}", numProcessedPages, pageIdToRectCollection.size(), pageId, zonesString);
log.info("{}/{} Page {} done, OCR regions {}",
numProcessedPages,
pageIdToRectCollection.size(),
pageId,
getAllOcrTextZonesAsString(pageIdToRectCollection, pageId));
// re-adding OCR pages
Page ocrPage = ocrPageDoc.getPage(1);
pdfDoc.pageInsert(pdfDoc.getPageIterator(pageId), ocrPage);
pdfDoc.pageRemove(pdfDoc.getPageIterator(pageId + 1));
ocrPageDoc.close();
replaceOriginalPageWithOcrPage(pdfDoc, pageId, singlePagePdfDoc);
singlePagePdfDoc.close();
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_STATUS_UPDATE_RESPONSE_QUEUE,
objectMapper.writeValueAsString(OCRStatusUpdateResponse.builder()
@@ -120,7 +119,7 @@ public class OCRService {
.build()));
} catch (PDFNetException e) {
log.error("failed to process page {}", pageId);
log.error("Failed to process page {}", pageId);
throw new RuntimeException(e);
}
}
@@ -134,7 +133,52 @@ public class OCRService {
.build()));
Optimizer.optimize(pdfDoc);
pdfDoc.save(out, SDFDoc.SaveMode.LINEARIZED, null);
try {
pdfDoc.save(out, SDFDoc.SaveMode.LINEARIZED, null);
} catch (Exception e) {
log.error("Processed File with fileId {} could not be saved", fileId);
throw new RuntimeException(e);
}
}
private void processOcr(Map<Integer, RectCollection> pageIdToRectCollection, Integer pageId, PDFDoc singlePagePdfDoc) throws PDFNetException {
OCROptions options = new OCROptions();
options.addTextZonesForPage(pageIdToRectCollection.get(pageId), 1);
options.addLang(ENGLISH);
options.addDPI(settings.getOcrDPI());
OCRModule.processPDF(singlePagePdfDoc, options);
}
private static PDFDoc extractSinglePagePdfDoc(PDFDoc pdfDoc, Integer pageId) throws PDFNetException {
PDFDoc singlePagePdfDoc = new PDFDoc();
Page page = pdfDoc.getPage(pageId);
page.setMediaBox(page.getCropBox()); // this line ensures the ocr text is placed correctly by PDFTron, see TestFile MediaBoxBiggerThanCropBox.pdf
singlePagePdfDoc.pagePushBack(page);
return singlePagePdfDoc;
}
private static void replaceOriginalPageWithOcrPage(PDFDoc pdfDoc, Integer pageId, PDFDoc ocrPageDoc) throws PDFNetException {
Page ocrPage = ocrPageDoc.getPage(1);
pdfDoc.pageInsert(pdfDoc.getPageIterator(pageId), ocrPage);
pdfDoc.pageRemove(pdfDoc.getPageIterator(pageId + 1));
}
private static StringBuilder getAllOcrTextZonesAsString(Map<Integer, RectCollection> pageIdToRectCollection, Integer pageId) throws PDFNetException {
StringBuilder zonesString = new StringBuilder();
for (int j = 0; j < pageIdToRectCollection.get(pageId).getNumRects(); ++j) {
var r = pageIdToRectCollection.get(pageId).getRectAt(j);
zonesString.append(format("[lower left (%.1f|%.1f) upper right (%.1f|%.1f)]", r.getX1(), r.getY1(), r.getX2(), r.getY2()));
}
return zonesString;
}
}
@@ -38,7 +38,6 @@ public class OcrMessageReceiver {
DocumentRequest ocrRequestMessage = objectMapper.readValue(in, DocumentRequest.class);
long start = System.currentTimeMillis();
log.info("Start ocr for file with dossierId {} and fileId {}", ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId());
setStatusOcrProcessing(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId());
@@ -48,19 +47,18 @@ public class OcrMessageReceiver {
fileStorageService.storeUntouchedFile(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), originalFile);
}
try (var out = new ByteArrayOutputStream()) {
ocrService.runOcrOnDocument(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), out);
fileStorageService.storeOriginalFile(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), new ByteArrayInputStream(out.toByteArray()));
try (var transferStream = new ByteArrayOutputStream()) {
ocrService.runOcrOnDocument(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), transferStream);
try (var inputStream = new ByteArrayInputStream(transferStream.toByteArray())) {
fileStorageService.storeOriginalFile(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), inputStream);
}
} catch (IOException e) {
log.error("Failed to store file with dossierId {} and fileId {}", ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId());
throw new RuntimeException(e);
}
long end = System.currentTimeMillis();
log.info("Successfully processed ocr for file with dossierId {} and fileId {}, took {}", ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId(), end - start);
fileStatusProcessingUpdateClient.ocrSuccessful(ocrRequestMessage.getDossierId(), ocrRequestMessage.getFileId());
}
@@ -5,12 +5,12 @@ import static org.assertj.core.api.Assertions.assertThat;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.util.ArrayList;
import java.util.List;
import java.util.concurrent.TimeUnit;
import io.micrometer.prometheus.PrometheusMeterRegistry;
import io.micrometer.prometheus.PrometheusTimer;
import org.junit.jupiter.api.AfterEach;
import org.junit.jupiter.api.BeforeEach;
import org.junit.jupiter.api.Disabled;
@@ -36,12 +36,15 @@ import com.iqser.red.service.ocr.v1.server.utils.FileSystemBackedStorageService;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.storage.commons.StorageAutoConfiguration;
import com.iqser.red.storage.commons.service.StorageService;
import com.pdftron.common.PDFNetException;
import com.pdftron.pdf.OCRModule;
import com.pdftron.pdf.PDFDoc;
import com.pdftron.pdf.Page;
import com.pdftron.pdf.PageIterator;
import com.pdftron.pdf.TextExtractor;
import io.micrometer.prometheus.PrometheusMeterRegistry;
import io.micrometer.prometheus.PrometheusTimer;
import lombok.SneakyThrows;
@ExtendWith(SpringExtension.class)
@@ -80,19 +83,20 @@ public class OcrServiceIntegrationTest {
@Test
@Disabled // OCRModule is not available on build server. If you want to run the test set the property at the top.
public void testOCRMetrics(){
public void testOCRMetrics() {
testOCR("Watermark");
testOCR("Watermark");
testOCR("Watermark");
var ocrOnDocumentMeter = registry.getMeters().stream()
.filter(m -> m.getId().getName().equalsIgnoreCase("redactmanager_runOcrOnDocument")).findAny();
var ocrOnDocumentMeter = registry.getMeters().stream().filter(m -> m.getId().getName().equalsIgnoreCase("redactmanager_runOcrOnDocument")).findAny();
assertThat(ocrOnDocumentMeter.isPresent()).isTrue();
PrometheusTimer timer = (PrometheusTimer) ocrOnDocumentMeter.get();
assertThat(timer.count()).isEqualTo(3);
assertThat(timer.mean(TimeUnit.SECONDS)).isGreaterThan(0.1);
}
@Test
@Disabled // OCRModule is not available on build server. If you want to run the test set the property at the top.
public void testOcr() {
@@ -153,30 +157,34 @@ public class OcrServiceIntegrationTest {
private String testOCR(String fileName) {
ClassPathResource pdfFileResource = new ClassPathResource("files/" + fileName + ".pdf");
var originId = FileStorageService.getStorageId("dossier", "file", FileType.ORIGIN);
try (var fileStream = pdfFileResource.getInputStream()) {
storageService.storeObject(originId, fileStream);
}
try (var out = new FileOutputStream(getTemporaryDirectory() + "/" + fileName + ".pdf")) {
ocrService.runOcrOnDocument("dossier", "file", out);
}
System.out.println("File:" + getTemporaryDirectory() + "/" + fileName + ".pdf");
try (var fileStream = new FileInputStream(getTemporaryDirectory() + "/" + fileName + ".pdf")) {
return extractAllTextFromDocument(fileStream);
}
}
private static String extractAllTextFromDocument(InputStream fileStream) throws IOException, PDFNetException {
PDFDoc pdfDoc = new PDFDoc(fileStream);
TextExtractor extractor = new TextExtractor();
List<String> texts = new ArrayList<>();
PDFDoc pdfDoc;
try (var fileStream = new FileInputStream(getTemporaryDirectory() + "/" + fileName + ".pdf")) {
pdfDoc = new PDFDoc(fileStream);
}
PageIterator iterator = pdfDoc.getPageIterator();
while (iterator.hasNext()) {
Page page = iterator.next();
extractor.begin(page);
texts.add(extractor.getAsText());
}
System.out.println("File:" + getTemporaryDirectory() + "/" + fileName + ".pdf");
return String.join("\n", texts);
}
@@ -184,7 +192,7 @@ public class OcrServiceIntegrationTest {
@SneakyThrows
public void dummyTest() {
// Build needs one text to not fail.
// Build needs one test to not fail.
assertThat(1).isEqualTo(1);
}
@@ -204,7 +212,7 @@ public class OcrServiceIntegrationTest {
@Bean
@Primary
public StorageService inmemoryStorage() {
public StorageService inMemoryStorage() {
return new FileSystemBackedStorageService();
}
@@ -122,6 +122,14 @@ class ImagePositionRetrievalServiceTest {
assertThat(allRectCoords.size()).isEqualTo(48);
}
@Test
@SneakyThrows
public void testEveryPointInDashedLineIsImage() {
String fileName = "everyPointInDashedLineIsImage";
List<int[]> allRectCoords = testImagePositionDetection(fileName);
assertThat(allRectCoords.size()).isEqualTo(0);
}
private List<int[]> testImagePositionDetection(String fileName) throws IOException, PDFNetException {