Compare commits
16
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0f6d33f882 | ||
|
|
7b4ca83563 | ||
|
|
d40d2fd58d | ||
|
|
39b323e69e | ||
|
|
412edec340 | ||
|
|
242b3ef3b8 | ||
|
|
e4aaecc750 | ||
|
|
c7be843a6f | ||
|
|
2cadfdb8bf | ||
|
|
b019ca8e2d | ||
|
|
670c505042 | ||
|
|
22049d81c2 | ||
|
|
050c399270 | ||
|
|
8f123fb865 | ||
|
|
21523a7796 | ||
|
|
fdb3ae46dc |
@@ -22,4 +22,5 @@ deploy:
|
||||
rules:
|
||||
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
|
||||
- if: $CI_COMMIT_BRANCH =~ /^release/
|
||||
- if: $CI_COMMIT_BRANCH =~ /^feature/
|
||||
- if: $CI_COMMIT_TAG
|
||||
|
||||
@@ -70,7 +70,7 @@ int concurrency = 8;
|
||||
int batchSize = 128;
|
||||
boolean debug; // writes the ocr layer visibly to the viewer doc pdf
|
||||
boolean idpEnabled; // Enables table detection, paragraph classification, section detection, key-value detection.
|
||||
boolean tableDetection; // writes the tables to the PDF as invisible lines.
|
||||
boolean drawTablesAsLines; // writes the tables to the PDF as invisible lines.
|
||||
boolean processAllPages; // if this parameter is set, ocr will be performed on any page, regardless if it has images or not
|
||||
boolean fontStyleDetection; // Enables bold detection using ghostscript and leptonica
|
||||
String contentFormat; // Either markdown or text. But, for whatever reason, with markdown enabled, key-values are not written by azure....
|
||||
|
||||
+4
-4
@@ -23,7 +23,7 @@ public class DocumentRequest {
|
||||
String viewerDocId;
|
||||
String idpResultId;
|
||||
|
||||
boolean removeWatermarks;
|
||||
boolean removeWatermark;
|
||||
|
||||
|
||||
public DocumentRequest(String dossierId, String fileId) {
|
||||
@@ -33,15 +33,15 @@ public class DocumentRequest {
|
||||
originDocumentId = null;
|
||||
viewerDocId = null;
|
||||
idpResultId = null;
|
||||
removeWatermarks = false;
|
||||
removeWatermark = false;
|
||||
}
|
||||
|
||||
// needed for backwards compatibility
|
||||
public DocumentRequest(String dossierId, String fileId, boolean removeWatermarks) {
|
||||
public DocumentRequest(String dossierId, String fileId, boolean removeWatermark) {
|
||||
|
||||
this.dossierId = dossierId;
|
||||
this.fileId = fileId;
|
||||
this.removeWatermarks = removeWatermarks;
|
||||
this.removeWatermark = removeWatermark;
|
||||
originDocumentId = null;
|
||||
viewerDocId = null;
|
||||
idpResultId = null;
|
||||
|
||||
@@ -21,8 +21,8 @@ dependencies {
|
||||
api("org.apache.commons:commons-math3:3.6.1")
|
||||
api("com.amazonaws:aws-java-sdk-kms:1.12.440")
|
||||
api("com.google.guava:guava:31.1-jre")
|
||||
api("com.iqser.red.commons:pdftron-logic-commons:2.27.0")
|
||||
api("com.knecon.fforesight:viewer-doc-processor:0.148.0")
|
||||
api("com.knecon.fforesight:viewer-doc-processor:0.177.0")
|
||||
api("com.azure:azure-ai-documentintelligence:1.0.0-beta.3")
|
||||
api("com.iqser.red.commons:pdftron-logic-commons:2.31.0")
|
||||
testImplementation("org.junit.jupiter:junit-jupiter:5.8.1")
|
||||
}
|
||||
|
||||
+1
-1
@@ -18,7 +18,7 @@ public class OcrServiceSettings {
|
||||
|
||||
boolean debug; // writes the ocr layer visibly to the viewer doc pdf
|
||||
boolean idpEnabled; // Enables table detection, paragraph classification, section detection, key-value detection.
|
||||
boolean tableDetection; // writes the tables to the PDF as invisible lines.
|
||||
boolean drawTablesAsLines; // writes the tables to the PDF as invisible lines.
|
||||
boolean processAllPages; // if this parameter is set, ocr will be performed on any page, regardless if it has images or not
|
||||
boolean fontStyleDetection; // Enables bold detection using ghostscript and leptonica
|
||||
String contentFormat; // Either markdown or text. But, for whatever reason, with markdown enabled, key-values are not written by azure....
|
||||
|
||||
+6
-4
@@ -20,10 +20,12 @@ public record PageInformation(Rectangle2D mediabox, int number, int rotationDegr
|
||||
|
||||
ConcurrentHashMap<Integer, PageInformation> pageInformationMap = new ConcurrentHashMap<>();
|
||||
int pageNumber = 1;
|
||||
for (PageIterator iterator = pdfDoc.getPageIterator(); iterator.hasNext(); pageNumber++) {
|
||||
|
||||
Page page = iterator.next();
|
||||
pageInformationMap.put(pageNumber, PageInformation.fromPage(pageNumber, page));
|
||||
try (PageIterator iterator = pdfDoc.getPageIterator()) {
|
||||
while (iterator.hasNext()) {
|
||||
Page page = iterator.next();
|
||||
pageInformationMap.put(pageNumber, PageInformation.fromPage(pageNumber, page));
|
||||
pageNumber++;
|
||||
}
|
||||
}
|
||||
return pageInformationMap;
|
||||
}
|
||||
|
||||
+2
@@ -16,6 +16,7 @@ import com.knecon.fforesight.service.ocr.processor.service.imageprocessing.Image
|
||||
import com.knecon.fforesight.service.ocr.processor.visualizations.layers.LayerFactory;
|
||||
import com.knecon.fforesight.service.ocr.processor.visualizations.layers.OcrResult;
|
||||
import com.pdftron.common.PDFNetException;
|
||||
import com.pdftron.pdf.Optimizer;
|
||||
import com.pdftron.pdf.PDFDoc;
|
||||
import com.pdftron.sdf.SDFDoc;
|
||||
|
||||
@@ -74,6 +75,7 @@ public class AsyncOcrService {
|
||||
|
||||
BinaryData docData;
|
||||
try (var smallerDoc = extractBatchDocument(pdfDoc, batch)) {
|
||||
Optimizer.optimize(smallerDoc);
|
||||
docData = BinaryData.fromBytes(smallerDoc.save(SDFDoc.SaveMode.LINEARIZED, null));
|
||||
}
|
||||
return docData;
|
||||
|
||||
+4
-1
@@ -72,8 +72,11 @@ public class ImageDetectionService {
|
||||
}
|
||||
case Element.e_form -> {
|
||||
reader.formBegin();
|
||||
findImagePositionsOnPage(reader);
|
||||
var found = findImagePositionsOnPage(reader);
|
||||
reader.end();
|
||||
if (found) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -53,7 +53,7 @@ public class ImageProcessingSupervisor {
|
||||
|
||||
public ImageFile awaitProcessedPage(Integer pageNumber) throws InterruptedException {
|
||||
|
||||
if (hasErros()) {
|
||||
if (hasErrors()) {
|
||||
return null;
|
||||
}
|
||||
getPageLatch(pageNumber).await();
|
||||
@@ -61,7 +61,7 @@ public class ImageProcessingSupervisor {
|
||||
}
|
||||
|
||||
|
||||
private boolean hasErros() {
|
||||
private boolean hasErrors() {
|
||||
|
||||
return errors.isEmpty();
|
||||
}
|
||||
|
||||
+1
-1
@@ -88,7 +88,7 @@ public class WritableOcrResultFactory {
|
||||
|
||||
var builder = WritableOcrResult.builder().pageNumber(pageInformation.number()).textPositionInImage(words);
|
||||
|
||||
if (settings.isTableDetection()) {
|
||||
if (settings.isDrawTablesAsLines()) {
|
||||
builder.tableLines(getTableLines(analyzeResult, pageInformation, pageCtm));
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -63,7 +63,7 @@ public class OcrMessageReceiver {
|
||||
|
||||
fileStorageService.downloadFiles(request, documentFile);
|
||||
|
||||
ocrService.runOcrOnDocument(dossierId, fileId, request.isRemoveWatermarks(), tmpDir, documentFile, viewerDocumentFile, analyzeResultFile);
|
||||
ocrService.runOcrOnDocument(dossierId, fileId, request.isRemoveWatermark(), tmpDir, documentFile, viewerDocumentFile, analyzeResultFile);
|
||||
|
||||
fileStorageService.storeFiles(request, documentFile, viewerDocumentFile, analyzeResultFile);
|
||||
|
||||
|
||||
+4
-5
@@ -8,7 +8,6 @@ import com.knecon.fforesight.service.ocr.processor.service.IOcrMessageSender;
|
||||
import com.knecon.fforesight.service.ocr.v1.api.model.DocumentRequest;
|
||||
import com.knecon.fforesight.service.ocr.v1.api.model.OCRStatusUpdateResponse;
|
||||
import com.knecon.fforesight.service.ocr.v1.server.configuration.MessagingConfiguration;
|
||||
import com.knecon.fforesight.tenantcommons.TenantContext;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
@@ -28,7 +27,7 @@ public class OcrMessageSender implements IOcrMessageSender {
|
||||
public void sendOcrFinished(String fileId, int totalImages) {
|
||||
|
||||
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_STATUS_UPDATE_RESPONSE_QUEUE,
|
||||
TenantContext.getTenantId(),
|
||||
|
||||
OCRStatusUpdateResponse.builder().fileId(fileId).numberOfPagesToOCR(totalImages).numberOfOCRedPages(totalImages).ocrFinished(true).build());
|
||||
}
|
||||
|
||||
@@ -36,7 +35,7 @@ public class OcrMessageSender implements IOcrMessageSender {
|
||||
public void sendOCRStarted(String fileId) {
|
||||
|
||||
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_STATUS_UPDATE_RESPONSE_QUEUE,
|
||||
TenantContext.getTenantId(),
|
||||
|
||||
OCRStatusUpdateResponse.builder().fileId(fileId).ocrStarted(true).build());
|
||||
|
||||
}
|
||||
@@ -45,7 +44,7 @@ public class OcrMessageSender implements IOcrMessageSender {
|
||||
public void sendUpdate(String fileId, int finishedImages, int totalImages) {
|
||||
|
||||
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_STATUS_UPDATE_RESPONSE_QUEUE,
|
||||
TenantContext.getTenantId(),
|
||||
|
||||
OCRStatusUpdateResponse.builder().fileId(fileId).numberOfPagesToOCR(totalImages).numberOfOCRedPages(finishedImages).build());
|
||||
|
||||
}
|
||||
@@ -53,7 +52,7 @@ public class OcrMessageSender implements IOcrMessageSender {
|
||||
|
||||
public void sendOcrResponse(String dossierId, String fileId) {
|
||||
|
||||
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_RESPONSE_QUEUE, TenantContext.getTenantId(), new DocumentRequest(dossierId, fileId));
|
||||
rabbitTemplate.convertAndSend(MessagingConfiguration.OCR_RESPONSE_QUEUE, new DocumentRequest(dossierId, fileId));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user