Compare commits

..
Author SHA1 Message Date
maverickstuder f90cb20156 RED-8666 2024-03-04 15:18:13 +01:00
Dominique Eifländer 72202f63dc More 2024-02-16 14:51:26 +01:00
Dominique Eifländer e14d953b04 More 2024-02-16 14:50:31 +01:00
Dominique Eifländer 9e5778d4b2 More 2024-02-16 14:08:59 +01:00
Dominique Eifländer e394f2fa7c More refactoring 2024-02-16 13:48:03 +01:00
Dominique Eifländer b2fb6829cb More refactoring 2024-02-16 11:15:44 +01:00
Dominique Eifländer 4871e55f2d More refactoring 2024-02-15 16:54:07 +01:00
Dominique Eifländer 4de6c12aec REmove more 2024-02-15 10:29:39 +01:00
Dominique Eifländer 4afa8daafa First working docstrum 2024-02-14 13:39:27 +01:00
Kilian Schüttler 3c9049dc8a Merge branch 'RED-8156' into 'main'
RED-8156: refactor ViewerDocumentService as a dependency for ocr-service

See merge request fforesight/layout-parser!99
2024-02-07 10:42:21 +01:00
Kilian Schuettler 015984891f RED-8156: refactor ViewerDocumentService as a dependency for ocr-service
* fix pmd
2024-02-06 17:17:26 +01:00
Kilian Schuettler 66fcb62833 RED-8156: refactor ViewerDocumentService as a dependency for ocr-service
* fix pmd
2024-02-06 17:09:21 +01:00
Kilian Schuettler 48824f56a8 RED-8156: refactor ViewerDocumentService as a dependency for ocr-service
* fix pmd
2024-02-06 17:06:53 +01:00
Kilian Schuettler 785628537f RED-8156: refactor ViewerDocumentService as a dependency for ocr-service
* various improvements to experimental parsing steps
* added embed fonts functionality to viewer doc
* fix checkstyle
2024-02-06 17:03:38 +01:00
Kilian Schuettler 23eb0c40a3 RED-8156: refactor ViewerDocumentService as a dependency for ocr-service
* various improvements to experimental parsing steps
* added embed fonts functionality to viewer doc
2024-02-06 16:59:51 +01:00
Dominique Eifländer 1b4aaf4454 Merge branch 'RED-8171' into 'main'
RED-8171: Traces do not stop at @Async

See merge request fforesight/layout-parser!98
2024-02-02 13:34:44 +01:00
Dominique Eifländer e4f3557b36 RED-8171: Traces do not stop at @Async 2024-02-02 13:22:57 +01:00
Timo Bejan 9be3c86297 Merge branch 'RED-8085' into 'main'
Red 8085

See merge request fforesight/layout-parser!96
2024-01-29 10:31:36 +01:00
Timo Bejan 88855de2da Red 8085 2024-01-29 10:31:36 +01:00
Dominique Eifländer 368a75e985 Merge branch 'RED-8106' into 'main'
RED-8106: Make documentdata serializable

See merge request fforesight/layout-parser!95
2023-12-22 13:33:02 +01:00
Dominique Eifländer 12344d57b2 RED-8106: Make documentdata serializable 2023-12-21 13:42:25 +01:00
Dominique Eifländer 9e854379e7 Merge branch 'RED-1137' into 'main'
RED-1137: Do not observe actuator endpoints

See merge request fforesight/layout-parser!92
2023-12-20 14:11:31 +01:00
Dominique Eifländer b779c72041 RED-1137: Do not observe actuator endpoints 2023-12-20 14:05:00 +01:00
Dominique Eifländer 760a809900 Merge branch 'RED-7384' into 'main'
RED-7384: fixes for migration

See merge request fforesight/layout-parser!91
2023-12-20 12:40:00 +01:00
Kilian Schüttler ba1c7c07ab RED-7384: fixes for migration 2023-12-20 12:40:00 +01:00
Dominique Eifländer ca0cbbcb49 Merge branch 'RED-5223' into 'main'
RED-5223: Use tracing-commons from fforesight

See merge request fforesight/layout-parser!90
2023-12-13 16:05:01 +01:00
Dominique Eifländer da2cdc288e RED-5223: Use tracing-commons from fforesight 2023-12-13 15:31:26 +01:00
Dominique Eifländer 68da328889 Merge branch 'queueHotfix' into 'main'
hotfix: removed dlq from response queue to be equal to persistence-service

See merge request fforesight/layout-parser!89
2023-12-13 09:54:43 +01:00
Dominique Eifländer 711548d1a7 hotfix: removed dlq from response queue to be equal to persistence-service 2023-12-13 09:47:27 +01:00
Dominique Eifländer 2bddcdafee Merge branch 'RED-5223' into 'main'
RED-5223: Enabled tracing, upgrade spring, use logstash-logback-encoder for json logs

See merge request fforesight/layout-parser!88
2023-12-11 15:13:22 +01:00
Dominique Eifländer 750ccf4ce2 RED-5223: Enabled tracing, upgrade spring, use logstash-logback-encoder for json logs 2023-12-11 15:06:23 +01:00
Ali Oezyetimoglu 57b5d3f48e Merge branch 'RED-7714' into 'main'
RED-7715 - Add log4j config to enable switching between json/line logs

See merge request fforesight/layout-parser!87
2023-12-06 13:00:28 +01:00
Andrei Isvoran d8c9659469 RED-7715 - Add log4j config to enable switching between json/line logs 2023-12-06 11:59:42 +02:00
Kilian Schüttler 30f060e36c Merge branch 'enable-caching' into 'main'
enable caching in build

See merge request fforesight/layout-parser!84
2023-11-24 10:36:55 +01:00
Kilian Schuettler 53a5824e6c enable caching in build 2023-11-24 10:24:50 +01:00
Dominique Eifländer e2bcf971c9 Merge branch 'DM-589' into 'main'
DM-589: Filter wrong detected cells that borders from rotation at scanning

See merge request fforesight/layout-parser!83
2023-11-20 16:07:27 +01:00
Dominique Eifländer dacc2f7f43 DM-589: Filter wrong detected cells that borders from rotation at scanning 2023-11-20 15:54:02 +01:00
Dominique Eifländer 144a9591a2 Merge branch 'TAAS-103-hotfix' into 'main'
* added back in if statement

See merge request fforesight/layout-parser!82
2023-11-16 12:48:48 +01:00
yhampe 207d9dec97 * added back in if statement
* removed not needed commentar
2023-11-16 12:40:49 +01:00
Yannik Hampe 09ee90222e Merge branch 'TAAS-103' into 'main'
TAAS-103: Table Detection and rotated text

See merge request fforesight/layout-parser!81
2023-11-16 09:13:41 +01:00
yhampe 1316a067fe * removed double chechking for height of cell 2023-11-16 08:51:12 +01:00
yhampe e203210ade * removed not needed properties 2023-11-16 08:23:58 +01:00
yhampe b25d46291a * checkstyle 2023-11-16 08:12:47 +01:00
yhampe 84148d3b6e * fixed tests 2023-11-16 07:51:08 +01:00
Dominique Eifländer a6ba66b1aa TAAS-103: Fixed values in wrong cells 2023-11-15 13:36:46 +01:00
yhampe c3e69b2cdf * fixed bug with incorrect empty cell count by adding threshhold to cell.contains 2023-11-15 10:44:47 +01:00
yhampe f69331e7d8 *renamed page to firstPage in DocumentStructure and Table 2023-11-07 10:21:19 +01:00
yhampe 01493dc033 TAAS-103: Table Detection and rotated text
* added page property to DocumentStructure to be able to get page of found tables

* added a method to TableExtractionService to get the table area

* added calculateMinCharWidthAndMaxCharHeightInsideTable to LayoutParsingPipeline to calculate the values based upon table area

* refactored PDFLinesTextStripper for better readability

*removed textMatrix from RedTextPosition as it is no longer needed
2023-11-07 08:47:28 +01:00
yhampe 459e0c8be7 TAAS-103: 2023-11-07 08:39:15 +01:00
Kilian Schüttler 1b1f777706 Merge branch 'RED-7806' into 'main'
RED-7806 - Specific customer document cannot be processed

See merge request fforesight/layout-parser!79
2023-10-25 10:50:34 +02:00
Corina Olariu 0e0a811f9d RED-7806 - Specific customer document cannot be processed
- add brackets
2023-10-25 11:36:54 +03:00
Corina Olariu efa3d75479 RED-7806 - Specific customer document cannot be processed
- check for font name null before using to avoid the NPE
2023-10-25 09:16:47 +03:00
Kilian Schüttler 9abdc6d44d Merge branch 'RED-7434' into 'main'
RED-7434 - Remove Section Grid entirely

See merge request fforesight/layout-parser!78
2023-10-20 10:07:01 +02:00
Corina Olariu 3bab61c446 RED-7434 - Remove Section Grid entirely
- remove sectionGrid relation (including SectionGridCreatorService)
- update junit tests
2023-10-20 09:09:22 +03:00
Dominique Eifländer d17517d3c3 Merge branch 'hotfix-bdr-doc' into 'main'
hotfix: Fixed parsing for specific taas document

See merge request fforesight/layout-parser!77
2023-10-18 16:12:15 +02:00
Dominique Eifländer 567cbc178b hotfix: Fixed parsing for specific taas document 2023-10-17 15:52:19 +02:00
Kilian Schüttler 3c53772765 Merge branch 'RED-7759' into 'main'
RED-7759: Upgraded storage-commons to newest windwos compatible version

See merge request fforesight/layout-parser!76
2023-10-13 12:21:04 +02:00
Dominique Eifländer 8647cf5a18 RED-7759: Upgraded storage-commons to newest windwos compatible version 2023-10-13 12:15:22 +02:00
Kilian Schüttler 310c07b200 Merge branch 'RED-7607-WIP' into 'main'
RED-7607 - Rotating pages leads to lost annotations (RM & DM)

See merge request fforesight/layout-parser!75
2023-10-05 13:34:12 +02:00
Corina Olariu daba0bf8a6 RED-7607 - Rotating pages leads to lost annotations (RM & DM)
- remove finally clause
2023-10-04 17:46:46 +03:00
Corina Olariu 3839de215c RED-7607 - Rotating pages leads to lost annotations (RM & DM)
- rollback to getDir().getDegrees()
2023-10-04 15:27:13 +03:00
Corina Olariu b4d68594f1 RED-7607 - Rotating pages leads to lost annotations (RM & DM)
- use rotation instead of getDir().getDegrees()
2023-10-04 14:22:15 +03:00
Corina Olariu 99ed331a1e RED-7607 - Rotating pages leads to lost annotations (RM & DM)
- use getXDirAdj instead of getX
- add fontSizeCounter for landscape pages also
2023-10-04 14:13:38 +03:00
Corina Olariu f2c0991987 RED-7607 - Rotating pages leads to lost annotations (RM & DM)
- fix PMD findings
2023-10-04 14:09:46 +03:00
Timo Bejan b8ef55e6e2 Merge branch 'TAAS-104' into 'main'
TAAS-104: merge visually intersecting Paragraphs

See merge request fforesight/layout-parser!73
2023-09-05 17:08:40 +02:00
Kilian Schuettler 5792ff4a93 TAAS-104: merge visually intersecting Paragraphs
* fix build
2023-09-05 16:54:23 +02:00
Kilian Schuettler 621c3f269d TAAS-104: merge visually intersecting Paragraphs 2023-09-05 16:09:05 +02:00
Dominique Eifländer 8dba392904 Merge branch 'RED-7461' into 'main'
RED-7461: Fixed wrong textblock classifation if footer is marked as header

See merge request fforesight/layout-parser!72
2023-09-01 12:14:38 +02:00
deiflaender 306a53ea79 RED-7461: Fixed wrong textblock classifation if footer is marked as header 2023-09-01 12:07:47 +02:00
Kilian Schüttler 754fd8f933 Merge branch 'TAAS-89' into 'main'
TAAS-89: added log entry and an end2end test

See merge request fforesight/layout-parser!71
2023-08-31 14:40:48 +02:00
Kilian Schuettler 28ec4c9ccb TAAS-89: added log entry and an end2end test 2023-08-31 14:28:18 +02:00
Kilian Schüttler aed4a55787 Merge branch 'TAAS-89' into 'main'
TAAS-89: fixed weird bug with empty sections

See merge request fforesight/layout-parser!70
2023-08-31 12:00:18 +02:00
Kilian Schuettler f87e2d75b5 TAAS-89: fixed weird bug with empty sections 2023-08-31 11:41:22 +02:00
Kilian Schüttler de6760abc1 Merge branch 'TAAS-89' into 'main'
TAAS-89: added some more documentation

See merge request fforesight/layout-parser!69
2023-08-31 10:55:45 +02:00
Kilian Schuettler 261ef4c367 TAAS-89: added some more documentation
* fixed weird bug with empty sections
2023-08-31 10:49:32 +02:00
Timo Bejan 11ba9c6bb9 Merge branch 'TAAS-89' into 'main'
Added some documentation

See merge request fforesight/layout-parser!64
2023-08-25 16:34:18 +02:00
Renovate Bot b7c3d02978 Merge branch 'renovate/main-spring-boot' into 'main'
Update spring boot to v3.1.3 (main)

See merge request fforesight/layout-parser!63
2023-08-24 21:15:35 +02:00
Kilian Schuettler bcf0bcbaf4 Added some documentation 2023-08-24 18:37:47 +02:00
Renovate Bot 84cde2a3db Update spring boot to v3.1.3 2023-08-24 13:16:58 +00:00
Renovate Bot 6f2dd4f823 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.40.0 (main)

See merge request fforesight/layout-parser!61
2023-08-24 09:17:14 +02:00
Renovate Bot a909724217 Update dependency com.iqser.red.commons:storage-commons to v2.40.0 2023-08-24 04:16:37 +00:00
Renovate Bot 67a981e7a8 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.536 (main)

See merge request fforesight/layout-parser!60
2023-08-24 03:15:54 +02:00
Renovate Bot 0e93fdd515 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.536 2023-08-23 22:17:08 +00:00
Renovate Bot d464239f9b Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.144.0 (main)

See merge request fforesight/layout-parser!59
2023-08-24 00:16:53 +02:00
Renovate Bot 88a20924b9 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.144.0 2023-08-23 19:14:51 +00:00
Renovate Bot f89243472c Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.39.0 (main)

See merge request fforesight/layout-parser!58
2023-08-23 09:17:15 +02:00
Renovate Bot ad3612acd4 Update dependency com.iqser.red.commons:storage-commons to v2.39.0 2023-08-23 04:17:21 +00:00
Renovate Bot 630eee6bd7 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.535 (main)

See merge request fforesight/layout-parser!57
2023-08-23 03:13:39 +02:00
Renovate Bot a951911ec8 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.535 2023-08-22 22:16:17 +00:00
Renovate Bot 75e6b88705 Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.140.0 (main)

See merge request fforesight/layout-parser!56
2023-08-22 21:18:37 +02:00
Renovate Bot 2e0adbdd9a Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.140.0 2023-08-22 16:17:49 +00:00
Renovate Bot b747742558 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.38.0 (main)

See merge request fforesight/layout-parser!55
2023-08-22 15:17:43 +02:00
Renovate Bot 192c9976c1 Update dependency com.iqser.red.commons:storage-commons to v2.38.0 2023-08-22 10:17:05 +00:00
Dominique Eifländer b251697492 Merge branch 'PDFBox-update' into 'main'
upgrade PDFBox to 3.0.0

See merge request fforesight/layout-parser!52
2023-08-22 09:39:53 +02:00
Renovate Bot 22d6b25fe4 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.534 (main)

See merge request fforesight/layout-parser!53
2023-08-22 09:17:26 +02:00
Renovate Bot e6bcd6fb2b Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.534 2023-08-22 04:17:20 +00:00
Renovate Bot 2847adde22 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.37.0 (main)

See merge request fforesight/layout-parser!54
2023-08-22 06:16:54 +02:00
Renovate Bot 7cf67d7121 Update dependency com.iqser.red.commons:storage-commons to v2.37.0 2023-08-22 01:17:23 +00:00
Kilian Schuettler 3a18923ef5 upgrade PDFBox to 3.0.0
* disable experimental ruling header stuff
2023-08-21 17:54:20 +02:00
Kilian Schuettler 2b15fd1d3c RED-7461: improve header/footer recognition 2023-08-21 17:49:13 +02:00
Dominique Eifländer 3722fff476 Merge branch 'RED-7461' into 'main'
Red 7461

See merge request fforesight/layout-parser!51
2023-08-21 17:08:10 +02:00
deiflaender 0cb8029f0a RED-7461: Fixed pr findings 2023-08-21 16:57:37 +02:00
deiflaender b270b9c942 RED-7461: Use marked content to classify headers and footers if available 2023-08-21 16:02:24 +02:00
deiflaender 60615ec5d8 RED-7461: First working iteration of header and footer improvement 2023-08-21 15:31:11 +02:00
Renovate Bot 880914a167 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.36.0 (main)

See merge request fforesight/layout-parser!50
2023-08-19 09:17:27 +02:00
Renovate Bot a80a93d2b0 Update dependency com.iqser.red.commons:storage-commons to v2.36.0 2023-08-19 04:15:48 +00:00
Renovate Bot 0afa7e5b12 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.533 (main)

See merge request fforesight/layout-parser!49
2023-08-19 03:14:12 +02:00
Renovate Bot 12516ebf22 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.533 2023-08-18 22:13:37 +00:00
Renovate Bot 0dca90c3fe Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.138.0 (main)

See merge request fforesight/layout-parser!48
2023-08-18 21:14:23 +02:00
Renovate Bot 2506c9e091 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.138.0 2023-08-18 16:15:53 +00:00
Timo Bejan 83d39ba3a5 Fixed issue with weird colors 2023-08-18 16:21:45 +03:00
Renovate Bot c09bb06da6 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.35.0 (main)

See merge request fforesight/layout-parser!46
2023-08-18 09:17:56 +02:00
Renovate Bot 1793b1138e Update dependency com.iqser.red.commons:storage-commons to v2.35.0 2023-08-18 04:17:52 +00:00
Renovate Bot d30735bc49 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.532 (main)

See merge request fforesight/layout-parser!45
2023-08-18 03:12:57 +02:00
Renovate Bot 9356db5373 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.532 2023-08-17 22:15:27 +00:00
Renovate Bot ee766e7150 Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.135.0 (main)

See merge request fforesight/layout-parser!44
2023-08-17 21:15:12 +02:00
Renovate Bot a33bbc9abc Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.135.0 2023-08-17 16:16:19 +00:00
Renovate Bot 5758295fac Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.132.0 (main)

See merge request fforesight/layout-parser!43
2023-08-17 12:15:34 +02:00
Renovate Bot 8142d0aa09 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.132.0 2023-08-17 07:17:11 +00:00
Renovate Bot dc80353a5b Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.34.0 (main)

See merge request fforesight/layout-parser!42
2023-08-17 09:16:50 +02:00
Renovate Bot 4d856b04b3 Update dependency com.iqser.red.commons:storage-commons to v2.34.0 2023-08-17 04:16:30 +00:00
Renovate Bot 6ba25ecaa0 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.531 (main)

See merge request fforesight/layout-parser!41
2023-08-17 06:16:08 +02:00
Renovate Bot fcdcaf16e9 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.531 2023-08-16 22:15:33 +00:00
Renovate Bot 086e338f4a Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.131.0 (main)

See merge request fforesight/layout-parser!40
2023-08-17 00:15:20 +02:00
Renovate Bot 9dbe73f376 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.131.0 2023-08-16 19:16:16 +00:00
Renovate Bot aaf4015c95 Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.130.0 (main)

See merge request fforesight/layout-parser!39
2023-08-16 18:16:44 +02:00
Renovate Bot 2b65ad4b4b Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.130.0 2023-08-16 13:13:52 +00:00
Renovate Bot c0c75f6a0e Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.33.0 (main)

See merge request fforesight/layout-parser!38
2023-08-16 09:12:41 +02:00
Renovate Bot 2f4af6e377 Update dependency com.iqser.red.commons:storage-commons to v2.33.0 2023-08-16 04:13:49 +00:00
Renovate Bot b9a305bf2d Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.530 (main)

See merge request fforesight/layout-parser!37
2023-08-16 03:13:39 +02:00
Renovate Bot db6b6af4d7 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.530 2023-08-15 22:12:09 +00:00
Renovate Bot d73addf7ed Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.127.0 (main)

See merge request fforesight/layout-parser!36
2023-08-15 18:14:12 +02:00
Renovate Bot c7978c93c2 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.127.0 2023-08-15 13:14:00 +00:00
Kilian Schüttler 457f7d9c66 Merge branch 'RED-7158' into 'main'
RED-7158: fix for all page rotations

See merge request fforesight/layout-parser!35
2023-08-15 15:07:04 +02:00
Kilian Schuettler 0387cdd143 RED-7158: fix for all page rotations
* also make lines thinner
2023-08-15 14:55:41 +02:00
Kilian Schüttler 5c6898b975 Merge branch 'RED-7158' into 'main'
RED-7158: add layoutgrid into new ViewerDocument as optional content

See merge request fforesight/layout-parser!34
2023-08-15 13:22:06 +02:00
Kilian Schuettler b7b273b47d RED-7158: layout grid
* downgraded storage-commons to working version
2023-08-15 13:15:44 +02:00
Kilian Schuettler 9aa9cb2d54 RED-7158: add layoutgrid into new ViewerDocument as optional content
* set layer to invisible by default
2023-08-15 13:14:16 +02:00
Renovate Bot ee6c21638f Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.32.0 (main)

See merge request fforesight/layout-parser!33
2023-08-15 09:12:24 +02:00
Renovate Bot 1e4475afdf Update dependency com.iqser.red.commons:storage-commons to v2.32.0 2023-08-15 04:13:57 +00:00
Renovate Bot 708d274ebc Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.529 (main)

See merge request fforesight/layout-parser!32
2023-08-15 03:13:02 +02:00
Renovate Bot a94faad870 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.529 2023-08-14 22:12:08 +00:00
Kilian Schüttler d854125867 Merge branch 'RED-7158' into 'main'
RED-7158: add layoutgrid into new ViewerDocument as optional content

See merge request fforesight/layout-parser!31
2023-08-14 16:14:23 +02:00
Kilian Schuettler 63de8ef82d RED-7158: add layoutgrid into new ViewerDocument as optional content
* downgraded storage-commons
2023-08-14 16:07:11 +02:00
Kilian Schuettler ea0af08c31 RED-7851: add layoutgrid to new viewer document as optional content 2023-08-14 16:06:23 +02:00
Renovate Bot 810caa0624 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.31.0 (main)

See merge request fforesight/layout-parser!30
2023-08-12 09:14:34 +02:00
Renovate Bot c282372dc8 Update dependency com.iqser.red.commons:storage-commons to v2.31.0 2023-08-12 04:15:18 +00:00
Renovate Bot 055ccd3366 Merge branch 'renovate/main-plugins-(non-major)' into 'main'
Update plugin io.freefair.lombok to v8.2.2 (main)

See merge request fforesight/layout-parser!29
2023-08-12 06:14:53 +02:00
Renovate Bot 4b4c73fb7b Update plugin io.freefair.lombok to v8.2.2 2023-08-12 01:13:35 +00:00
Renovate Bot 35b9cfd1c2 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.528 (main)

See merge request fforesight/layout-parser!28
2023-08-12 03:13:18 +02:00
Renovate Bot 9a73b952cf Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.528 2023-08-11 22:13:57 +00:00
Renovate Bot a1c73094f1 Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.126.0 (main)

See merge request fforesight/layout-parser!27
2023-08-11 18:17:30 +02:00
Renovate Bot b79d9946a9 Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.126.0 2023-08-11 13:13:58 +00:00
Renovate Bot a9735daa04 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.30.0 (main)

See merge request fforesight/layout-parser!26
2023-08-11 15:13:39 +02:00
Renovate Bot d00491c15e Update dependency com.iqser.red.commons:storage-commons to v2.30.0 2023-08-11 10:15:55 +00:00
deiflaender ed48b6a4bf RED-6725: Fixed wrong file encoding in container, that leads to not working rules on terms with special chars 2023-08-11 11:16:07 +02:00
Renovate Bot 62eade84b9 Merge branch 'renovate/main-plugins-(non-major)' into 'main'
Update plugin io.freefair.lombok to v8.2.1 (main)

See merge request fforesight/layout-parser!25
2023-08-11 09:13:51 +02:00
Renovate Bot d6a217fe70 Update plugin io.freefair.lombok to v8.2.1 2023-08-11 04:15:01 +00:00
Renovate Bot f1e4d0d52b Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.527 (main)

See merge request fforesight/layout-parser!24
2023-08-11 06:14:44 +02:00
Renovate Bot 38a1e8b95f Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.527 2023-08-11 01:14:44 +00:00
Renovate Bot a371558f8c Merge branch 'renovate/main-plugins-(non-major)' into 'main'
Update plugin io.spring.dependency-management to v1.1.3 (main)

See merge request fforesight/layout-parser!23
2023-08-10 21:14:45 +02:00
Renovate Bot b716b187eb Update plugin io.spring.dependency-management to v1.1.3 2023-08-10 16:14:50 +00:00
Kilian Schuettler 0be6454a7e add script for pushing custom images 2023-08-10 15:47:52 +02:00
Renovate Bot 5fde631e04 Merge branch 'renovate/main-spring-boot' into 'main'
Update spring boot to v3.1.2 (main)

See merge request fforesight/layout-parser!20
2023-08-10 12:15:11 +02:00
Renovate Bot c076c10840 Update spring boot to v3.1.2 2023-08-10 07:14:23 +00:00
Renovate Bot 24104f8cc1 Merge branch 'renovate/main-com.iqser.red.commons-storage-commons-2.x' into 'main'
Update dependency com.iqser.red.commons:storage-commons to v2.29.0 (main)

See merge request fforesight/layout-parser!17
2023-08-10 09:14:03 +02:00
Renovate Bot 3632dd4667 Update dependency com.iqser.red.commons:storage-commons to v2.29.0 2023-08-10 04:14:44 +00:00
Renovate Bot 063aa8bfe1 Merge branch 'renovate/main-com.iqser.red.commons-jackson-commons-1.x' into 'main'
Update dependency com.iqser.red.commons:jackson-commons to v1.3.0 (main)

See merge request fforesight/layout-parser!16
2023-08-10 06:14:25 +02:00
Renovate Bot d2716a60e9 Update dependency com.iqser.red.commons:jackson-commons to v1.3.0 2023-08-10 01:13:56 +00:00
Renovate Bot 8f08a8c62b Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.526 (main)

See merge request fforesight/layout-parser!22
2023-08-10 03:13:38 +02:00
Renovate Bot 091cb73622 Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.526 2023-08-09 22:14:12 +00:00
Renovate Bot d3b0bc430f Merge branch 'renovate/main-plugins-(non-major)' into 'main'
Update Plugins (non-major) (main)

See merge request fforesight/layout-parser!15
2023-08-10 00:13:58 +02:00
Renovate Bot e3c12bc1bb Update Plugins (non-major) 2023-08-09 19:14:15 +00:00
Renovate Bot f6f7a0a952 Merge branch 'renovate/main-jacksonversion' into 'main'
Update jacksonVersion to v2.15.2 (main)

See merge request fforesight/layout-parser!13
2023-08-09 21:13:57 +02:00
Renovate Bot 96df6e3145 Update jacksonVersion to v2.15.2 2023-08-09 16:12:54 +00:00
Renovate Bot 574e5ad425 Merge branch 'renovate/main-aws-java-sdk-monorepo' into 'main'
Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.525 (main)

See merge request fforesight/layout-parser!12
2023-08-09 18:12:37 +02:00
Renovate Bot 0611e56baa Update dependency com.amazonaws:aws-java-sdk-s3 to v1.12.525 2023-08-09 13:13:33 +00:00
Kilian Schüttler 442c1dafea Merge branch 'update-pdfbox' into 'main'
update PDFBox Version

See merge request fforesight/layout-parser!19
2023-08-09 12:48:14 +02:00
Kilian Schüttler 33bc532eac Merge branch 'renovate/main-com.iqser.red.service-persistence-service-shared-api-v1-2.x' into 'main'
Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.124.0 (main)

See merge request fforesight/layout-parser!18
2023-08-09 12:42:38 +02:00
Kilian Schuettler 4bd6e7e343 update PDFBox Version 2023-08-09 12:41:28 +02:00
Renovate Bot 159ac6348c Update dependency com.iqser.red.service:persistence-service-shared-api-v1 to v2.124.0 2023-08-09 10:14:27 +00:00
Kilian Schuettler 17259ed805 add renovate, fix checkstyle 2023-08-09 10:11:02 +02:00
Dominique Eifländer 67bf5cbaa8 Merge branch 'RED-6725' into 'main'
RED-6725: Install fonts

See merge request fforesight/layout-parser!11
2023-08-09 10:07:05 +02:00
deiflaender f8a3cbc147 RED-6725: Install fonts 2023-08-09 10:00:07 +02:00
Andrei Isvoran a3d4fbe3a3 Merge branch 'RED-6864' into 'main'
RED-6864 - Switch to DELETE_ON_CLOSE

See merge request fforesight/layout-parser!10
2023-08-09 08:37:07 +02:00
Andrei Isvoran 5c1dca5933 RED-6864 - Switch to DELETE_ON_CLOSE 2023-08-09 09:30:37 +03:00
Timo Bejan f56ab8fa49 Merge branch 'RED-6864' into 'main'
RED-6864 - Switch to new storage-commons download

See merge request fforesight/layout-parser!9
2023-08-08 17:16:40 +02:00
Andrei Isvoran cfca5376a0 RED-6864 - Switch to new storage-commons download 2023-08-08 17:16:40 +02:00
Kevin Tumma 0633fa04fb Update file .gitlab-ci.yml 2023-08-08 13:00:40 +02:00
Dominique Eifländer 659a9abaa5 Merge branch 'DM-165' into 'main'
DM-165: Fixed numberFormatException on german local machines

See merge request fforesight/layout-parser!8
2023-08-07 12:29:37 +02:00
deiflaender 5877aea3f7 DM-165: Fixed numberFormatException on german local machines 2023-08-07 12:12:00 +02:00
deiflaender f2b92de827 DM-165: Fixed indexOutOfBounds error in TableNodeFactory 2023-08-05 10:20:05 +02:00
Kilian Schuettler 4a5464d6aa Refactoring to make downstream refactoring easier 2023-08-04 15:16:36 +02:00
Dominique Eifländer d9a3bbbd30 Merge branch 'RED-5253' into 'main'
RED-5253: Ported last documine changes

See merge request fforesight/layout-parser!7
2023-08-04 09:59:57 +02:00
deiflaender 150aea55c0 RED-5253: Ported last documine changes 2023-08-04 09:55:35 +02:00
Kilian Schuettler 676f0c9d09 cleanup dependency versions 2023-08-01 10:27:14 +02:00
Kilian Schuettler ded00df11e fix build 2023-08-01 09:57:58 +02:00
Kilian Schuettler 286556cbb6 mark non nullable fields in request 2023-08-01 00:50:33 +02:00
Kilian Schuettler d6a74dc9f9 add field id to image data 2023-07-31 16:32:11 +02:00
Kilian Schuettler 2a55654fcf add simplifiedText 2023-07-31 15:30:03 +02:00
Kilian Schüttler 7496914b37 Merge branch 'RED-6725' into 'main'
include openfeign for tenant-commons

See merge request fforesight/layout-parser!6
2023-07-31 15:02:33 +02:00
Kilian Schuettler c8ace585e1 include openfeign for tenant-commons 2023-07-31 14:40:57 +02:00
Kilian Schüttler 69c5f80c8c Merge branch 'RED-6725' into 'main'
Red 6725

See merge request fforesight/layout-parser!5
2023-07-31 13:19:39 +02:00
Kilian Schuettler 79d27189fd move Properties names to DocumentStructure 2023-07-31 12:54:49 +02:00
Kilian Schuettler 75bac72c05 fix dsljson 2023-07-29 02:44:22 +02:00
Kilian Schuettler c5e6271dc3 fix fileIds 2023-07-29 01:59:32 +02:00
Kilian Schüttler 5561dd5e95 Merge branch 'RED-6725' into 'main'
Red 6725

See merge request fforesight/layout-parser!4
2023-07-28 17:46:40 +02:00
Kilian Schuettler 041b633742 use correct repo 2023-07-28 17:42:18 +02:00
Kilian Schuettler 715426bd3b remove hardcoded version 2023-07-28 17:34:59 +02:00
Kilian Schuettler 464b8053fe configure maven-publish plugin 2023-07-28 17:26:05 +02:00
Kilian Schuettler 2fece83c7c remove root build.gradle 2023-07-28 16:15:53 +02:00
Kilian Schuettler ad03ef1922 move queue to server package, so i can easily import processor as a library for redaction-service 2023-07-28 16:15:31 +02:00
Christoph Schabert cc44100e4e add root build.gradle.kts 2023-07-28 13:46:06 +02:00
Christoph Schabert 5d1c1ae406 Update gradle.properties.kts 2023-07-27 18:00:31 +02:00
Kilian Schuettler f72838b0be removed unnecessary dependency 2023-07-27 17:15:22 +02:00
Kilian Schuettler 6388898cc0 added comments for native build investigation 2023-07-27 17:12:16 +02:00
194 changed files with 6211 additions and 1339 deletions
+4
View File
@@ -38,3 +38,7 @@ build/
### VS Code ###
.vscode/
gradlew.bat
gradlew
gradle.properties
gradle/
+18 -1
View File
@@ -1,4 +1,21 @@
include:
- project: 'gitlab/gitlab'
ref: 'main'
file: 'ci-templates/gradle_java.yml'
file: 'ci-templates/gradle_java.yml'
deploy:
stage: deploy
tags:
- dind
script:
- echo "Building with gradle version ${BUILDVERSION}"
- gradle -Pversion=${BUILDVERSION} publish
- gradle bootBuildImage --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
- echo "BUILDVERSION=$BUILDVERSION" >> version.env
artifacts:
reports:
dotenv: version.env
rules:
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
- if: $CI_COMMIT_BRANCH =~ /^release/
- if: $CI_COMMIT_TAG
+88
View File
@@ -1 +1,89 @@
# PDF Layout Parser Micro-Service: layout-parser
## Introduction
The layout-parser micro-service is a powerful tool designed to efficiently extract structured information from PDF documents. Written in Java and utilizing Spring Boot 3, Apache PDFBox, and RabbitMQ, this micro-service excels at parsing PDFs and organizing their content into a meaningful and coherent layout structure. Notably, the layout-parser micro-service distinguishes itself by relying solely on advanced algorithms, rather than machine learning techniques.
### Key Steps in the PDF Layout Parsing Process:
* **Text Position Extraction:**
The micro-service leverages Apache PDFBox to extract precise text positions for each individual character within the PDF document.
* **Word Segmentation and Text Block Formation:**
Employing an array of diverse algorithms, the micro-service initially identifies and segments words, creating distinct text blocks.
* **Text Block Classification:**
The segmented text blocks are then subjected to classification algorithms. These algorithms categorize the text blocks based on their content and visual properties, distinguishing between sections, subsections, headlines, paragraphs, images, tables, table cells, headers, and footers.
* **Layout Coherence Establishment:**
The classified text blocks are subsequently orchestrated into a cohesive layout structure. This process involves arranging sections, subsections, paragraphs, images, and other elements in a logical and structured manner.
* **Output Generation in Various Formats:**
Once the layout structure is established, the micro-service generates output in multiple formats. These formats are designed for seamless integration with downstream micro-services. The supported formats include JSON, XML, and others, ensuring flexibility in downstream data consumption.
### Optional Enhancements:
* **ML-Based Table Extraction:**
For enhanced results, users have the option to incorporate machine learning-based table extraction. This feature can be activated by providing ML-generated results as a JSON file, which are then integrated seamlessly into the layout structure.
* **Image Classification using ML:**
Additionally, for more accurate image classification, users can optionally feed ML-generated image classification results into the micro-service. Similar to the table extraction option, the micro-service processes the pre-parsed results in JSON format, thus optimizing the accuracy of image content identification.
In conclusion, the layout-parser micro-service is a versatile PDF layout parsing solution crafted entirely around advanced algorithms, without reliance on machine learning. It proficiently extracts text positions, segments content into meaningful blocks, classifies these blocks, arranges them coherently, and outputs structured data for downstream micro-services. Optional integration with ML-generated table extractions and image classifications further enhances its capabilities.
## Installation
### Prerequisites
Before building and using the layout-parser micro-service, please ensure you have the following software and tools installed:
Java Development Kit (JDK) 17 or later
Gradle build tool (preinstalled)
Build and Test
To build and test the micro-service, follow these steps:
### Clone the Repository:
bash
```
git clone ssh://git@git.knecon.com:22222/fforesight/layout-parser.git
cd layout-parser
```
### Build the Project:
Use the following command to build the project using Gradle:
```
gradle clean build
```
### Run Tests:
Run the test suite using the following command:
```
gradle test
```
## Building a Custom Docker Image
To create a custom Docker image for the layout-parser micro-service, execute the provided script:
### Ensure Docker is Installed:
Ensure that Docker is installed and running on your system.
### Run the Image Building Script:
Execute the publish-custom-image script in the project directory:
```
./publish-custom-image
```
## Publishing to Internal Maven Repository
To publish the layout-parser micro-service to your internal Maven repository, execute the following command:
```
gradle -Pversion=buildVersion publish
```
Replace buildVersion with the desired version number.
## Additional Notes
Make sure to configure any necessary application properties before deploying the micro-service.
For advanced usage and configurations, refer to Kilian or Dom or preferably the source code.
@@ -1,24 +1,16 @@
plugins {
java
`java-library`
`maven-publish`
pmd
checkstyle
jacoco
}
group = "com.knecon.fforesight"
version = "0.1-SNAPSHOT"
java.sourceCompatibility = JavaVersion.VERSION_17
java.targetCompatibility = JavaVersion.VERSION_17
tasks.jacocoTestReport {
reports {
xml.required.set(false)
csv.required.set(false)
html.outputLocation.set(layout.buildDirectory.dir("jacocoHtml"))
}
}
tasks.pmdMain {
pmd.ruleSetFiles = files("${rootDir}/config/pmd/pmd.xml")
}
@@ -29,6 +21,11 @@ tasks.pmdTest {
tasks.named<Test>("test") {
useJUnitPlatform()
reports {
junitXml.outputLocation.set(layout.buildDirectory.dir("reports/junit"))
}
minHeapSize = "512m"
maxHeapSize = "2048m"
}
tasks.test {
@@ -40,14 +37,38 @@ tasks.jacocoTestReport {
reports {
xml.required.set(true)
csv.required.set(false)
html.outputLocation.set(layout.buildDirectory.dir("jacocoHtml"))
}
}
allprojects {
publishing {
publications {
create<MavenPublication>(name) {
from(components["java"])
}
}
repositories {
maven {
url = uri("https://nexus.knecon.com/repository/red-platform-releases/")
credentials {
username = providers.gradleProperty("mavenUser").getOrNull();
password = providers.gradleProperty("mavenPassword").getOrNull();
}
}
}
}
}
java {
withJavadocJar()
}
repositories {
mavenLocal()
mavenCentral()
maven {
url = uri("https://nexus.knecon.com/repository/gindev/");
url = uri("https://nexus.knecon.com/repository/gindev/")
credentials {
username = providers.gradleProperty("mavenUser").getOrNull();
password = providers.gradleProperty("mavenPassword").getOrNull();
+5 -5
View File
@@ -9,13 +9,13 @@
</description>
<rule ref="category/java/errorprone.xml">
<exclude name="DataflowAnomalyAnalysis"/>
<exclude name="MissingSerialVersionUID"/>
<exclude name="NullAssignment"/>
<exclude name="BeanMembersShouldSerialize"/>
<exclude name="AvoidLiteralsInIfCondition"/>
<exclude name="AvoidDuplicateLiterals"/>
<exclude name="AvoidFieldNameMatchingMethodName"/>
<exclude name="NullAssignment"/>
<exclude name="AssignmentInOperand"/>
<exclude name="BeanMembersShouldSerialize"/>
</rule>
</ruleset>
</ruleset>
+5 -5
View File
@@ -10,14 +10,14 @@
<rule ref="category/java/errorprone.xml">
<exclude name="DataflowAnomalyAnalysis"/>
<exclude name="MissingSerialVersionUID"/>
<exclude name="NullAssignment"/>
<exclude name="BeanMembersShouldSerialize"/>
<exclude name="AvoidLiteralsInIfCondition"/>
<exclude name="AvoidDuplicateLiterals"/>
<exclude name="AvoidFieldNameMatchingMethodName"/>
<exclude name="AvoidFieldNameMatchingTypeName"/>
<exclude name="NullAssignment"/>
<exclude name="AssignmentInOperand"/>
<exclude name="TestClassWithoutTestCases"/>
<exclude name="BeanMembersShouldSerialize"/>
</rule>
</ruleset>
</ruleset>
+1 -1
View File
@@ -1 +1 @@
version = 0.1.0
version = 0.1-SNAPSHOT
@@ -1,6 +1,10 @@
plugins {
id("com.knecon.fforesight.java-conventions")
id("io.freefair.lombok") version "8.1.0"
id("io.freefair.lombok") version "8.4"
}
description = "layoutparser-service-internal-api"
dependencies {
implementation("io.swagger.core.v3:swagger-annotations:2.2.15")
}
@@ -1,5 +1,8 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.io.Serializable;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -10,11 +13,16 @@ import lombok.experimental.FieldDefaults;
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class DocumentData {
@Schema(description = "Object containing the complete document layout parsing information. It is split into 4 categories, structure, text, positions and pages: " + "The document tree structure of SemanticNodes such as Section, Paragraph, Headline, etc. " + "The text, which is stored as separate blocks of data. " + "The text positions, which are also stored as separate blocks. The Blocks are equal to the text blocks in length and order. " + "The page information.")
public class DocumentData implements Serializable {
@Schema(description = "Contains information about the document's pages.")
DocumentPage[] documentPages;
@Schema(description = "Contains information about the document's text.")
DocumentTextData[] documentTextData;
@Schema(description = "Contains information about the document's text positions.")
DocumentPositionData[] documentPositions;
@Schema(description = "Contains information about the document's semantic structure.")
DocumentStructure documentStructure;
}
@@ -1,5 +1,8 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.io.Serializable;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -12,11 +15,16 @@ import lombok.experimental.FieldDefaults;
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DocumentPage {
@Schema(description = "Object containing information about the document's pages.")
public class DocumentPage implements Serializable {
@Schema(description = "The page number, starting with 1.")
int number;
@Schema(description = "The page height in PDF user units.", example = "792")
int height;
@Schema(description = "The page width in PDF user units.", example = "694")
int width;
@Schema(description = "The page rotation as specified by the PDF.", example = "90", allowableValues = {"0", "90", "180", "270"})
int rotation;
}
@@ -1,5 +1,8 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.io.Serializable;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -12,10 +15,14 @@ import lombok.experimental.FieldDefaults;
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DocumentPositionData {
@Schema(description = "Object containing text positional information of a specific text block. A document is split into multiple text blocks, which are supposed to be read in order. Every text block can only occur on a single page.")
public class DocumentPositionData implements Serializable {
@Schema(description = "Identifier of the text block.")
Long id;
@Schema(description = "For each string coordinate in the search text of the text block, the array contains an entry relating the string coordinate to the position coordinate. This is required due to the text and position coordinates not being equal.")
int[] stringIdxToPositionIdx;
@Schema(description = "The bounding box for each glyph as a rectangle. This matrix is of size (n,4), where n is the number of glyphs in the text block. The second dimension specifies the rectangle with the value x, y, width, height, with x, y specifying the lower left corner. In order to access this information, the stringIdxToPositionIdx array must be used to transform the coordinates.")
float[][] positions;
}
@@ -1,9 +1,13 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.awt.geom.Rectangle2D;
import java.io.Serializable;
import java.util.Arrays;
import java.util.List;
import java.util.Map;
import java.util.stream.Stream;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -16,10 +20,49 @@ import lombok.experimental.FieldDefaults;
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DocumentStructure {
@Schema(description = "Object containing information about the parsed tree structure of the SemanticNodes, such as Section, Paragraph, Headline etc inside of the document.")
public class DocumentStructure implements Serializable {
@Schema(description = "The root EntryData represents the Document.")
EntryData root;
@Schema(description = "Object containing the extra field names, a table has in its properties field.")
public static class TableProperties implements Serializable {
public static final String NUMBER_OF_ROWS = "numberOfRows";
public static final String NUMBER_OF_COLS = "numberOfCols";
}
@Schema(description = "Object containing the extra field names, an Image has in its properties field.")
public static class ImageProperties implements Serializable {
public static final String TRANSPARENT = "transparent";
public static final String IMAGE_TYPE = "imageType";
public static final String POSITION = "position";
public static final String ID = "id";
}
@Schema(description = "Object containing the extra field names, a table cell has in its properties field.")
public static class TableCellProperties implements Serializable {
public static final String B_BOX = "bBox";
public static final String ROW = "row";
public static final String COL = "col";
public static final String HEADER = "header";
}
public static final String RECTANGLE_DELIMITER = ";";
public static Rectangle2D parseRectangle2D(String bBox) {
List<Float> floats = Arrays.stream(bBox.split(RECTANGLE_DELIMITER)).map(Float::parseFloat).toList();
return new Rectangle2D.Float(floats.get(0), floats.get(1), floats.get(2), floats.get(3));
}
public EntryData get(List<Integer> tocId) {
@@ -57,15 +100,23 @@ public class DocumentStructure {
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public static class EntryData {
@Schema(description = "Object containing information of a SemanticNode and also structuring the layout with children.")
public static class EntryData implements Serializable {
@Schema(description = "Type of the semantic node.", allowableValues = {"DOCUMENT", "SECTION", "PARAGRAPH", "HEADLINE", "TABLE", "TABLE_CELL", "HEADER", "FOOTER", "IMAGE"})
NodeType type;
@Schema(description = "Specifies the position in the parsed tree structure.", example = "[1, 0, 2]")
int[] treeId;
@Schema(description = "Specifies the text block IDs associated with this semantic node. The value should be joined with the DocumentTextData/DocumentPositionData. Is empty, if no text block is directly associated with this semantic node. Only Paragraph, Headline, Header or Footer is directly associated with a text block.", example = "[1]")
Long[] atomicBlockIds;
@Schema(description = "Specifies the pages this semantic node appears on. The value should be joined with the PageData.", example = "[1, 2, 3]")
Long[] pageNumbers;
@Schema(description = "Some semantic nodes have additional information, this information is stored in this Map. The extra fields are specified by the Properties subclasses.", example = "For a Table: {\"numberOfRows\": 3, \"numberOfCols\": 4}")
Map<String, String> properties;
@Schema(description = "All child Entries of this Entry.", example = "[1, 2, 3]")
List<EntryData> children;
@Override
public String toString() {
@@ -1,7 +1,8 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.io.Serializable;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -14,14 +15,22 @@ import lombok.experimental.FieldDefaults;
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DocumentTextData {
@Schema(description = "Object containing text information of a specific text block. A document is split into multiple text blocks, which are supposed to be read in order. Every text block can only occur on a single page.")
public class DocumentTextData implements Serializable {
@Schema(description = "Identifier of the text block.")
Long id;
@Schema(description = "The page the text block occurs on.")
Long page;
@Schema(description = "The text the text block.")
String searchText;
@Schema(description = "Each text block is assigned a number on a page, starting from 0.")
int numberOnPage;
@Schema(description = "The text blocks are ordered, this number represents the start of the text block as a string offset.")
int start;
@Schema(description = "The text blocks are ordered, this number represents the end of the text block as a string offset.")
int end;
@Schema(description = "The line breaks in the text of this semantic node in string offsets. They are exclusive end. At the end of each semantic node there is an implicit linebreak.", example = "[5, 10]")
int[] lineBreaks;
}
@@ -1,8 +1,9 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.io.Serializable;
import java.util.Locale;
public enum NodeType {
public enum NodeType implements Serializable {
DOCUMENT,
SECTION,
HEADLINE,
@@ -0,0 +1,21 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@Schema(description = "Object containing a simplified version, which contains almost exclusively text, of the document structure Section class.")
public class SimplifiedSectionText {
@Schema(description = "The number of this Section. This is used to map the simplified section text back to the original Section.")
private int sectionNumber;
@Schema(description = "The text in this Section.")
private String text;
}
@@ -0,0 +1,24 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.redaction;
import java.util.ArrayList;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@Schema(description = "Object containing a simplified version, which contains almost exclusively text, of the document structure.")
public class SimplifiedText {
@Schema(description = "Number of pages in the entire document.")
private int numberOfPages;
@Schema(description = "A List of simplified Sections, which contains almost exclusively text.")
private List<SimplifiedSectionText> sectionTexts = new ArrayList<>();
}
@@ -2,20 +2,29 @@ package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Builder;
import lombok.Data;
@Data
@Builder
@Schema(description = "Object containing information about a Paragraph/Headline/Header/Footer.")
public class ParagraphData {
@Schema(description = "The text of this Semantic Node, without any linebreaks.", example = "This is some text.")
private String text;
@Schema(description = "A list of text ranges in string offsets. Every character in any of the ranges is bold.", example = "[0, 15]")
List<Range> boldTextBoundaries;
@Schema(description = "A list of text ranges in string offsets. Every character in any of the ranges is italic.", example = "[0, 15]")
List<Range> italicTextBoundaries;
@Schema(description = "The line breaks in the text of this semantic node in string offsets. They are exclusive end. At the end of each semantic node there is an implicit linebreak.", example = "[5, 10]")
List<Integer> linebreaks;
@Schema(description = "The classification of this Paragraph.", allowableValues = "{paragraph, headline, header, footer}")
private String classification;
@Schema(description = "Describes the text orientation of this semantic node. Any semantic node only has a single text orientation.", allowableValues = "{ZERO, QUARTER_CIRCLE, HALF_CIRCLE, THREE_QUARTER_CIRCLE}")
private String orientation;
@Schema(description = "Describes the text direction in degrees of this semantic node. Any semantic node only has a single text direction.", minimum = "0", maximum = "359")
private int textDirection;
}
@@ -1,5 +1,8 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import io.swagger.v3.oas.annotations.media.Schema;
@Schema(description = "Object specifying the start and end offsets of a text range in string offsets.")
public record Range(int start, int end) {
}
@@ -2,6 +2,7 @@ package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -9,8 +10,12 @@ import lombok.Data;
@Builder
@Data
@AllArgsConstructor
@Schema(description = "Object containing a simplified version of the document structure. This simplified form only knows Paragraphs and Tables. The Paragraph Objects might be a Paragraph, Headline, Header or Footer.")
public class ResearchDocumentData {
@Schema(description = "File name of the original uploaded file.")
String originalFile;
@Schema(description = "A List of all paragraphs/headline or table objects, that have been parsed in this document.")
List<StructureObject> structureObjects;
}
@@ -2,14 +2,19 @@ package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
@Schema(description = "Object containing information about a Table Row.")
public class RowData {
@Schema(description = "Boolean indicating whether this table row is classified as a header row.")
boolean header;
@Schema(description = "A list of Objects containing information about the text in each cell of this row.")
List<ParagraphData> cellText;
@Schema(description = "The bounding box of this StructureObject. Is always exactly 4 values representing x, y, w, h, where x, y specify the lower left corner.")
float[] bBox;
}
@@ -1,5 +1,6 @@
package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -7,13 +8,20 @@ import lombok.Data;
@Data
@Builder
@AllArgsConstructor
@Schema(description = "Object containing information about either a Paragraph/Headline/Header/Footer or a Table.")
public class StructureObject {
@Schema(description = "The ID of this StructureObject.")
Integer structureObjectNumber;
@Schema(description = "This value indicates the start of the string offsets in this Object, with respect to the reading order.")
int page;
@Schema(description = "This stringOffset indicates the start of the string offsets in this Object, with respect to the reading order of the entire document. It is equal to the previous' StructureObject stringOffset + its length.")
int stringOffset;
@Schema(description = "The bounding box of this StructureObject. Is always exactly 4 values representing x, y, w, h, where x, y specify the lower left corner.", example = "[100, 100, 50, 50]")
float[] boundingBox;
@Schema(description = "Object containing information about a Paragraph/Headline/Header/Footer. Either this or table is null.")
ParagraphData paragraph;
@Schema(description = "Object containing information about a Table. Either this or paragraph is null.")
TableData table;
}
@@ -2,15 +2,20 @@ package com.knecon.fforesight.service.layoutparser.internal.api.data.taas;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
@Schema(description = "Object containing information about a Table.")
public class TableData {
@Schema(description = "A list of Objects containing information about all rows in this table.")
List<RowData> rowData;
@Schema(description = "Number of columns in this table.")
Integer numberOfCols;
@Schema(description = "Number of rows in this table.")
Integer numberOfRows;
}
@@ -2,9 +2,19 @@ package com.knecon.fforesight.service.layoutparser.internal.api.queue;
import java.util.Map;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Builder;
@Builder
public record LayoutParsingFinishedEvent(Map<String, String> identifier, long duration, int numberOfPages, String message) {
@Schema(description = "Object containing information about the layout parsing.")
public record LayoutParsingFinishedEvent(
@Schema(description = "General purpose identifier. It is returned exactly the same way it is inserted with the LayoutParsingRequest.")
Map<String, String> identifier,//
@Schema(description = "The duration of a single layout parsing in ms.")
long duration,//
@Schema(description = "The number of pages of the parsed document.")
int numberOfPages,//
@Schema(description = "A general message. It contains some information useful for a developer, like the paths where the files are stored. Not meant to be machine readable.")
String message) {
}
@@ -3,21 +3,39 @@ package com.knecon.fforesight.service.layoutparser.internal.api.queue;
import java.util.Map;
import java.util.Optional;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Builder;
import lombok.NonNull;
@Builder
@Schema(description = "Object containing all storage paths the service needs to know.")
public record LayoutParsingRequest(
LayoutParsingType layoutParsingType,
@Schema(description = "Enum specifying the type of layout parsing to be performed.", allowableValues = "{RedactManager, DocuMine, TAAS}")//
@NonNull LayoutParsingType layoutParsingType,
@Schema(description = "General purpose identifiers. They are not changed by the service at all and are returned as is in the response queue.")//
Map<String, String> identifier,
String originFileStorageId,
Optional<String> tablesFileStorageId,
Optional<String> imagesFileStorageId,
String structureFileStorageId,
String researchDocumentStorageId,
String textBlockFileStorageId,
String positionBlockFileStorageId,
String pageFileStorageId,
String sectionGridStorageId) {
@Schema(description = "Path to the original PDF file.")//
@NonNull String originFileStorageId,//
@Schema(description = "Optional Path to the table extraction file.")//
Optional<String> tablesFileStorageId,//
@Schema(description = "Optional Path to the image classification file.")//
Optional<String> imagesFileStorageId,//
@Schema(description = "Path where the Document Structure File will be stored.")//
@NonNull String structureFileStorageId,//
@Schema(description = "Path where the Research Data File will be stored.")//
String researchDocumentStorageId,//
@Schema(description = "Path where the Document Text File will be stored.")//
@NonNull String textBlockFileStorageId,//
@Schema(description = "Path where the Document Positions File will be stored.")//
@NonNull String positionBlockFileStorageId,//
@Schema(description = "Path where the Document Pages File will be stored.")//
@NonNull String pageFileStorageId,//
@Schema(description = "Path where the Simplified Text File will be stored.")//
@NonNull String simplifiedTextStorageId,//
@Schema(description = "Path where the Viewer Document PDF will be stored.")//
@NonNull String viewerDocumentStorageId) {
}
@@ -1,21 +1,27 @@
plugins {
id("com.knecon.fforesight.java-conventions")
id("io.freefair.lombok") version "8.1.0"
}
dependencies {
implementation(project(":layoutparser-service-internal-api"))
implementation("com.iqser.red.service:persistence-service-shared-api-v1:2.36.0")
implementation("com.knecon.fforesight:tenant-commons:0.10.0")
implementation("com.iqser.red.commons:storage-commons:2.1.0")
implementation("org.apache.pdfbox:pdfbox:3.0.0-alpha2")
implementation("org.apache.pdfbox:pdfbox-tools:3.0.0-alpha2")
implementation("com.fasterxml.jackson.module:jackson-module-afterburner:2.15.0-rc2")
implementation("com.fasterxml.jackson.datatype:jackson-datatype-jsr310:2.15.0-rc2")
implementation("org.springframework.boot:spring-boot-starter-web:3.0.6")
implementation("org.springframework.boot:spring-boot-starter-amqp:3.0.6")
id("io.freefair.lombok") version "8.4"
}
description = "layoutparser-service-processor"
val jacksonVersion = "2.15.2"
val pdfBoxVersion = "3.0.0"
dependencies {
implementation(project(":layoutparser-service-internal-api"))
implementation(project(":viewer-doc-processor"))
implementation("com.iqser.red.service:persistence-service-shared-api-v1:2.144.0") {
exclude("org.springframework.boot", "spring-boot-starter-security")
exclude("org.springframework.boot", "spring-boot-starter-validation")
}
implementation("com.knecon.fforesight:tenant-commons:0.21.0")
implementation("com.iqser.red.commons:storage-commons:2.45.0")
implementation("org.apache.pdfbox:pdfbox:${pdfBoxVersion}")
implementation("org.apache.pdfbox:pdfbox-tools:${pdfBoxVersion}")
implementation("com.fasterxml.jackson.module:jackson-module-afterburner:${jacksonVersion}")
implementation("com.fasterxml.jackson.datatype:jackson-datatype-jsr310:${jacksonVersion}")
implementation("org.springframework.boot:spring-boot-starter-web:3.1.3")
}
@@ -2,142 +2,389 @@ package com.knecon.fforesight.service.layoutparser.processor;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.io.File;
import java.io.IOException;
import java.nio.file.Files;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.concurrent.atomic.AtomicInteger;
import java.util.concurrent.atomic.AtomicReference;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.documentinterchange.markedcontent.PDMarkedContent;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingFinishedEvent;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingRequest;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationSection;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Document;
import com.knecon.fforesight.service.layoutparser.processor.model.image.ClassifiedImage;
import com.knecon.fforesight.service.layoutparser.processor.model.table.CleanRulings;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.python_api.adapter.CvTableParsingAdapter;
import com.knecon.fforesight.service.layoutparser.processor.python_api.adapter.ImageServiceResponseAdapter;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.image.ImageServiceResponse;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableCells;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableServiceResponse;
import com.knecon.fforesight.service.layoutparser.processor.services.factory.DocumentGraphFactory;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Document;
import com.knecon.fforesight.service.layoutparser.processor.services.mapper.DocumentDataMapper;
import com.knecon.fforesight.service.layoutparser.processor.services.mapper.TaasDocumentDataMapper;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.services.PdfParsingService;
import com.knecon.fforesight.service.layoutparser.processor.services.SectionGridCreatorService;
import com.knecon.fforesight.service.layoutparser.processor.services.BodyTextFrameService;
import com.knecon.fforesight.service.layoutparser.processor.services.RulingCleaningService;
import com.knecon.fforesight.service.layoutparser.processor.services.SectionsBuilderService;
import com.knecon.fforesight.service.layoutparser.processor.services.SimplifiedSectionTextService;
import com.knecon.fforesight.service.layoutparser.processor.services.TableExtractionService;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.DocuMineBlockificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.RedactManagerBlockificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.TaasBlockificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.classification.DocuMineClassificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.classification.RedactManagerClassificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.classification.TaasClassificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.DocstrumSegmentationService;
import com.knecon.fforesight.service.layoutparser.processor.services.factory.DocumentGraphFactory;
import com.knecon.fforesight.service.layoutparser.processor.services.mapper.DocumentDataMapper;
import com.knecon.fforesight.service.layoutparser.processor.services.mapper.TaasDocumentDataMapper;
import com.knecon.fforesight.service.layoutparser.processor.services.parsing.PDFLinesTextStripper;
import com.knecon.fforesight.service.layoutparser.processor.services.visualization.LayoutGridService;
import com.knecon.fforesight.service.layoutparser.processor.utils.MarkedContentUtils;
import io.micrometer.observation.Observation;
import io.micrometer.observation.ObservationRegistry;
import io.micrometer.observation.annotation.Observed;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@SuppressWarnings("PMD.CloseResource")
@Slf4j
@Service
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class LayoutParsingPipeline {
private final ImageServiceResponseAdapter imageServiceResponseAdapter;
private final CvTableParsingAdapter cvTableParsingAdapter;
private final LayoutParsingStorageService layoutParsingStorageService;
private final PdfParsingService pdfParsingService;
private final SectionsBuilderService sectionsBuilderService;
private final SectionGridCreatorService sectionGridCreatorService;
private final TaasClassificationService taasClassificationService;
private final RedactManagerClassificationService redactManagerClassificationService;
private final DocuMineClassificationService docuMineClassificationService;
ImageServiceResponseAdapter imageServiceResponseAdapter;
CvTableParsingAdapter cvTableParsingAdapter;
LayoutParsingStorageService layoutParsingStorageService;
SectionsBuilderService sectionsBuilderService;
TaasClassificationService taasClassificationService;
RedactManagerClassificationService redactManagerClassificationService;
DocuMineClassificationService docuMineClassificationService;
SimplifiedSectionTextService simplifiedSectionTextService;
BodyTextFrameService bodyTextFrameService;
RulingCleaningService rulingCleaningService;
TableExtractionService tableExtractionService;
TaasBlockificationService taasBlockificationService;
DocuMineBlockificationService docuMineBlockificationService;
RedactManagerBlockificationService redactManagerBlockificationService;
LayoutGridService layoutGridService;
ObservationRegistry observationRegistry;
DocstrumSegmentationService docstrumSegmentationService;
public LayoutParsingFinishedEvent parseLayoutAndSaveFilesToStorage(LayoutParsingRequest layoutParsingRequest) throws IOException {
long start = System.currentTimeMillis();
log.info("Starting layout parsing for {}", layoutParsingRequest.identifier());
try (PDDocument originDocument = layoutParsingStorageService.getOriginFile(layoutParsingRequest.originFileStorageId())) {
ImageServiceResponse imageServiceResponse = new ImageServiceResponse();
if (layoutParsingRequest.imagesFileStorageId().isPresent()) {
imageServiceResponse = layoutParsingStorageService.getImagesFile(layoutParsingRequest.pageFileStorageId());
}
File originFile = layoutParsingStorageService.getOriginFile(layoutParsingRequest.originFileStorageId());
File viewerDocumentFile = layoutParsingStorageService.getViewerDocFile(layoutParsingRequest.viewerDocumentStorageId()).orElse(originFile);
TableServiceResponse tableServiceResponse = new TableServiceResponse();
if (layoutParsingRequest.tablesFileStorageId().isPresent()) {
tableServiceResponse = layoutParsingStorageService.getTablesFile(layoutParsingRequest.pageFileStorageId());
}
Document documentGraph = parseLayout(layoutParsingRequest.layoutParsingType(), originDocument, imageServiceResponse, tableServiceResponse);
int numberOfPages = originDocument.getNumberOfPages();
layoutParsingStorageService.storeSectionGrid(layoutParsingRequest, sectionGridCreatorService.createSectionGrid(documentGraph));
layoutParsingStorageService.storeDocumentData(layoutParsingRequest, DocumentDataMapper.toDocumentData(documentGraph));
if (layoutParsingRequest.layoutParsingType().equals(LayoutParsingType.TAAS)) {
var researchDocumentData = TaasDocumentDataMapper.fromDocument(documentGraph);
layoutParsingStorageService.storeResearchDocumentData(layoutParsingRequest, researchDocumentData);
}
return LayoutParsingFinishedEvent.builder()
.identifier(layoutParsingRequest.identifier())
.numberOfPages(numberOfPages)
.duration(System.currentTimeMillis() - start)
.message(format("Layout parsing is finished and files have been saved with Ids:\n Structure: %s\nText: %s\nPositions: %s\nPageData: %s",
layoutParsingRequest.structureFileStorageId(),
layoutParsingRequest.textBlockFileStorageId(),
layoutParsingRequest.positionBlockFileStorageId(),
layoutParsingRequest.pageFileStorageId()))
.build();
ImageServiceResponse imageServiceResponse = new ImageServiceResponse();
if (layoutParsingRequest.imagesFileStorageId().isPresent()) {
imageServiceResponse = layoutParsingStorageService.getImagesFile(layoutParsingRequest.imagesFileStorageId().get());
}
TableServiceResponse tableServiceResponse = new TableServiceResponse();
if (layoutParsingRequest.tablesFileStorageId().isPresent()) {
tableServiceResponse = layoutParsingStorageService.getTablesFile(layoutParsingRequest.tablesFileStorageId().get());
}
ClassificationDocument classificationDocument = parseLayout(layoutParsingRequest.layoutParsingType(),
originFile,
imageServiceResponse,
tableServiceResponse,
layoutParsingRequest.identifier().toString());
log.info("Building document graph for {}", layoutParsingRequest.identifier());
Document documentGraph = observeBuildDocumentGraph(classificationDocument);
log.info("Creating viewer document for {}", layoutParsingRequest.identifier());
layoutGridService.addLayoutGrid(viewerDocumentFile, documentGraph, viewerDocumentFile, false);
log.info("Storing resulting files for {}", layoutParsingRequest.identifier());
layoutParsingStorageService.storeDocumentData(layoutParsingRequest, DocumentDataMapper.toDocumentData(documentGraph));
layoutParsingStorageService.storeSimplifiedText(layoutParsingRequest, simplifiedSectionTextService.toSimplifiedText(documentGraph));
layoutParsingStorageService.storeViewerDocument(layoutParsingRequest, viewerDocumentFile);
if (layoutParsingRequest.layoutParsingType().equals(LayoutParsingType.TAAS)) {
log.info("Building research document data for {}", layoutParsingRequest.identifier());
var researchDocumentData = TaasDocumentDataMapper.fromDocument(documentGraph);
layoutParsingStorageService.storeResearchDocumentData(layoutParsingRequest, researchDocumentData);
}
if (!viewerDocumentFile.equals(originFile)) {
viewerDocumentFile.delete();
}
originFile.delete();
return LayoutParsingFinishedEvent.builder()
.identifier(layoutParsingRequest.identifier())
.numberOfPages(documentGraph.getNumberOfPages())
.duration(System.currentTimeMillis() - start)
.message(format("""
Layout parsing has finished in %.02f s.
identifiers: %s
%s
Files have been saved with Ids:
Structure: %s
Text: %s
Positions: %s
PageData: %s
Simplified Text: %s
Viewer Doc: %s""",
((float) (System.currentTimeMillis() - start)) / 1000,
layoutParsingRequest.identifier(),
buildSemanticNodeCountMessage(documentGraph.getNumberOfPages(), documentGraph.buildSemanticNodeCounts()),
layoutParsingRequest.structureFileStorageId(),
layoutParsingRequest.textBlockFileStorageId(),
layoutParsingRequest.positionBlockFileStorageId(),
layoutParsingRequest.pageFileStorageId(),
layoutParsingRequest.simplifiedTextStorageId(),
layoutParsingRequest.viewerDocumentStorageId()))
.build();
}
public Document parseLayout(LayoutParsingType layoutParsingType,
PDDocument originDocument,
ImageServiceResponse imageServiceResponse,
TableServiceResponse tableServiceResponse) {
private Document observeBuildDocumentGraph(ClassificationDocument classificationDocument) {
ClassificationDocument classificationDocument = pdfParsingService.parseDocument(layoutParsingType,
originDocument,
cvTableParsingAdapter.buildCvParsedTablesPerPage(tableServiceResponse),
imageServiceResponseAdapter.buildClassifiedImagesPerPage(imageServiceResponse));
AtomicReference<Document> documentReference = new AtomicReference<>();
Observation.createNotStarted("LayoutParsingPipeline", observationRegistry).contextualName("build-document-graph").observe(() -> {
documentReference.set(DocumentGraphFactory.buildDocumentGraph(classificationDocument));
});
return documentReference.get();
}
private String buildSemanticNodeCountMessage(int numberOfPages, Map<NodeType, Long> semanticNodeCounts) {
return String.format("%d pages with %d sections, %d headlines, %d paragraphs, %d tables with %d cells, %d headers, and %d footers parsed",
numberOfPages,
semanticNodeCounts.get(NodeType.SECTION) == null ? 0 : semanticNodeCounts.get(NodeType.SECTION),
semanticNodeCounts.get(NodeType.HEADLINE) == null ? 0 : semanticNodeCounts.get(NodeType.HEADLINE),
semanticNodeCounts.get(NodeType.PARAGRAPH) == null ? 0 : semanticNodeCounts.get(NodeType.PARAGRAPH),
semanticNodeCounts.get(NodeType.TABLE) == null ? 0 : semanticNodeCounts.get(NodeType.TABLE),
semanticNodeCounts.get(NodeType.TABLE_CELL) == null ? 0 : semanticNodeCounts.get(NodeType.TABLE_CELL),
semanticNodeCounts.get(NodeType.HEADER) == null ? 0 : semanticNodeCounts.get(NodeType.HEADER),
semanticNodeCounts.get(NodeType.FOOTER) == null ? 0 : semanticNodeCounts.get(NodeType.FOOTER));
}
@SneakyThrows
@Observed(name = "LayoutParsingPipeline", contextualName = "parse-layout")
public ClassificationDocument parseLayout(LayoutParsingType layoutParsingType,
File originFile,
ImageServiceResponse imageServiceResponse,
TableServiceResponse tableServiceResponse,
String identifier) {
PDDocument originDocument = openDocument(originFile);
addNumberOfPagesToTrace(originDocument.getNumberOfPages(), Files.size(originFile.toPath()));
Map<Integer, List<TableCells>> pdfTableCells = cvTableParsingAdapter.buildCvParsedTablesPerPage(tableServiceResponse);
Map<Integer, List<ClassifiedImage>> pdfImages = imageServiceResponseAdapter.buildClassifiedImagesPerPage(imageServiceResponse);
ClassificationDocument classificationDocument = new ClassificationDocument();
List<ClassificationPage> classificationPages = new ArrayList<>();
long pageCount = originDocument.getNumberOfPages();
for (int pageNumber = 1; pageNumber <= pageCount; pageNumber++) {
if (pageNumber % 100 == 0) {
// re-open document every once in a while to save on RAM. This has no significant performance impact.
// This is due to PDFBox caching all images and some other stuff with Soft References. This dereferences them and forces the freeing of memory.
originDocument.close();
originDocument = openDocument(originFile);
}
if (pageNumber % 100 == 0 || pageNumber == pageCount || pageNumber == 1) {
log.info("Extracting text on Page {} for {}", pageNumber, identifier);
}
classificationDocument.setPages(classificationPages);
PDFLinesTextStripper stripper = new PDFLinesTextStripper();
PDPage pdPage = originDocument.getPage(pageNumber - 1);
stripper.setPageNumber(pageNumber);
stripper.setStartPage(pageNumber);
stripper.setEndPage(pageNumber);
stripper.setPdpage(pdPage);
if (layoutParsingType.equals(LayoutParsingType.DOCUMINE)) {
stripper.setSortByPosition(true);
}
stripper.getText(originDocument);
PDRectangle pdr = pdPage.getMediaBox();
int rotation = pdPage.getRotation();
boolean isLandscape = pdr.getWidth() > pdr.getHeight() && (rotation == 0 || rotation == 180) || pdr.getHeight() > pdr.getWidth() && (rotation == 90 || rotation == 270);
PDRectangle cropbox = pdPage.getCropBox();
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(pdfTableCells.get(pageNumber), stripper.getRulings());
// Docstrum
AtomicInteger num = new AtomicInteger(pageNumber);
var zones = docstrumSegmentationService.segmentPage(stripper.getTextPositionSequences());
List<AbstractPageBlock> pageBlocks = new ArrayList<>();
AtomicInteger numOnPage = new AtomicInteger(1);
// List<TextPositionSequence> textPositionSequences = new ArrayList<>();
zones.forEach(zone -> {
List<TextPositionSequence> textPositionSequences = new ArrayList<>();
zone.getLines().forEach(line -> {
line.getWords().forEach(word -> {
textPositionSequences.add(new TextPositionSequence(word.getTextPositions(), num.get()));
});
});
var cps = redactManagerBlockificationService.blockify(textPositionSequences, cleanRulings.getHorizontal(), cleanRulings.getVertical());
cps.getTextBlocks().forEach(cp -> {
pageBlocks.add(redactManagerBlockificationService.buildTextBlock(((TextPageBlock) cp).getSequences(), numOnPage.getAndIncrement()));
});
});
// ClassificationPage classificationPage = switch (layoutParsingType) {
// case REDACT_MANAGER -> redactManagerBlockificationService.blockify(textPositionSequences, cleanRulings.getHorizontal(), cleanRulings.getVertical());
// case TAAS -> taasBlockificationService.blockify(textPositionSequences, cleanRulings.getHorizontal(), cleanRulings.getVertical());
// case DOCUMINE -> docuMineBlockificationService.blockify(textPositionSequences, cleanRulings.getHorizontal(), cleanRulings.getVertical());
// };
ClassificationPage classificationPage = new ClassificationPage(pageBlocks);
classificationPage.setCleanRulings(cleanRulings);
classificationPage.setRotation(rotation);
classificationPage.setLandscape(isLandscape);
classificationPage.setPageNumber(pageNumber);
classificationPage.setPageWidth(cropbox.getWidth());
classificationPage.setPageHeight(cropbox.getHeight());
// MarkedContent needs to be converted at this point, otherwise it leads to GC Problems in Pdfbox.
classificationPage.setMarkedContentBboxPerType(convertMarkedContents(stripper.getMarkedContents()));
// If images is ocr needs to be calculated before textBlocks are moved into tables, otherwise findOcr algorithm needs to be adopted.
if (pdfImages != null && pdfImages.containsKey(pageNumber)) {
classificationPage.setImages(pdfImages.get(pageNumber));
imageServiceResponseAdapter.findOcr(classificationPage);
}
tableExtractionService.extractTables(cleanRulings, classificationPage);
buildPageStatistics(classificationPage);
increaseDocumentStatistics(classificationPage, classificationDocument);
classificationPages.add(classificationPage);
}
originDocument.close();
log.info("Calculating BodyTextFrame for {}", identifier);
bodyTextFrameService.setBodyTextFrames(classificationDocument, layoutParsingType);
log.info("Classify TextBlocks for {}", identifier);
switch (layoutParsingType) {
case TAAS -> taasClassificationService.classifyDocument(classificationDocument);
case DOCUMINE -> docuMineClassificationService.classifyDocument(classificationDocument);
case REDACT_MANAGER -> redactManagerClassificationService.classifyDocument(classificationDocument);
}
sectionsBuilderService.buildSections(classificationDocument);
sectionsBuilderService.addImagesToSections(classificationDocument);
List<ClassificationSection> sections = new ArrayList<>();
for (var page : classificationPages) {
page.getTextBlocks().forEach(block -> {
block.setPage(page.getPageNumber());
var section = sectionsBuilderService.buildTextBlock(List.of(block), "a");
sections.add(section);
});
}
classificationDocument.setSections(sections);
return DocumentGraphFactory.buildDocumentGraph(classificationDocument);
log.info("Building Sections for {}", identifier);
// sectionsBuilderService.buildSections(classificationDocument);
// sectionsBuilderService.addImagesToSections(classificationDocument);
return classificationDocument;
}
public Document parseLayoutWithTimer(LayoutParsingType layoutParsingType,
PDDocument originDocument,
ImageServiceResponse imageServiceResponse,
TableServiceResponse tableServiceResponse) {
private void addNumberOfPagesToTrace(int numberOfPages, long size) {
long start = System.currentTimeMillis();
ClassificationDocument classificationDocument = pdfParsingService.parseDocument(layoutParsingType,
originDocument,
cvTableParsingAdapter.buildCvParsedTablesPerPage(tableServiceResponse),
imageServiceResponseAdapter.buildClassifiedImagesPerPage(imageServiceResponse));
System.out.printf("parsed %d ms", System.currentTimeMillis() - start);
start = System.currentTimeMillis();
switch (layoutParsingType) {
case TAAS -> taasClassificationService.classifyDocument(classificationDocument);
case DOCUMINE -> docuMineClassificationService.classifyDocument(classificationDocument);
case REDACT_MANAGER -> redactManagerClassificationService.classifyDocument(classificationDocument);
if (observationRegistry.getCurrentObservation() != null) {
observationRegistry.getCurrentObservation().highCardinalityKeyValue("numberOfPages", String.valueOf(numberOfPages));
observationRegistry.getCurrentObservation().highCardinalityKeyValue("fileSize", String.valueOf(size));
}
System.out.printf(", classified %d ms", System.currentTimeMillis() - start);
}
start = System.currentTimeMillis();
sectionsBuilderService.buildSections(classificationDocument);
System.out.printf(", sections built %d ms", System.currentTimeMillis() - start);
start = System.currentTimeMillis();
Document document = DocumentGraphFactory.buildDocumentGraph(classificationDocument);
System.out.printf(", graph constructed %d ms", System.currentTimeMillis() - start);
@SneakyThrows
private PDDocument openDocument(File originFile) {
PDDocument document = Loader.loadPDF(originFile);
document.setAllSecurityToBeRemoved(true);
return document;
}
private Map<String, List<Rectangle2D>> convertMarkedContents(List<PDMarkedContent> pdMarkedContents) {
Map<String, List<Rectangle2D>> markedContentBboxes = new HashMap<>();
markedContentBboxes.put(MarkedContentUtils.HEADER, MarkedContentUtils.getMarkedContentBboxPerLine(pdMarkedContents, MarkedContentUtils.HEADER));
markedContentBboxes.put(MarkedContentUtils.FOOTER, MarkedContentUtils.getMarkedContentBboxPerLine(pdMarkedContents, MarkedContentUtils.FOOTER));
return markedContentBboxes;
}
private void increaseDocumentStatistics(ClassificationPage classificationPage, ClassificationDocument document) {
if (!classificationPage.isLandscape()) {
document.getFontSizeCounter().addAll(classificationPage.getFontSizeCounter().getCountPerValue());
}
document.getFontCounter().addAll(classificationPage.getFontCounter().getCountPerValue());
document.getTextHeightCounter().addAll(classificationPage.getTextHeightCounter().getCountPerValue());
document.getFontStyleCounter().addAll(classificationPage.getFontStyleCounter().getCountPerValue());
}
private void buildPageStatistics(ClassificationPage classificationPage) {
// Collect all statistics for the classificationPage, except from blocks inside tables, as tables will always be added to BodyTextFrame.
for (AbstractPageBlock textBlock : classificationPage.getTextBlocks()) {
if (textBlock instanceof TextPageBlock) {
if (((TextPageBlock) textBlock).getSequences() == null) {
continue;
}
for (TextPositionSequence word : ((TextPageBlock) textBlock).getSequences()) {
classificationPage.getTextHeightCounter().add(word.getTextHeight());
classificationPage.getFontCounter().add(word.getFont());
classificationPage.getFontSizeCounter().add(word.getFontSize());
classificationPage.getFontStyleCounter().add(word.getFontStyle());
}
}
}
}
}
@@ -1,10 +1,24 @@
package com.knecon.fforesight.service.layoutparser.processor;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.ComponentScan;
import org.springframework.context.annotation.Configuration;
import com.knecon.fforesight.service.viewerdoc.service.ViewerDocumentService;
import io.micrometer.observation.ObservationRegistry;
@Configuration
@ComponentScan
public class LayoutParsingServiceProcessorConfiguration {
@Bean
@Autowired
public ViewerDocumentService viewerDocumentService(ObservationRegistry registry) {
return new ViewerDocumentService(registry);
}
}
@@ -1,32 +1,32 @@
package com.knecon.fforesight.service.layoutparser.processor;
import java.io.File;
import java.io.FileOutputStream;
import java.io.FileInputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.nio.file.StandardOpenOption;
import java.util.Optional;
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.io.MemoryUsageSetting;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.stereotype.Service;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionGrid;
import com.iqser.red.storage.commons.service.StorageService;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPositionData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentTextData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentStructure;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPage;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.SimplifiedText;
import com.knecon.fforesight.service.layoutparser.internal.api.data.taas.ResearchDocumentData;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingRequest;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.image.ImageServiceResponse;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableServiceResponse;
import com.knecon.fforesight.tenantcommons.TenantContext;
import io.micrometer.observation.annotation.Observed;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@@ -37,38 +37,50 @@ public class LayoutParsingStorageService {
private final StorageService storageService;
private final ObjectMapper objectMapper;
@Observed(name = "LayoutParsingStorageService", contextualName = "get-origin-file")
public File getOriginFile(String storageId) throws IOException {
public PDDocument getOriginFile(String storageId) throws IOException {
File tempFile = createTempFile("document", ".pdf");
storageService.downloadTo(TenantContext.getTenantId(), storageId, tempFile);
return tempFile;
}
try (var originDocumentInputStream = storageService.getObject(TenantContext.getTenantId(), storageId).getInputStream()) {
File tempFile = createTempFile("document", ".pdf");
try (var tempFileOutputStream = new FileOutputStream(tempFile)) {
IOUtils.copy(originDocumentInputStream, tempFileOutputStream);
}
return Loader.loadPDF(tempFile, MemoryUsageSetting.setupMixed(67108864L));
@Observed(name = "LayoutParsingStorageService", contextualName = "get-viewer-doc-file")
public Optional<File> getViewerDocFile(String storageId) throws IOException {
if (!storageService.objectExists(TenantContext.getTenantId(), storageId)) {
return Optional.empty();
}
File tempFile = createTempFile("viewerDocument", ".pdf");
storageService.downloadTo(TenantContext.getTenantId(), storageId, tempFile);
return Optional.of(tempFile);
}
public ImageServiceResponse getImagesFile(String storageId) throws IOException {
try (InputStream inputStream = storageService.getObject(TenantContext.getTenantId(), storageId).getInputStream()) {
try (InputStream inputStream = getObject(storageId)) {
return objectMapper.readValue(inputStream, ImageServiceResponse.class);
ImageServiceResponse imageServiceResponse = objectMapper.readValue(inputStream, ImageServiceResponse.class);
inputStream.close();
return imageServiceResponse;
}
}
public TableServiceResponse getTablesFile(String storageId) throws IOException {
try (var tableClassificationStream = storageService.getObject(TenantContext.getTenantId(), storageId).getInputStream()) {
return objectMapper.readValue(tableClassificationStream, TableServiceResponse.class);
try (var tableClassificationStream = getObject(storageId)) {
TableServiceResponse tableServiceResponse = objectMapper.readValue(tableClassificationStream, TableServiceResponse.class);
tableClassificationStream.close();
return tableServiceResponse;
}
}
@Observed(name = "LayoutParsingStorageService", contextualName = "store-document-data")
public void storeDocumentData(LayoutParsingRequest layoutParsingRequest, DocumentData documentData) {
storageService.storeJSONObject(TenantContext.getTenantId(), layoutParsingRequest.structureFileStorageId(), documentData.getDocumentStructure());
@@ -78,38 +90,12 @@ public class LayoutParsingStorageService {
}
public void storeSectionGrid(LayoutParsingRequest layoutParsingRequest, SectionGrid sectionGrid) {
storageService.storeJSONObject(TenantContext.getTenantId(), layoutParsingRequest.sectionGridStorageId(), sectionGrid);
}
public void storeResearchDocumentData(LayoutParsingRequest layoutParsingRequest, ResearchDocumentData researchDocumentData) {
storageService.storeJSONObject(TenantContext.getTenantId(), layoutParsingRequest.researchDocumentStorageId(), researchDocumentData);
}
public DocumentData readDocumentData(LayoutParsingRequest layoutParsingRequest) throws IOException {
DocumentPage[] documentPageData = storageService.readJSONObject(TenantContext.getTenantId(), layoutParsingRequest.pageFileStorageId(), DocumentPage[].class);
DocumentTextData[] documentTextDataBlockData = storageService.readJSONObject(TenantContext.getTenantId(),
layoutParsingRequest.textBlockFileStorageId(),
DocumentTextData[].class);
DocumentPositionData[] atomicPositionBlockData = storageService.readJSONObject(TenantContext.getTenantId(),
layoutParsingRequest.positionBlockFileStorageId(),
DocumentPositionData[].class);
DocumentStructure tableOfContentsData = storageService.readJSONObject(TenantContext.getTenantId(), layoutParsingRequest.structureFileStorageId(), DocumentStructure.class);
return DocumentData.builder()
.documentStructure(tableOfContentsData)
.documentPositions(atomicPositionBlockData)
.documentTextData(documentTextDataBlockData)
.documentPages(documentPageData)
.build();
}
private File createTempFile(String filenamePrefix, String filenameSuffix) throws IOException {
File tempFile = Files.createTempFile(filenamePrefix, filenameSuffix).toFile();
@@ -134,4 +120,31 @@ public class LayoutParsingStorageService {
}
}
@Observed(name = "LayoutParsingStorageService", contextualName = "store-simplified-text")
public void storeSimplifiedText(LayoutParsingRequest layoutParsingRequest, SimplifiedText simplifiedText) {
storageService.storeJSONObject(TenantContext.getTenantId(), layoutParsingRequest.simplifiedTextStorageId(), simplifiedText);
}
@SneakyThrows
private InputStream getObject(String storageId) {
File tempFile = File.createTempFile("temp", ".data");
storageService.downloadTo(TenantContext.getTenantId(), storageId, tempFile);
Path path = Paths.get(tempFile.getPath());
return Files.newInputStream(path, StandardOpenOption.DELETE_ON_CLOSE);
}
@SneakyThrows
@Observed(name = "LayoutParsingStorageService", contextualName = "store-viewer-document")
public void storeViewerDocument(LayoutParsingRequest layoutParsingRequest, File out) {
try (var in = new FileInputStream(out)) {
storageService.storeObject(TenantContext.getTenantId(), layoutParsingRequest.viewerDocumentStorageId(), in);
}
}
}
@@ -72,9 +72,30 @@ public abstract class AbstractPageBlock {
}
public boolean intersectsY(AbstractPageBlock atc) {
public boolean intersectsY(AbstractPageBlock apb) {
return this.minY <= atc.getMaxY() && this.maxY >= atc.getMinY();
return this.minY <= apb.getMaxY() && this.maxY >= apb.getMinY();
}
public boolean almostIntersects(AbstractPageBlock apb, float yThreshold, float xThreshold) {
return this.almostIntersectsX(apb, xThreshold) && this.almostIntersectsY(apb, yThreshold);
}
private boolean almostIntersectsY(AbstractPageBlock apb, float threshold) {
return this.minY - threshold <= apb.getMaxY() && this.maxY + threshold >= apb.getMinY();
}
private boolean almostIntersectsX(AbstractPageBlock apb, float threshold) {
return this.minX - threshold <= apb.getMaxX() && this.maxX + threshold >= apb.getMinX();
}
public abstract boolean isEmpty();
}
@@ -3,7 +3,6 @@ package com.knecon.fforesight.service.layoutparser.processor.model;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionGrid;
import com.knecon.fforesight.service.layoutparser.processor.model.text.StringFrequencyCounter;
import com.knecon.fforesight.service.layoutparser.processor.model.text.UnclassifiedText;
@@ -25,7 +24,6 @@ public class ClassificationDocument {
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private boolean headlines;
private SectionGrid sectionGrid = new SectionGrid();
private long rulesVersion;
}
@@ -1,13 +1,18 @@
package com.knecon.fforesight.service.layoutparser.processor.model;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.knecon.fforesight.service.layoutparser.processor.model.image.ClassifiedImage;
import com.knecon.fforesight.service.layoutparser.processor.model.table.CleanRulings;
import com.knecon.fforesight.service.layoutparser.processor.model.text.StringFrequencyCounter;
import lombok.Data;
import lombok.Getter;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
@@ -16,6 +21,7 @@ import lombok.RequiredArgsConstructor;
public class ClassificationPage {
@NonNull
@Getter
private List<AbstractPageBlock> textBlocks;
private List<ClassifiedImage> images = new ArrayList<>();
@@ -35,4 +41,8 @@ public class ClassificationPage {
private float pageWidth;
private float pageHeight;
CleanRulings cleanRulings;
private Map<String, List<Rectangle2D>> markedContentBboxPerType = new HashMap<>();
}
@@ -2,6 +2,7 @@ package com.knecon.fforesight.service.layoutparser.processor.model;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Collectors;
import com.knecon.fforesight.service.layoutparser.processor.model.image.ClassifiedImage;
import com.knecon.fforesight.service.layoutparser.processor.model.table.TablePageBlock;
@@ -29,4 +30,10 @@ public class ClassificationSection {
return tables;
}
public List<AbstractPageBlock> getNonEmptyPageBlocks() {
return pageBlocks.stream().filter(pageBlock -> !pageBlock.isEmpty()).collect(Collectors.toList());
}
}
@@ -3,6 +3,7 @@ package com.knecon.fforesight.service.layoutparser.processor.model;
import java.awt.geom.Rectangle2D;
import java.util.List;
import com.knecon.fforesight.service.layoutparser.processor.model.table.Ruling;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import lombok.AllArgsConstructor;
@@ -17,5 +18,5 @@ public class PageContents {
List<TextPositionSequence> sortedTextPositionSequences;
Rectangle2D cropBox;
Rectangle2D mediaBox;
List<Ruling> rulings;
}
@@ -12,6 +12,7 @@ import lombok.Setter;
@Setter
@EqualsAndHashCode
@SuppressWarnings("PMD.AvoidFieldNameMatchingMethodName")
public class Boundary implements Comparable<Boundary> {
private int start;
@@ -10,7 +10,6 @@ import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import com.amazonaws.services.kms.model.NotFoundException;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.DocumentTree;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.entity.RedactionEntity;
@@ -84,7 +83,7 @@ public class Document implements GenericSemanticNode {
@Override
public Headline getHeadline() {
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node).findFirst().orElseThrow(() -> new NotFoundException("No Headlines found in this document!"));
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node).findFirst().orElse(Headline.builder().build());
}
@@ -100,6 +99,12 @@ public class Document implements GenericSemanticNode {
}
public Map<NodeType, Long> buildSemanticNodeCounts() {
return streamAllSubNodes().collect(Collectors.groupingBy(SemanticNode::getType, Collectors.counting()));
}
@Override
public String toString() {
@@ -1,7 +1,9 @@
package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
@@ -34,6 +36,9 @@ public class Footer implements GenericSemanticNode {
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
@Override
public NodeType getType() {
@@ -62,4 +67,14 @@ public class Footer implements GenericSemanticNode {
return treeId + ": " + NodeType.FOOTER + ": " + leafTextBlock.buildSummary();
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = GenericSemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -1,7 +1,9 @@
package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
@@ -34,6 +36,9 @@ public class Header implements GenericSemanticNode {
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
@Override
public boolean isLeaf() {
@@ -62,4 +67,14 @@ public class Header implements GenericSemanticNode {
return treeId + ": " + NodeType.HEADER + ": " + leafTextBlock.buildSummary();
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = GenericSemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -1,7 +1,9 @@
package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
@@ -34,6 +36,9 @@ public class Headline implements GenericSemanticNode {
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
@Override
public NodeType getType() {
@@ -69,4 +74,14 @@ public class Headline implements GenericSemanticNode {
return this;
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = GenericSemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -1,7 +1,9 @@
package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
@@ -32,6 +34,9 @@ public class Paragraph implements GenericSemanticNode {
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
@Override
public NodeType getType() {
@@ -60,4 +65,14 @@ public class Paragraph implements GenericSemanticNode {
return treeId + ": " + NodeType.PARAGRAPH + ": " + leafTextBlock.buildSummary();
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = GenericSemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -1,7 +1,9 @@
package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
@@ -35,6 +37,9 @@ public class Section implements GenericSemanticNode {
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
@Override
public NodeType getType() {
@@ -74,4 +79,14 @@ public class Section implements GenericSemanticNode {
.orElseGet(() -> getParent().getHeadline());
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = GenericSemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -398,12 +398,11 @@ public interface SemanticNode {
*/
default Map<Page, Rectangle2D> getBBox() {
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
if (isLeaf()) {
return getBBoxFromLeafTextBlock(bBoxPerPage);
return getBBoxFromLeafTextBlock();
}
return getBBoxFromChildren(bBoxPerPage);
return getBBoxFromChildren();
}
@@ -426,25 +425,31 @@ public interface SemanticNode {
/**
* TODO: this produces unwanted results for sections spanning multiple columns.
*
* @param bBoxPerPage initial empty BoundingBox
* Computes the Union of the bounding boxes of all children recursively.
* @return The union of the BoundingBoxes of all children
*/
private Map<Page, Rectangle2D> getBBoxFromChildren(Map<Page, Rectangle2D> bBoxPerPage) {
private Map<Page, Rectangle2D> getBBoxFromChildren() {
return streamChildren().map(SemanticNode::getBBox).reduce((map1, map2) -> {
map1.forEach((page, rectangle) -> map2.merge(page, rectangle, (rect1, rect2) -> rect1.createUnion(rect2).getBounds2D()));
return map2;
}).orElse(bBoxPerPage);
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
List<Map<Page, Rectangle2D>> childrenBBoxes = streamChildren().map(SemanticNode::getBBox).toList();
Set<Page> pages = childrenBBoxes.stream().flatMap(map -> map.keySet().stream()).collect(Collectors.toSet());
for (Page page : pages) {
Rectangle2D bBoxOnPage = childrenBBoxes.stream()
.filter(childBboxPerPage -> childBboxPerPage.containsKey(page))
.map(childBboxPerPage -> childBboxPerPage.get(page))
.collect(RectangleTransformations.collectBBox());
bBoxPerPage.put(page, bBoxOnPage);
}
return bBoxPerPage;
}
/**
* @param bBoxPerPage initial empty BoundingBox
* @return The union of all BoundingBoxes of the TextBlock of this node
*/
private Map<Page, Rectangle2D> getBBoxFromLeafTextBlock(Map<Page, Rectangle2D> bBoxPerPage) {
private Map<Page, Rectangle2D> getBBoxFromLeafTextBlock() {
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
Map<Page, List<AtomicTextBlock>> atomicTextBlockPerPage = getTextBlock().getAtomicTextBlocks().stream().collect(Collectors.groupingBy(AtomicTextBlock::getPage));
atomicTextBlockPerPage.forEach((page, atbs) -> bBoxPerPage.put(page, RectangleTransformations.bBoxUnionAtomicTextBlock(atbs)));
return bBoxPerPage;
@@ -2,10 +2,12 @@ package com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.Collection;
import java.util.HashSet;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.Set;
import java.util.stream.IntStream;
import java.util.stream.Stream;
@@ -34,13 +36,14 @@ public class Table implements SemanticNode {
int numberOfRows;
int numberOfCols;
TextBlock textBlock;
@Builder.Default
@EqualsAndHashCode.Exclude
Set<RedactionEntity> entities = new HashSet<>();
@EqualsAndHashCode.Exclude
Map<Page, Rectangle2D> bBoxCache;
/**
* Streams all entities in this table, that appear in a row, which contains any of the provided strings.
@@ -208,7 +211,6 @@ public class Table implements SemanticNode {
return IntStream.range(0, numberOfCols).boxed().map(col -> getCell(row, col));
}
/**
* Streams all TableCells row-wise and filters them with header == true.
*
@@ -313,5 +315,12 @@ public class Table implements SemanticNode {
return treeId.toString() + ": " + NodeType.TABLE + ": #cols: " + numberOfCols + ", #rows: " + numberOfRows + ", " + this.getTextBlock().buildSummary();
}
@Override
public Map<Page, Rectangle2D> getBBox() {
if (bBoxCache == null) {
bBoxCache = SemanticNode.super.getBBox();
}
return bBoxCache;
}
}
@@ -76,16 +76,22 @@ public class TableCell implements GenericSemanticNode {
}
if (textBlock == null) {
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
textBlock = buildTextBlock();
}
return textBlock;
}
private TextBlock buildTextBlock() {
return streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
}
@Override
public String toString() {
return treeId + ": " + NodeType.TABLE_CELL + ": " + this.getTextBlock().buildSummary();
return treeId + ": " + NodeType.TABLE_CELL + ": " + this.buildTextBlock().buildSummary();
}
}
@@ -28,6 +28,7 @@ public class Ruling extends Line2D.Float {
super(p1, p2);
}
public Ruling straightenVertical() {
double y1 = Math.min(getY1(), getY2());
@@ -36,6 +37,7 @@ public class Ruling extends Line2D.Float {
return new Ruling(new Point2D.Double(x, y1), new Point2D.Double(x, y2));
}
public Ruling straightenHorizontal() {
double x1 = Math.min(getX1(), getX2());
@@ -444,6 +446,16 @@ public class Ruling extends Line2D.Float {
}
public boolean almostMatches(Ruling ruling) {
final float TOLERANCE = 1;
return Math.abs(ruling.getX1() - x1) < TOLERANCE &&//
Math.abs(ruling.getY1() - y1) < TOLERANCE &&//
Math.abs(ruling.getX2() - x2) < TOLERANCE &&//
Math.abs(ruling.getY2() - y2) < TOLERANCE;
}
private enum SOType {
VERTICAL,
HRIGHT,
@@ -1,12 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.model.table;
import java.awt.geom.Point2D;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.Collections;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.TreeMap;
import java.util.stream.Collectors;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.PageBlockType;
@@ -42,6 +44,12 @@ public class TablePageBlock extends AbstractPageBlock {
}
@Override
public boolean isEmpty() {
return getColCount() == 0 || getRowCount() == 0;
}
public List<List<Cell>> getRows() {
if (rows == null) {
@@ -246,7 +254,8 @@ public class TablePageBlock extends AbstractPageBlock {
if (prevY != null && prevX != null) {
var cell = new Cell(new Point2D.Float(prevX, prevY), new Point2D.Float(x, y));
var intersectionCell = cells.stream().filter(c -> cell.intersects(c) && cell.overlapRatio(c) > 0.1f).findFirst();
var intersectionCell = cells.stream().filter(c -> intersects(cell, c)).findFirst();
intersectionCell.ifPresent(value -> cell.getTextBlocks().addAll(value.getTextBlocks()));
if (cell.hasMinimumSize()) {
row.add(cell);
@@ -267,6 +276,21 @@ public class TablePageBlock extends AbstractPageBlock {
}
public boolean intersects(Cell cell1, Cell cell2) {
if (cell1.getHeight() <= 0 || cell2.getHeight() <= 0) {
return false;
}
double x0 = cell1.getX() + 2;
double y0 = cell1.getY() + 2;
return (cell2.x + cell2.width > x0 &&
cell2.y + cell2.height > y0 &&
cell2.x < x0 + cell1.getWidth() -2 &&
cell2.y < y0 + cell1.getHeight() -2);
}
@Override
public String getText() {
@@ -304,6 +328,8 @@ public class TablePageBlock extends AbstractPageBlock {
}
public String getTextAsHtml() {
StringBuilder sb = new StringBuilder();
@@ -17,7 +17,6 @@ import lombok.SneakyThrows;
@AllArgsConstructor
public class RedTextPosition {
private String textMatrix;
private float[] position;
@JsonIgnore
@@ -46,6 +45,9 @@ public class RedTextPosition {
@JsonIgnore
private String fontName;
@JsonIgnore
private RedTextPosition parent;
@SneakyThrows
public static RedTextPosition fromTextPosition(TextPosition textPosition) {
@@ -56,8 +58,6 @@ public class RedTextPosition {
pos.setFontSizeInPt(textPosition.getFontSizeInPt());
pos.setTextMatrix(textPosition.getTextMatrix().toString());
var position = new float[4];
position[0] = textPosition.getXDirAdj();
@@ -1,17 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.model.text;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedSectionText {
private int sectionNumber;
private String text;
}
@@ -1,20 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.model.text;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedText {
private int numberOfPages;
private List<SimplifiedSectionText> sectionTexts = new ArrayList<>();
}
@@ -17,6 +17,7 @@ import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.Getter;
import lombok.NoArgsConstructor;
@EqualsAndHashCode(callSuper = true)
@@ -27,6 +28,7 @@ import lombok.NoArgsConstructor;
public class TextPageBlock extends AbstractPageBlock {
@Builder.Default
@Getter
private List<TextPositionSequence> sequences = new ArrayList<>();
@JsonIgnore
@@ -74,6 +76,7 @@ public class TextPageBlock extends AbstractPageBlock {
return sequences.get(0).getPageWidth();
}
public static TextPageBlock merge(List<TextPageBlock> textBlocksToMerge) {
List<TextPositionSequence> sequences = textBlocksToMerge.stream().map(TextPageBlock::getSequences).flatMap(java.util.Collection::stream).toList();
@@ -81,6 +84,7 @@ public class TextPageBlock extends AbstractPageBlock {
return fromTextPositionSequences(sequences);
}
public static TextPageBlock fromTextPositionSequences(List<TextPositionSequence> wordBlockList) {
TextPageBlock textBlock = null;
@@ -132,7 +136,6 @@ public class TextPageBlock extends AbstractPageBlock {
}
/**
* Returns the minX value in pdf coordinate system.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
@@ -364,4 +367,11 @@ public class TextPageBlock extends AbstractPageBlock {
}
@Override
public boolean isEmpty() {
return sequences.isEmpty();
}
}
@@ -9,8 +9,6 @@ import java.util.stream.Collectors;
import org.apache.pdfbox.text.TextPosition;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Point;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
@@ -57,6 +55,18 @@ public class TextPositionSequence implements CharSequence {
}
public TextPositionSequence(List<RedTextPosition> textPositions, int page) {
this.textPositions = textPositions;
this.page = page;
this.dir = TextDirection.fromDegrees(textPositions.get(0).getDir());
this.rotation = textPositions.get(0).getRotation();
this.pageHeight = textPositions.get(0).getPageHeight();
this.pageWidth = textPositions.get(0).getPageWidth();
this.isParagraphStart = false;
}
@Override
public int length() {
@@ -142,8 +152,7 @@ public class TextPositionSequence implements CharSequence {
*
* @return the text direction adjusted minX value
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getMinXDirAdj() {
return textPositions.get(0).getXDirAdj();
@@ -157,8 +166,7 @@ public class TextPositionSequence implements CharSequence {
*
* @return the text direction adjusted maxX value
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getMaxXDirAdj() {
return textPositions.get(textPositions.size() - 1).getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + HEIGHT_PADDING;
@@ -172,8 +180,7 @@ public class TextPositionSequence implements CharSequence {
*
* @return the text direction adjusted minY value. The upper border of the bounding box of the word.
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getMinYDirAdj() {
return textPositions.get(0).getYDirAdj() - getTextHeight();
@@ -187,8 +194,7 @@ public class TextPositionSequence implements CharSequence {
*
* @return the text direction adjusted maxY value. The lower border of the bounding box of the word.
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getMaxYDirAdj() {
return textPositions.get(0).getYDirAdj();
@@ -196,42 +202,38 @@ public class TextPositionSequence implements CharSequence {
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getTextHeight() {
return textPositions.get(0).getHeightDir() + HEIGHT_PADDING;
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getHeight() {
return getMaxYDirAdj() - getMinYDirAdj();
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getWidth() {
return getMaxXDirAdj() - getMinXDirAdj();
}
@JsonIgnore
@JsonAttribute(ignore = true)
public String getFont() {
if (textPositions.get(0).getFontName() == null) {
return "none";
}
return textPositions.get(0).getFontName().toLowerCase(Locale.ROOT).replaceAll(",bold", "").replaceAll(",italic", "");
}
@JsonIgnore
@JsonAttribute(ignore = true)
public String getFontStyle() {
if (textPositions.get(0).getFontName() == null) {
return "standard";
}
String lowercaseFontName = textPositions.get(0).getFontName().toLowerCase(Locale.ROOT);
if (lowercaseFontName.contains("bold") && lowercaseFontName.contains("italic")) {
@@ -243,20 +245,15 @@ public class TextPositionSequence implements CharSequence {
} else {
return "standard";
}
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getFontSize() {
return textPositions.get(0).getFontSizeInPt();
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getSpaceWidth() {
return textPositions.get(0).getWidthOfSpace();
@@ -272,8 +269,7 @@ public class TextPositionSequence implements CharSequence {
*
* @return bounding box of the word in Pdf Coordinate System
*/
@JsonIgnore
@JsonAttribute(ignore = true)
@SneakyThrows
public Rectangle getRectangle() {
@@ -8,6 +8,7 @@ import java.util.Map;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.PageInfo;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableCells;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableServiceResponse;
@@ -24,24 +25,26 @@ public class CvTableParsingAdapter {
Map<Integer, List<TableCells>> tableCells = new HashMap<>();
tableServiceResponse.getData()
.forEach(tableData -> tableCells.computeIfAbsent(tableData.getPageInfo().getNumber(), tableCell -> new ArrayList<>())
.addAll(convertTableCells(tableData.getTableCells())));
.addAll(convertTableCells(tableData.getTableCells(), tableData.getPageInfo())));
return tableCells;
}
private Collection<TableCells> convertTableCells(List<TableCells> tableCells) {
private Collection<TableCells> convertTableCells(List<TableCells> tableCells, PageInfo pageInfo) {
List<TableCells> cvParsedTableCells = new ArrayList<>();
tableCells.forEach(t -> cvParsedTableCells.add(TableCells.builder()
.y0(t.getY0())
.x1(t.getX1())
.y1(t.getY1())
.x0(t.getX0())
.width(t.getWidth())
.height(t.getHeight())
.build()));
tableCells.stream()
.filter(cell -> cell.getWidth() < pageInfo.getWidth() * 0.98 && cell.getHeight() < pageInfo.getHeight() * 0.98)
.forEach(t -> cvParsedTableCells.add(TableCells.builder()
.y0(t.getY0())
.x1(t.getX1())
.y1(t.getY1())
.x0(t.getX0())
.width(t.getWidth())
.height(t.getHeight())
.build()));
return cvParsedTableCells;
}
@@ -3,9 +3,15 @@ package com.knecon.fforesight.service.layoutparser.processor.python_api.model.im
import java.util.HashMap;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Classification {
private Map<String, Float> probabilities = new HashMap<>();
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class FilterGeometry {
private ImageSize imageSize;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Filters {
private FilterGeometry geometry;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Geometry {
private float width;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImageFormat {
private float quotient;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImageMetadata {
private Classification classification;
@@ -6,9 +6,15 @@ import java.util.List;
import com.fasterxml.jackson.annotation.JsonAlias;
import com.fasterxml.jackson.annotation.JsonProperty;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImageServiceResponse {
private String dossierId;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImageSize {
private float quotient;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Position {
private float x1;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.image;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Probability {
private boolean unconfident;
@@ -1,8 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.table;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class PageInfo {
private int number;
@@ -3,12 +3,12 @@ package com.knecon.fforesight.service.layoutparser.processor.python_api.model.ta
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.RequiredArgsConstructor;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@RequiredArgsConstructor
public class PdfTableCell {
private float x0;
@@ -1,10 +1,14 @@
package com.knecon.fforesight.service.layoutparser.processor.python_api.model.table;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class TableCells {
private float x0;
@@ -3,9 +3,15 @@ package com.knecon.fforesight.service.layoutparser.processor.python_api.model.ta
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class TableData {
private PageInfo pageInfo;
@@ -3,9 +3,15 @@ package com.knecon.fforesight.service.layoutparser.processor.python_api.model.ta
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class TableServiceResponse {
private String dossierId;
@@ -1,40 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.queue;
import static com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingQueueNames.LAYOUT_PARSING_FINISHED_EVENT_QUEUE;
import static com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingQueueNames.LAYOUT_PARSING_DLQ;
import static com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingQueueNames.LAYOUT_PARSING_REQUEST_QUEUE;
import org.springframework.amqp.core.Queue;
import org.springframework.amqp.core.QueueBuilder;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import lombok.RequiredArgsConstructor;
@Configuration
@RequiredArgsConstructor
public class MessagingConfiguration {
@Bean
public Queue layoutparsingRequestQueue() {
return QueueBuilder.durable(LAYOUT_PARSING_REQUEST_QUEUE)//
.withArgument("x-dead-letter-exchange", "").withArgument("x-dead-letter-routing-key", LAYOUT_PARSING_DLQ).build();
}
@Bean
public Queue layoutparsingResponseQueue() {
return QueueBuilder.durable(LAYOUT_PARSING_FINISHED_EVENT_QUEUE)//
.withArgument("x-dead-letter-exchange", "").withArgument("x-dead-letter-routing-key", LAYOUT_PARSING_DLQ).build();
}
@Bean
public Queue layoutparsingDLQ() {
return QueueBuilder.durable(LAYOUT_PARSING_DLQ).build();
}
}
@@ -1,5 +1,6 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.util.Comparator;
import java.util.List;
import org.springframework.stereotype.Service;
@@ -8,17 +9,74 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlo
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.FloatFrequencyCounter;
import com.knecon.fforesight.service.layoutparser.processor.model.table.Cell;
import com.knecon.fforesight.service.layoutparser.processor.model.table.Ruling;
import com.knecon.fforesight.service.layoutparser.processor.model.table.TablePageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.utils.MarkedContentUtils;
import com.knecon.fforesight.service.layoutparser.processor.utils.PositionUtils;
@Service
public class BodyTextFrameService {
private static final float RULING_HEIGHT_THRESHOLD = 0.15f; // multiplied with page height. Header/Footer Rulings must be within that border of the page.
private static final float RULING_WIDTH_THRESHOLD = 0.75f; // multiplied with page width. Header/Footer Rulings must be at least that wide.
public void setBodyTextFrames(ClassificationDocument classificationDocument, LayoutParsingType layoutParsingType) {
Rectangle bodyTextFrame = calculateBodyTextFrame(classificationDocument.getPages(), classificationDocument.getFontSizeCounter(), false, layoutParsingType);
Rectangle landscapeBodyTextFrame = calculateBodyTextFrame(classificationDocument.getPages(), classificationDocument.getFontSizeCounter(), true, layoutParsingType);
for (ClassificationPage page : classificationDocument.getPages()) {
// var updatedBodyTextFrame = getBodyTextFrameFromRulings(page, bodyTextFrame, landscapeBodyTextFrame);
setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
}
}
private Rectangle getBodyTextFrameFromRulings(ClassificationPage page, Rectangle bodyTextFrame, Rectangle landscapeBodyTextFrame) {
List<Ruling> potentialFooterRulings = getPotentialFooterRulings(page);
List<Ruling> potentialHeaderRulings = getPotentialHeaderRulings(page);
var x = bodyTextFrame.getTopLeft().getX();
var y = bodyTextFrame.getTopLeft().getY();
var w = bodyTextFrame.getWidth();
var h = bodyTextFrame.getHeight();
if (!potentialFooterRulings.isEmpty()) {
h = y + h - potentialFooterRulings.get(0).getTop();
y = potentialFooterRulings.get(0).getTop();
}
if (!potentialHeaderRulings.isEmpty()) {
h = potentialHeaderRulings.get(0).getBottom() - bodyTextFrame.getTopLeft().getY();
}
return new Rectangle(new Point(x, y), w, h, page.getPageNumber());
}
private List<Ruling> getPotentialFooterRulings(ClassificationPage page) {
return page.getCleanRulings()
.getHorizontal()
.stream()
.filter(ruling -> ruling.getY1() < page.getPageHeight() * RULING_HEIGHT_THRESHOLD)
.filter(ruling -> ruling.getWidth() > RULING_WIDTH_THRESHOLD * page.getPageWidth())
.sorted(Comparator.comparingDouble(Ruling::getTop))
.toList();
}
private List<Ruling> getPotentialHeaderRulings(ClassificationPage page) {
return page.getCleanRulings()
.getHorizontal()
.stream()
.filter(ruling -> ruling.getY1() > page.getPageHeight() * (1 - RULING_HEIGHT_THRESHOLD))
.filter(ruling -> ruling.getWidth() > RULING_WIDTH_THRESHOLD * page.getPageWidth())
.sorted(Comparator.comparingDouble(Ruling::getBottom).reversed())
.toList();
}
/**
@@ -34,7 +92,7 @@ public class BodyTextFrameService {
* @param bodyTextFrame frame that contains the main text on portrait pages
* @param landscapeBodyTextFrame frame that contains the main text on landscape pages
*/
public void setBodyTextFrameAdjustedToPage(ClassificationPage page, Rectangle bodyTextFrame, Rectangle landscapeBodyTextFrame) {
private void setBodyTextFrameAdjustedToPage(ClassificationPage page, Rectangle bodyTextFrame, Rectangle landscapeBodyTextFrame) {
Rectangle textFrame = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
@@ -69,7 +127,10 @@ public class BodyTextFrameService {
* @param landscape Calculate for landscape or portrait
* @return Rectangle of the text frame
*/
public Rectangle calculateBodyTextFrame(List<ClassificationPage> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape, LayoutParsingType layoutParsingType) {
protected Rectangle calculateBodyTextFrame(List<ClassificationPage> pages,
FloatFrequencyCounter documentFontSizeCounter,
boolean landscape,
LayoutParsingType layoutParsingType) {
float approximateHeaderLineCount;
if (layoutParsingType.equals(LayoutParsingType.TAAS)) {
@@ -94,9 +155,14 @@ public class BodyTextFrameService {
continue;
}
if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.HEADER)
|| MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.FOOTER)) {
continue;
}
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
if (layoutParsingType.equals(LayoutParsingType.DOCUMINE) && approxLineCount < approximateHeaderLineCount && textBlock.getMaxY() >= page.getPageHeight() - (page.getPageHeight() / 10)
|| !layoutParsingType.equals(LayoutParsingType.DOCUMINE) && approxLineCount < approximateHeaderLineCount){
if (layoutParsingType.equals(LayoutParsingType.DOCUMINE) && approxLineCount < approximateHeaderLineCount && textBlock.getMaxY() >= page.getPageHeight() - (page.getPageHeight() / 10) || !layoutParsingType.equals(
LayoutParsingType.DOCUMINE) && approxLineCount < approximateHeaderLineCount) {
continue;
}
@@ -170,4 +236,4 @@ public class BodyTextFrameService {
}
}
}
@@ -0,0 +1,73 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.io.IOException;
import java.util.Collection;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.springframework.core.io.ClassPathResource;
import com.knecon.fforesight.service.layoutparser.processor.model.PageContents;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.services.parsing.PDFLinesTextStripper;
import com.knecon.fforesight.service.layoutparser.processor.utils.RectangleTransformations;
import lombok.experimental.UtilityClass;
@UtilityClass
public class PageContentExtractor {
public List<PageContents> getSortedPageContents(String filename) throws IOException {
List<PageContents> textPositionSequencesPerPage = new LinkedList<>();
ClassPathResource pdfResource = new ClassPathResource(filename);
try (PDDocument pdDocument = Loader.loadPDF(pdfResource.getFile())) {
for (int pageNumber = 1; pageNumber < pdDocument.getNumberOfPages() + 1; pageNumber++) {
PDFLinesTextStripper stripper = new PDFLinesTextStripper();
PDPage pdPage = pdDocument.getPage(pageNumber - 1);
stripper.setPageNumber(pageNumber);
stripper.setSortByPosition(true);
stripper.setStartPage(pageNumber);
stripper.setEndPage(pageNumber);
stripper.setPdpage(pdPage);
stripper.getText(pdDocument);
Map<Float, List<TextPositionSequence>> sortedTextPositionSequencesPerDir = stripper.getTextPositionSequences()
.stream()
.collect(Collectors.groupingBy(textPositionSequence -> textPositionSequence.getDir().getDegrees()));
var sortedTextPositionSequences = sortByDirAccordingToPageRotation(sortedTextPositionSequencesPerDir, pdPage.getRotation());
textPositionSequencesPerPage.add(new PageContents(sortedTextPositionSequences,
RectangleTransformations.toRectangle2D(pdPage.getCropBox()),
RectangleTransformations.toRectangle2D(pdPage.getMediaBox()),
stripper.getRulings()));
}
}
return textPositionSequencesPerPage;
}
public List<TextPositionSequence> sortByDirAccordingToPageRotation(Map<Float, List<TextPositionSequence>> sortedTextPositionSequencesPerDir, int rotation) {
LinkedList<Float> sortedKeys = new LinkedList<>(sortedTextPositionSequencesPerDir.keySet().stream().sorted().toList());
for (int i = 0; i < sortedKeys.size(); i++) {
if (sortedKeys.get(i) < rotation) {
Float keyToSwap = sortedKeys.remove(i);
sortedKeys.addLast(keyToSwap);
}
}
return sortedKeys.stream().map(sortedTextPositionSequencesPerDir::get).flatMap(Collection::stream).toList();
}
}
@@ -1,151 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.python_api.adapter.ImageServiceResponseAdapter;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableCells;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.image.ClassifiedImage;
import com.knecon.fforesight.service.layoutparser.processor.model.table.CleanRulings;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.services.parsing.PDFLinesTextStripper;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.DocuMineBlockificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.RedactManagerBlockificationService;
import com.knecon.fforesight.service.layoutparser.processor.services.blockification.TaasBlockificationService;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class PdfParsingService {
private final RulingCleaningService rulingCleaningService;
private final TableExtractionService tableExtractionService;
private final ImageServiceResponseAdapter imageServiceResponseAdapter;
private final TaasBlockificationService taasBlockificationService;
private final DocuMineBlockificationService docuMineBlockificationService;
private final RedactManagerBlockificationService redactManagerBlockificationService;
public ClassificationDocument parseDocument(LayoutParsingType layoutParsingType,
PDDocument originDocument,
Map<Integer, List<TableCells>> pdfTableCells,
Map<Integer, List<ClassifiedImage>> pdfImages) {
ClassificationDocument document = new ClassificationDocument();
List<ClassificationPage> classificationPages = new ArrayList<>();
originDocument.setAllSecurityToBeRemoved(true);
long pageCount = originDocument.getNumberOfPages();
for (int pageNumber = 1; pageNumber <= pageCount; pageNumber++) {
parsePage(layoutParsingType, pdfImages, originDocument, pdfTableCells, document, classificationPages, pageNumber);
}
document.setPages(classificationPages);
return document;
}
@SneakyThrows
private void parsePage(LayoutParsingType layoutParsingType,
Map<Integer, List<ClassifiedImage>> pdfImages,
PDDocument pdDocument,
Map<Integer, List<TableCells>> pdfTableCells,
ClassificationDocument document,
List<ClassificationPage> classificationPages,
int pageNumber) {
PDFLinesTextStripper stripper = new PDFLinesTextStripper();
PDPage pdPage = pdDocument.getPage(pageNumber - 1);
stripper.setPageNumber(pageNumber);
stripper.setStartPage(pageNumber);
stripper.setEndPage(pageNumber);
stripper.setPdpage(pdPage);
stripper.getText(pdDocument);
PDRectangle pdr = pdPage.getMediaBox();
int rotation = pdPage.getRotation();
boolean isLandscape = pdr.getWidth() > pdr.getHeight() && (rotation == 0 || rotation == 180) || pdr.getHeight() > pdr.getWidth() && (rotation == 90 || rotation == 270);
PDRectangle cropbox = pdPage.getCropBox();
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(pdfTableCells.get(pageNumber),
stripper.getRulings(),
stripper.getMinCharWidth(),
stripper.getMaxCharHeight());
ClassificationPage classificationPage = switch (layoutParsingType) {
case REDACT_MANAGER -> redactManagerBlockificationService.blockify(stripper.getTextPositionSequences(), cleanRulings.getHorizontal(), cleanRulings.getVertical());
case TAAS -> taasBlockificationService.blockify(stripper.getTextPositionSequences(), cleanRulings.getHorizontal(), cleanRulings.getVertical());
case DOCUMINE -> docuMineBlockificationService.blockify(stripper.getTextPositionSequences(), cleanRulings.getHorizontal(), cleanRulings.getVertical());
};
classificationPage.setRotation(rotation);
classificationPage.setLandscape(isLandscape);
classificationPage.setPageNumber(pageNumber);
classificationPage.setPageWidth(cropbox.getWidth());
classificationPage.setPageHeight(cropbox.getHeight());
// If images is ocr needs to be calculated before textBlocks are moved into tables, otherwise findOcr algorithm needs to be adopted.
if (pdfImages != null && pdfImages.containsKey(pageNumber)) {
classificationPage.setImages(pdfImages.get(pageNumber));
imageServiceResponseAdapter.findOcr(classificationPage);
}
tableExtractionService.extractTables(cleanRulings, classificationPage, layoutParsingType);
buildPageStatistics(classificationPage);
increaseDocumentStatistics(classificationPage, document);
classificationPages.add(classificationPage);
}
private void increaseDocumentStatistics(ClassificationPage classificationPage, ClassificationDocument document) {
if (!classificationPage.isLandscape()) {
document.getFontSizeCounter().addAll(classificationPage.getFontSizeCounter().getCountPerValue());
}
document.getFontCounter().addAll(classificationPage.getFontCounter().getCountPerValue());
document.getTextHeightCounter().addAll(classificationPage.getTextHeightCounter().getCountPerValue());
document.getFontStyleCounter().addAll(classificationPage.getFontStyleCounter().getCountPerValue());
}
private void buildPageStatistics(ClassificationPage classificationPage) {
// Collect all statistics for the classificationPage, except from blocks inside tables, as tables will always be added to BodyTextFrame.
for (AbstractPageBlock textBlock : classificationPage.getTextBlocks()) {
if (textBlock instanceof TextPageBlock) {
if (((TextPageBlock) textBlock).getSequences() == null) {
continue;
}
for (TextPositionSequence word : ((TextPageBlock) textBlock).getSequences()) {
classificationPage.getTextHeightCounter().add(word.getTextHeight());
classificationPage.getFontCounter().add(word.getFont());
classificationPage.getFontSizeCounter().add(word.getFontSize());
classificationPage.getFontStyleCounter().add(word.getFontStyle());
}
}
}
}
}
@@ -12,9 +12,9 @@ import java.util.Map;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableCells;
import com.knecon.fforesight.service.layoutparser.processor.model.table.CleanRulings;
import com.knecon.fforesight.service.layoutparser.processor.model.table.Ruling;
import com.knecon.fforesight.service.layoutparser.processor.python_api.model.table.TableCells;
import com.knecon.fforesight.service.layoutparser.processor.utils.DoubleComparisons;
import lombok.RequiredArgsConstructor;
@@ -25,10 +25,13 @@ import lombok.extern.slf4j.Slf4j;
@RequiredArgsConstructor
public class RulingCleaningService {
public CleanRulings getCleanRulings(List<TableCells> tableCells, List<Ruling> rulings, float minCharWidth, float maxCharHeight) {
private static final float THRESHOLD = 6;
public CleanRulings getCleanRulings(List<TableCells> tableCells, List<Ruling> rulings) {
if (!rulings.isEmpty()) {
snapPoints(rulings, minCharWidth, maxCharHeight);
snapPoints(rulings);
}
List<Ruling> vrs = new ArrayList<>();
@@ -53,14 +56,11 @@ public class RulingCleaningService {
}
List<Ruling> horizontalRulingLines = collapseOrientedRulings(hrs);
return CleanRulings.builder()
.vertical(verticalRulingLines)
.horizontal(horizontalRulingLines)
.build();
return CleanRulings.builder().vertical(verticalRulingLines).horizontal(horizontalRulingLines).build();
}
public void snapPoints(List<? extends Line2D.Float> rulings, float xThreshold, float yThreshold) {
public void snapPoints(List<? extends Line2D.Float> rulings) {
// collect points and keep a Line -> p1,p2 map
Map<Line2D.Float, Point2D[]> linesToPoints = new HashMap<>();
@@ -81,7 +81,7 @@ public class RulingCleaningService {
for (Point2D p : points.subList(1, points.size() - 1)) {
List<Point2D> last = groupedPoints.get(groupedPoints.size() - 1);
if (Math.abs(p.getX() - last.get(0).getX()) < xThreshold) {
if (Math.abs(p.getX() - last.get(0).getX()) < THRESHOLD) {
groupedPoints.get(groupedPoints.size() - 1).add(p);
} else {
groupedPoints.add(new ArrayList<>(Collections.singletonList(p)));
@@ -108,7 +108,7 @@ public class RulingCleaningService {
for (Point2D p : points.subList(1, points.size() - 1)) {
List<Point2D> last = groupedPoints.get(groupedPoints.size() - 1);
if (Math.abs(p.getY() - last.get(0).getY()) < yThreshold) {
if (Math.abs(p.getY() - last.get(0).getY()) < THRESHOLD) {
groupedPoints.get(groupedPoints.size() - 1).add(p);
} else {
groupedPoints.add(new ArrayList<>(Collections.singletonList(p)));
@@ -1,146 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.awt.geom.Rectangle2D;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Set;
import java.util.function.BiConsumer;
import java.util.function.BinaryOperator;
import java.util.function.Function;
import java.util.function.Supplier;
import java.util.stream.Collector;
import java.util.stream.Stream;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Point;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.CellRectangle;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionGrid;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionRectangle;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.NodeType;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Document;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Page;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.SemanticNode;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Table;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.TableCell;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class SectionGridCreatorService {
public SectionGrid createSectionGrid(Document document) {
Map<Integer, List<SectionRectangle>> sectionBBox = document.streamAllSubNodesOfType(NodeType.SECTION).map(SemanticNode::getBBox).collect(new SectionGridCollector());
Map<Integer, List<SectionRectangle>> paragraphBBox = document.streamAllSubNodesOfType(NodeType.PARAGRAPH).map(SemanticNode::getBBox).collect(new SectionGridCollector());
Map<Integer, List<SectionRectangle>> headlineBBox = document.streamAllSubNodesOfType(NodeType.HEADLINE).map(SemanticNode::getBBox).collect(new SectionGridCollector());
Map<Integer, List<SectionRectangle>> tableBBox = document.streamAllSubNodesOfType(NodeType.TABLE).map(node -> (Table) node).collect(new TableGridCollector());
var sectionGrid = new SectionGrid();
sectionGrid.setRectanglesPerPage(mergeMapsByConcatenatingLists(//
mergeMapsByConcatenatingLists(paragraphBBox, headlineBBox), //
mergeMapsByConcatenatingLists(sectionBBox, tableBBox)));
return sectionGrid;
}
private static abstract class GridCollector<T> implements Collector<T, Map<Integer, List<SectionRectangle>>, Map<Integer, List<SectionRectangle>>> {
@Override
public Supplier<Map<Integer, List<SectionRectangle>>> supplier() {
return HashMap::new;
}
@Override
public Function<Map<Integer, List<SectionRectangle>>, Map<Integer, List<SectionRectangle>>> finisher() {
return Function.identity();
}
@Override
public BinaryOperator<Map<Integer, List<SectionRectangle>>> combiner() {
return SectionGridCreatorService::mergeMapsByConcatenatingLists;
}
@Override
public Set<Characteristics> characteristics() {
return Set.of(Characteristics.IDENTITY_FINISH, Characteristics.CONCURRENT, Characteristics.UNORDERED);
}
}
private static class TableGridCollector extends GridCollector<Table> {
@Override
public BiConsumer<Map<Integer, List<SectionRectangle>>, Table> accumulator() {
return (map, table) -> table.getPages()
.forEach(page -> map.merge(page.getNumber(), List.of(toSectionRectangle(table, page, table.getPages().size())), SectionGridCreatorService::concatLists));
}
private static SectionRectangle toSectionRectangle(Table table, Page page, int numberOfParts) {
Rectangle2D rect = table.getBBox().get(page);
List<CellRectangle> tableCellRectangles = table.streamTableCells()
.map(TableCell::getBBox)
.map(map -> map.get(page))
.filter(Objects::nonNull)
.map(rectangle2D -> new CellRectangle(new Point((float) rectangle2D.getX(), (float) rectangle2D.getY()),
(float) rectangle2D.getWidth(),
(float) rectangle2D.getHeight()))
.toList();
return new SectionRectangle(new Point((float) rect.getX(), (float) rect.getY()),
(float) rect.getWidth(),
(float) rect.getHeight(),
1,
numberOfParts,
tableCellRectangles);
}
}
private static class SectionGridCollector extends GridCollector<Map<Page, Rectangle2D>> {
@Override
public BiConsumer<Map<Integer, List<SectionRectangle>>, Map<Page, Rectangle2D>> accumulator() {
return (mapToKeep, mapToMerge) -> mapToMerge.forEach((page, rectangle) -> mapToKeep.merge(page.getNumber(),
List.of(toSectionRectangle(rectangle, mapToMerge.values().size())),
SectionGridCreatorService::concatLists));
}
private static SectionRectangle toSectionRectangle(Rectangle2D rect, int numberOfParts) {
return new SectionRectangle(new Point((float) rect.getX(), (float) rect.getY()), (float) rect.getWidth(), (float) rect.getHeight(), 1, numberOfParts, null);
}
}
private static Map<Integer, List<SectionRectangle>> mergeMapsByConcatenatingLists(Map<Integer, List<SectionRectangle>> mapToKeep,
Map<Integer, List<SectionRectangle>> mapToMerge) {
mapToMerge.forEach((page, rectangle) -> mapToKeep.merge(page, rectangle, SectionGridCreatorService::concatLists));
return mapToKeep;
}
private static List<SectionRectangle> concatLists(List<SectionRectangle> l1, List<SectionRectangle> l2) {
return Stream.concat(l1.stream(), l2.stream()).toList();
}
}
@@ -240,7 +240,7 @@ public class SectionsBuilderService {
}
private ClassificationSection buildTextBlock(List<AbstractPageBlock> wordBlockList, String lastHeadline) {
public ClassificationSection buildTextBlock(List<AbstractPageBlock> wordBlockList, String lastHeadline) {
ClassificationSection section = new ClassificationSection();
@@ -0,0 +1,26 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.SimplifiedSectionText;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Document;
import com.knecon.fforesight.service.layoutparser.processor.model.graph.nodes.Section;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.SimplifiedText;
@Service
public class SimplifiedSectionTextService {
public SimplifiedText toSimplifiedText(Document document) {
List<SimplifiedSectionText> simplifiedSectionTexts = document.getMainSections().stream().map(this::toSimplifiedSectionText).toList();
return SimplifiedText.builder().numberOfPages(document.getNumberOfPages()).sectionTexts(simplifiedSectionTexts).build();
}
private SimplifiedSectionText toSimplifiedSectionText(Section section) {
return SimplifiedSectionText.builder().sectionNumber(section.getTreeId().get(0)).text(section.getTextBlock().getSearchText()).build();
}
}
@@ -12,7 +12,6 @@ import java.util.Set;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.table.Cell;
@@ -66,6 +65,17 @@ public class TableExtractionService {
};
public boolean contains(Cell cell, double x, double y, double w, double h) {
if (cell.isEmpty() || w <= 0 || h <= 0) {
return false;
}
double x0 = cell.getX();
double y0 = cell.getY();
return (x >= x0 - 2 && y >= y0 - 2 && (x + w) <= x0 + cell.getWidth() + 2 && (y + h) <= y0 + cell.getHeight() + 2);
}
/**
* Finds tables on a page and moves textblocks into cells of the found tables.
* Note: This algorithm uses Pdf Coordinate System where {0,0} rotated with the page rotation.
@@ -79,16 +89,17 @@ public class TableExtractionService {
* @param cleanRulings The lines used to build the table.
* @param page Page object that contains textblocks and statistics.
*/
public void extractTables(CleanRulings cleanRulings, ClassificationPage page, LayoutParsingType layoutParsingType) {
public void extractTables(CleanRulings cleanRulings, ClassificationPage page) {
List<Cell> cells = findCells(cleanRulings.getHorizontal(), cleanRulings.getVertical(), layoutParsingType);
List<Cell> cells = findCells(cleanRulings.getHorizontal(), cleanRulings.getVertical());
List<TextPageBlock> toBeRemoved = new ArrayList<>();
for (AbstractPageBlock abstractPageBlock : page.getTextBlocks()) {
TextPageBlock textBlock = (TextPageBlock) abstractPageBlock;
for (Cell cell : cells) {
if (cell.hasMinimumSize() && cell.intersects(textBlock.getPdfMinX(),
if (cell.hasMinimumSize() && contains(cell,
textBlock.getPdfMinX(),
textBlock.getPdfMinY(),
textBlock.getPdfMaxX() - textBlock.getPdfMinX(),
textBlock.getPdfMaxY() - textBlock.getPdfMinY())) {
@@ -102,7 +113,7 @@ public class TableExtractionService {
cells = new ArrayList<>(new HashSet<>(cells));
DoubleComparisons.sort(cells, Rectangle.ILL_DEFINED_ORDER);
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells).stream().filter(r -> r.getWidth() > 0f && r.getHeight() > 0f).toList();
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells);
List<TablePageBlock> tables = new ArrayList<>();
for (Rectangle area : spreadsheetAreas) {
@@ -135,16 +146,14 @@ public class TableExtractionService {
}
public List<Cell> findCells(List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines, LayoutParsingType layoutParsingType) {
public List<Cell> findCells(List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
if (layoutParsingType.equals(LayoutParsingType.TAAS)) {
// TODO: breaks some tables, for example "1 Abamectin Prr.pdf" try to fix this upstream in RulingCleaningService
for (Ruling r : horizontalRulingLines) {
if (r.getX2() < r.getX1()) {
double a = r.getX2();
r.x2 = (float) r.getX1();
r.x1 = (float) a;
}
// Fix for 211.pdf
for (Ruling r : horizontalRulingLines) {
if (r.getX2() < r.getX1()) {
double a = r.getX2();
r.x2 = (float) r.getX1();
r.x1 = (float) a;
}
}
@@ -1,74 +0,0 @@
package com.knecon.fforesight.service.layoutparser.processor.services;
import java.io.IOException;
import java.io.InputStream;
import java.util.Collection;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.springframework.core.io.ClassPathResource;
import com.knecon.fforesight.service.layoutparser.processor.model.PageContents;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.services.parsing.PDFLinesTextStripper;
import com.knecon.fforesight.service.layoutparser.processor.utils.RectangleTransformations;
import lombok.experimental.UtilityClass;
@UtilityClass
public class TextPositionSequenceSorter {
public List<PageContents> getSortedTextPositionsWithPages(String filename) throws IOException {
List<PageContents> textPositionSequencesPerPage = new LinkedList<>();
try (InputStream inputStream = new ClassPathResource(filename).getInputStream()) {
try (PDDocument pdDocument = Loader.loadPDF(inputStream)) {
for (int pageNumber = 1; pageNumber < pdDocument.getNumberOfPages() + 1; pageNumber++) {
PDFLinesTextStripper stripper = new PDFLinesTextStripper();
PDPage pdPage = pdDocument.getPage(pageNumber - 1);
stripper.setPageNumber(pageNumber);
stripper.setSortByPosition(true);
stripper.setStartPage(pageNumber);
stripper.setEndPage(pageNumber);
stripper.setPdpage(pdPage);
stripper.getText(pdDocument);
Map<Float, List<TextPositionSequence>> sortedTextPositionSequencesPerDir = stripper.getTextPositionSequences()
.stream()
.collect(Collectors.groupingBy(textPositionSequence -> textPositionSequence.getDir().getDegrees()));
var sortedTextPositionSequences = sortByDirAccordingToPageRotation(sortedTextPositionSequencesPerDir, pdPage.getRotation());
textPositionSequencesPerPage.add(new PageContents(sortedTextPositionSequences,
RectangleTransformations.toRectangle2D(pdPage.getCropBox()),
RectangleTransformations.toRectangle2D(pdPage.getMediaBox())));
}
}
}
return textPositionSequencesPerPage;
}
public List<TextPositionSequence> sortByDirAccordingToPageRotation(Map<Float, List<TextPositionSequence>> sortedTextPositionSequencesPerDir, int rotation) {
LinkedList<Float> sortedKeys = new LinkedList<>(sortedTextPositionSequencesPerDir.keySet().stream().sorted().toList());
for (int i = 0; i < sortedKeys.size(); i++) {
if (sortedKeys.get(i) < rotation) {
Float keyToSwap = sortedKeys.remove(i);
sortedKeys.addLast(keyToSwap);
}
}
return sortedKeys.stream().map(sortedTextPositionSequencesPerDir::get).flatMap(Collection::stream).toList();
}
}
@@ -5,6 +5,9 @@ import static java.util.stream.Collectors.toSet;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import org.springframework.stereotype.Service;
@@ -23,6 +26,8 @@ public class DocuMineBlockificationService {
static final float THRESHOLD = 1f;
Pattern pattern = Pattern.compile("^(\\d{1,2}\\.){1,3}\\d{1,2}\\.?\\s[0-9A-Za-z ()-]{2,50}", Pattern.CASE_INSENSITIVE);
/**
* This method is building blocks by expanding the minX/maxX and minY/maxY value on each word that is not split by the conditions.
@@ -43,7 +48,6 @@ public class DocuMineBlockificationService {
float maxX = 0;
float minY = 1000;
float maxY = 0;
TextPositionSequence prev = null;
boolean wasSplitted = false;
@@ -60,7 +64,10 @@ public class DocuMineBlockificationService {
boolean splitByOtherFontAndOtherY = prev != null && prev.getMaxYDirAdj() != word.getMaxYDirAdj() && (word.getFontStyle().contains("bold") && !prev.getFontStyle()
.contains("bold") || prev.getFontStyle().contains("bold") && !word.getFontStyle().contains("bold"));
if (prev != null && (lineSeparation || startFromTop || splitByDir || isSplitByRuling || splitByOtherFontAndOtherY || negativeXGap)) {
Matcher matcher = pattern.matcher(chunkWords.stream().collect(Collectors.joining(" ")).toString());
boolean startsOnSameX = Math.abs(minX - word.getMinXDirAdj()) < 5 && matcher.matches();
if (prev != null && (lineSeparation || startFromTop || splitByDir || isSplitByRuling || splitByOtherFontAndOtherY || negativeXGap || startsOnSameX)) {
Orientation prevOrientation = null;
if (!chunkBlockList1.isEmpty()) {
@@ -231,3 +238,4 @@ public class DocuMineBlockificationService {
}
}
@@ -57,7 +57,7 @@ public class RedactManagerBlockificationService {
boolean isSplitByRuling = isSplitByRuling(minX, minY, maxX, maxY, word, horizontalRulingLines, verticalRulingLines);
boolean splitByDir = prev != null && !prev.getDir().equals(word.getDir());
if (prev != null && (lineSeparation || startFromTop || splitByX || splitByDir || isSplitByRuling)) {
if (prev != null && (splitByDir || isSplitByRuling)) {
Orientation prevOrientation = null;
if (!chunkBlockList.isEmpty()) {
@@ -167,7 +167,7 @@ public class RedactManagerBlockificationService {
}
private TextPageBlock buildTextBlock(List<TextPositionSequence> wordBlockList, int indexOnPage) {
public TextPageBlock buildTextBlock(List<TextPositionSequence> wordBlockList, int indexOnPage) {
TextPageBlock textBlock = null;
@@ -1,13 +1,8 @@
package com.knecon.fforesight.service.layoutparser.processor.services.blockification;
import java.util.ArrayList;
import java.util.Iterator;
import java.util.LinkedList;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import org.springframework.stereotype.Service;
// TODO: figure out, why this fails the build
// import static com.knecon.fforesight.service.layoutparser.processor.services.factory.SearchTextWithTextPositionFactory.HEIGHT_PADDING;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
@@ -16,14 +11,25 @@ import com.knecon.fforesight.service.layoutparser.processor.model.table.Ruling;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.utils.RulingTextDirAdjustUtil;
import org.springframework.stereotype.Service;
import java.util.*;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Stream;
@Service
@SuppressWarnings("all")
public class TaasBlockificationService {
private static final float THRESHOLD = 1f;
private static final float Y_GAP_SPLIT_HEIGHT_MODIFIER = 1.25f;
private static final float Y_GAP_SPLIT_HEIGHT_MODIFIER = 1.25f; // multiplied with text height
private static final float INTERSECTS_Y_THRESHOLD = 4;// 2 * HEIGHT_PADDING // This is exactly 2 times our position height padding. This is required to find boxes that are visually intersecting.
private static final int X_GAP_SPLIT_CONSTANT = 50;
public static final int X_ALIGNMENT_THRESHOLD = 1;
public static final int NEGATIVE_X_GAP_THRESHOLD = -5;
private Pattern listIdentifier = Pattern.compile("^(?:(?:[1-9]|1\\d|20|[ivxlc]|[a-z])\\s*(?:[.)]))|\\uF0B7", Pattern.CASE_INSENSITIVE);
/**
@@ -39,14 +45,28 @@ public class TaasBlockificationService {
public ClassificationPage blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
List<TextPageBlock> classificationTextBlocks = constructFineGranularTextPageBlocks(textPositions, horizontalRulingLines, verticalRulingLines);
classificationTextBlocks = mergeFineGranularTextPageBlocks(classificationTextBlocks);
classificationTextBlocks = mergeTextPageBlocksAligningX(classificationTextBlocks);
classificationTextBlocks = mergeIntersectingTextBlocksUntilConvergence(classificationTextBlocks);
return new ClassificationPage(new ArrayList<>(classificationTextBlocks.stream().map(classificationTextBlock -> (AbstractPageBlock) classificationTextBlock).toList()));
}
private List<TextPageBlock> mergeFineGranularTextPageBlocks(List<TextPageBlock> classificationTextBlocks) {
private List<TextPageBlock> mergeIntersectingTextBlocksUntilConvergence(List<TextPageBlock> classificationTextBlocks) {
int currentSize = classificationTextBlocks.size();
while (true) {
classificationTextBlocks = mergeTextPageBlocksAlmostIntersecting(classificationTextBlocks);
if (classificationTextBlocks.size() == currentSize) {
break;
}
currentSize = classificationTextBlocks.size();
}
return classificationTextBlocks;
}
private List<TextPageBlock> mergeTextPageBlocksAligningX(List<TextPageBlock> classificationTextBlocks) {
if (classificationTextBlocks.isEmpty()) {
return new ArrayList<>();
@@ -55,16 +75,29 @@ public class TaasBlockificationService {
List<TextPageBlock> currentTextBlocksToMerge = new LinkedList<>();
textBlocksToMerge.add(currentTextBlocksToMerge);
TextPageBlock previousTextBlock = null;
Float lastLineGap = null;
for (TextPageBlock currentTextBlock : classificationTextBlocks) {
if (previousTextBlock == null) {
currentTextBlocksToMerge.add(currentTextBlock);
previousTextBlock = currentTextBlock;
continue;
}
boolean alignsXRight = Math.abs(currentTextBlock.getPdfMaxX() - previousTextBlock.getPdfMaxX()) < 1;
boolean smallYGap = Math.abs(currentTextBlock.getPdfMaxY() - previousTextBlock.getPdfMinY()) < 5;
if (alignsXRight && smallYGap) {
Matcher listIdentifierPattern = listIdentifier.matcher(currentTextBlock.getText());
boolean isListIdentifier = listIdentifierPattern.find();
boolean yGap = Math.abs(currentTextBlock.getPdfMaxY() - previousTextBlock.getPdfMinY()) < previousTextBlock.getMostPopularWordHeight() * Y_GAP_SPLIT_HEIGHT_MODIFIER;
boolean sameFont = previousTextBlock.getMostPopularWordFont().equals(currentTextBlock.getMostPopularWordFont()) && previousTextBlock.getMostPopularWordFontSize() == currentTextBlock.getMostPopularWordFontSize();
// boolean yGap = previousTextBlock != null && currentTextBlock.getMinYDirAdj() - maxY > Math.min(word.getHeight(), prev.getHeight()) * Y_GAP_SPLIT_HEIGHT_MODIFIER;
boolean alignsXRight = Math.abs(currentTextBlock.getPdfMaxX() - previousTextBlock.getPdfMaxX()) < X_ALIGNMENT_THRESHOLD;
boolean alignsXLeft = Math.abs(currentTextBlock.getPdfMinX() - previousTextBlock.getPdfMinX()) < X_ALIGNMENT_THRESHOLD;
// boolean smallYGap = Math.abs(currentTextBlock.getPdfMaxY() - previousTextBlock.getPdfMinY()) < yGap;
if (yGap && sameFont && !isListIdentifier) {
currentTextBlocksToMerge.add(currentTextBlock);
} else {
currentTextBlocksToMerge = new LinkedList<>();
currentTextBlocksToMerge.add(currentTextBlock);
@@ -76,6 +109,23 @@ public class TaasBlockificationService {
}
private List<TextPageBlock> mergeTextPageBlocksAlmostIntersecting(List<TextPageBlock> textPageBlocks) {
Set<TextPageBlock> alreadyMerged = new HashSet<>();
List<List<TextPageBlock>> textBlocksToMerge = new LinkedList<>();
for (TextPageBlock textPageBlock : textPageBlocks) {
if (alreadyMerged.contains(textPageBlock)) {
continue;
}
alreadyMerged.add(textPageBlock);
textBlocksToMerge.add(Stream.concat(Stream.of(textPageBlock),
textPageBlocks.stream().filter(textPageBlock2 -> textPageBlock.almostIntersects(textPageBlock2, INTERSECTS_Y_THRESHOLD, 0) && !alreadyMerged.contains(textPageBlock2)).peek(alreadyMerged::add))
.toList());
}
return textBlocksToMerge.stream().map(TextPageBlock::merge).toList();
}
private void assignOrientations(List<TextPageBlock> classificationTextBlocks) {
Iterator<TextPageBlock> itty = classificationTextBlocks.iterator();
@@ -128,8 +178,8 @@ public class TaasBlockificationService {
private List<TextPageBlock> constructFineGranularTextPageBlocks(List<TextPositionSequence> textPositions,
List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
int indexOnPage = 0;
List<TextPositionSequence> wordClusterToCombine = new ArrayList<>();
@@ -138,18 +188,18 @@ public class TaasBlockificationService {
float minX = 1000, maxX = 0, minY = 1000, maxY = 0;
TextPositionSequence prev = null;
// TODO: make static final constant
var listIdentitifier = Pattern.compile("\\b(?:[1-9]|1\\d|20|[ivxlc]|[a-z])\\s*(?:[.)])", Pattern.CASE_INSENSITIVE);
boolean wasSplitted = false;
Float splitX1 = null;
for (TextPositionSequence word : textPositions) {
Matcher listIdentifierPattern = listIdentitifier.matcher(word.toString());
Matcher listIdentifierPattern = listIdentifier.matcher(word.toString());
boolean yGap = prev != null && word.getMinYDirAdj() - maxY > Math.min(word.getHeight(), prev.getHeight()) * Y_GAP_SPLIT_HEIGHT_MODIFIER;
boolean sameLine = prev != null && equalsWithThreshold(prev.getMinYDirAdj(), word.getMinYDirAdj());
boolean positiveXGapInline = prev != null && maxX + X_GAP_SPLIT_CONSTANT < word.getMinXDirAdj() && sameLine;
boolean negativeXGap = prev != null && word.getMinXDirAdj() - minX < -5;
boolean negativeXGap = prev != null && word.getMinXDirAdj() - minX < NEGATIVE_X_GAP_THRESHOLD;
boolean startFromTop = prev != null && word.getMinYDirAdj() < prev.getMinYDirAdj() - prev.getTextHeight();
boolean newLineAfterSplit = prev != null && word.getMinYDirAdj() != prev.getMinYDirAdj() && wasSplitted && splitX1 != word.getMinXDirAdj();
boolean splitByRuling = isSplitByRuling(minX, minY, maxX, maxY, word, horizontalRulingLines, verticalRulingLines);
@@ -164,7 +214,7 @@ public class TaasBlockificationService {
Orientation prevOrientation = null;
if (!classificationTextBlocks.isEmpty()) {
prevOrientation = classificationTextBlocks.get(classificationTextBlocks.size() - 1).getOrientation();
prevOrientation = classificationTextBlocks.get(classificationTextBlocks.size() - X_ALIGNMENT_THRESHOLD).getOrientation();
}
TextPageBlock classificationTextBlock = TextPageBlock.fromTextPositionSequences(wordClusterToCombine);
@@ -5,43 +5,36 @@ import java.util.Locale;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import com.knecon.fforesight.service.layoutparser.processor.utils.MarkedContentUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.PageBlockType;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.services.BodyTextFrameService;
import com.knecon.fforesight.service.layoutparser.processor.utils.PositionUtils;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class DocuMineClassificationService {
private final BodyTextFrameService bodyTextFrameService;
private static final Pattern pattern = Pattern.compile("^(\\d{1,1}\\.?){1,3}\\d{1,2}\\.?\\s[0-9A-Za-z\\[\\]\\-]{2,50}", Pattern.CASE_INSENSITIVE);
private static final Pattern pattern2 = Pattern.compile(".*\\d{4}$", Pattern.CASE_INSENSITIVE);
private static final Pattern pattern = Pattern.compile("^(\\d{1,2}\\.){1,3}\\d{1,2}\\.?\\s[0-9A-Za-z \\[\\]]{2,50}", Pattern.CASE_INSENSITIVE);
private static final Pattern pattern2 = Pattern.compile("\\p{L}{3,}", Pattern.CASE_INSENSITIVE);
private static final Pattern pattern3 = Pattern.compile("^(\\d{1,1}\\.){1,3}\\d{1,2}\\.?\\s[a-z]{1,2}\\/[a-z]{1,2}.*");
public void classifyDocument(ClassificationDocument document) {
Rectangle bodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false, LayoutParsingType.DOCUMINE);
Rectangle landscapeBodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true, LayoutParsingType.DOCUMINE);
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
for (ClassificationPage page : document.getPages()) {
bodyTextFrameService.setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
classifyPage(page, document, headlineFontSizes);
}
}
@@ -70,12 +63,16 @@ public class DocuMineClassificationService {
textBlock.setClassification(PageBlockType.OTHER);
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.HEADER)
|| PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())
) {
textBlock.setClassification(PageBlockType.HEADER);
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
} else if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.FOOTER)
|| PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())
) {
textBlock.setClassification(PageBlockType.FOOTER);
} else if (page.getPageNumber() == 1 && (PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock,
document.getTextHeightCounter().getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter().getMostPopular() || page.getTextBlocks()
@@ -115,4 +112,4 @@ public class DocuMineClassificationService {
}
}
}
}
@@ -3,16 +3,14 @@ package com.knecon.fforesight.service.layoutparser.processor.services.classifica
import java.util.List;
import java.util.regex.Pattern;
import com.knecon.fforesight.service.layoutparser.processor.utils.MarkedContentUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
import com.knecon.fforesight.service.layoutparser.processor.model.PageBlockType;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.services.BodyTextFrameService;
import com.knecon.fforesight.service.layoutparser.processor.utils.PositionUtils;
import lombok.RequiredArgsConstructor;
@@ -23,19 +21,14 @@ import lombok.extern.slf4j.Slf4j;
@RequiredArgsConstructor
public class RedactManagerClassificationService {
private final BodyTextFrameService bodyTextFrameService;
public void classifyDocument(ClassificationDocument document) {
Rectangle bodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false, LayoutParsingType.REDACT_MANAGER);
Rectangle landscapeBodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true, LayoutParsingType.REDACT_MANAGER);
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
for (ClassificationPage page : document.getPages()) {
bodyTextFrameService.setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
classifyPage(page, document, headlineFontSizes);
}
}
@@ -59,11 +52,13 @@ public class RedactManagerClassificationService {
textBlock.setClassification(PageBlockType.OTHER);
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.HEADER)
|| PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification(PageBlockType.HEADER);
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
} else if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.FOOTER)
|| PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification(PageBlockType.FOOTER);
} else if (page.getPageNumber() == 1 && (PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock,
@@ -3,10 +3,9 @@ package com.knecon.fforesight.service.layoutparser.processor.services.classifica
import java.util.List;
import java.util.regex.Pattern;
import com.knecon.fforesight.service.layoutparser.processor.utils.MarkedContentUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.knecon.fforesight.service.layoutparser.internal.api.queue.LayoutParsingType;
import com.knecon.fforesight.service.layoutparser.processor.model.AbstractPageBlock;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationDocument;
import com.knecon.fforesight.service.layoutparser.processor.model.ClassificationPage;
@@ -28,14 +27,13 @@ public class TaasClassificationService {
public void classifyDocument(ClassificationDocument document) {
Rectangle bodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false, LayoutParsingType.TAAS);
Rectangle landscapeBodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true, LayoutParsingType.TAAS);
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
for (ClassificationPage page : document.getPages()) {
bodyTextFrameService.setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
classifyPage(page, document, headlineFontSizes);
}
}
@@ -59,9 +57,11 @@ public class TaasClassificationService {
textBlock.setClassification(PageBlockType.OTHER);
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation())) {
if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.HEADER)
|| PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation())) {
textBlock.setClassification(PageBlockType.HEADER);
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation())) {
} else if (MarkedContentUtils.intersects(textBlock, page.getMarkedContentBboxPerType(), MarkedContentUtils.FOOTER)
|| PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation())) {
textBlock.setClassification(PageBlockType.FOOTER);
} else if (page.getPageNumber() == 1 && (PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock,
document.getTextHeightCounter().getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter().getMostPopular() || page.getTextBlocks()
@@ -0,0 +1,48 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum;
import java.util.List;
import java.util.stream.Collectors;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Character;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Zone;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service.LineBuilderService;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service.NearestNeighbourService;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service.ReadingOrderService;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service.SpacingService;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service.ZoneBuilderService;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class DocstrumSegmentationService {
private final NearestNeighbourService nearestNeighbourService;
private final SpacingService spacingService;
private final LineBuilderService lineBuilderService;
private final ZoneBuilderService zoneBuilderService;
private final ReadingOrderService readingOrderService;
public List<Zone> segmentPage(List<TextPositionSequence> textPositions) {
var positions = textPositions.stream().map(TextPositionSequence::getTextPositions).flatMap(List::stream).toList();
var characters = positions.stream().map(Character::new).collect(Collectors.toList());
nearestNeighbourService.findNearestNeighbors(characters);
var characterSpacing = spacingService.computeCharacterSpacing(characters);
var lineSpacing = spacingService.computeLineSpacing(characters);
var lines = lineBuilderService.buildLines(characters, characterSpacing, lineSpacing);
var zones = zoneBuilderService.buildZones(lines, characterSpacing, lineSpacing);
return readingOrderService.resolve(zones);
}
}
@@ -0,0 +1,90 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
/**
* Filter class for neighbor objects that checks if the angle of the
* neighbor is within specified range.
*/
public abstract class AngleFilter {
private final double lowerAngle;
private final double upperAngle;
private AngleFilter(double lowerAngle, double upperAngle) {
this.lowerAngle = lowerAngle;
this.upperAngle = upperAngle;
}
/**
* Constructs new angle filter.
*
* @param lowerAngle minimum angle in range [-3*pi/2, pi/2)
* @param upperAngle maximum angle in range [-pi/2, 3*pi/2)
* @return newly constructed angle filter
*/
public static AngleFilter newInstance(double lowerAngle, double upperAngle) {
if (lowerAngle < -Math.PI / 2) {
lowerAngle += Math.PI;
}
if (upperAngle >= Math.PI / 2) {
upperAngle -= Math.PI;
}
if (lowerAngle <= upperAngle) {
return new AndFilter(lowerAngle, upperAngle);
} else {
return new OrFilter(lowerAngle, upperAngle);
}
}
public double getLowerAngle() {
return lowerAngle;
}
public double getUpperAngle() {
return upperAngle;
}
public abstract boolean matches(Neighbor neighbor);
public static final class AndFilter extends AngleFilter {
private AndFilter(double lowerAngle, double upperAngle) {
super(lowerAngle, upperAngle);
}
@Override
public boolean matches(Neighbor neighbor) {
return getLowerAngle() <= neighbor.getAngle() && neighbor.getAngle() < getUpperAngle();
}
}
public static final class OrFilter extends AngleFilter {
private OrFilter(double lowerAngle, double upperAngle) {
super(lowerAngle, upperAngle);
}
@Override
public boolean matches(Neighbor neighbor) {
return getLowerAngle() <= neighbor.getAngle() || neighbor.getAngle() < getUpperAngle();
}
}
}
@@ -0,0 +1,48 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.awt.geom.Rectangle2D;
import lombok.Data;
@Data
public abstract class BoundingBox {
private Rectangle2D bBox;
public double getX() {
return bBox.getX();
}
public double getY() {
return bBox.getY();
}
public double getWidth() {
return bBox.getWidth();
}
public double getHeight() {
return bBox.getHeight();
}
public double getArea() {
return (bBox.getHeight() * bBox.getWidth());
}
public boolean contains(Rectangle2D contained, double tolerance) {
return bBox.getX() <= contained.getX() + tolerance && bBox.getY() <= contained.getY() + tolerance && bBox.getX() + bBox.getWidth() >= contained.getX() + contained.getWidth() - tolerance && bBox.getY() + bBox.getHeight() >= contained.getY() + contained.getHeight() - tolerance;
}
}
@@ -0,0 +1,69 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.util.ArrayList;
import java.util.List;
import com.knecon.fforesight.service.layoutparser.processor.model.text.RedTextPosition;
import lombok.Data;
@Data
public class Character {
private final double x;
private final double y;
private final RedTextPosition textPosition;
private List<Neighbor> neighbors = new ArrayList<>();
public Character(RedTextPosition chunk) {
this.x = chunk.getXDirAdj() + chunk.getWidthDirAdj() / 2;
this.y = chunk.getYDirAdj() + chunk.getHeightDir() / 2;
this.textPosition = chunk;
}
public double getHeight() {
return textPosition.getHeightDir();
}
public double distance(Character character) {
double dx = getX() - character.getX();
double dy = getY() - character.getY();
return Math.sqrt(dx * dx + dy * dy);
}
public double horizontalDistance(Character character) {
return Math.abs(getX() - character.getX());
}
public double verticalDistance(Character character) {
return Math.abs(getY() - character.getY());
}
public void setNeighbors(List<Neighbor> neighbors) {
this.neighbors = neighbors;
}
public double angle(Character character) {
if (getX() > character.getX()) {
return Math.atan2(getY() - character.getY(), getX() - character.getX());
} else {
return Math.atan2(character.getY() - getY(), character.getX() - getX());
}
}
}
@@ -0,0 +1,212 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.util.AbstractSet;
import java.util.Collection;
import java.util.HashMap;
import java.util.Iterator;
import java.util.Map;
import java.util.NoSuchElementException;
import java.util.Set;
public class DisjointSets<E> implements Iterable<Set<E>> {
private final Map<E, Entry<E>> map = new HashMap<E, Entry<E>>();
/**
* Constructs a new set of singletons.
*
* @param c elements of singleton sets
*/
public DisjointSets(Collection<? extends E> c) {
for (E element : c) {
map.put(element, new Entry<E>(element));
}
}
/**
* Checks if elements are in the same subsets.
*
* @param e1 element from a subset
* @param e2 element from a subset
* @return true if elements are in the same subset; false otherwise
*/
public boolean areTogether(E e1, E e2) {
return map.get(e1).findRepresentative() == map.get(e2).findRepresentative();
}
/**
* Merges subsets which elements e1 and e2 belong to.
*
* @param e1 element from a subset
* @param e2 element from a subset
*/
public void union(E e1, E e2) {
Entry<E> r1 = map.get(e1).findRepresentative();
Entry<E> r2 = map.get(e2).findRepresentative();
if (r1 != r2) {
if (r1.size <= r2.size) {
r2.mergeWith(r1);
} else {
r1.mergeWith(r2);
}
}
}
@Override
public Iterator<Set<E>> iterator() {
return new Iterator<Set<E>>() {
private final Iterator<Entry<E>> iterator = map.values().iterator();
private Entry<E> nextRepresentative;
{
findNextRepresentative();
}
@Override
public boolean hasNext() {
return nextRepresentative != null;
}
@Override
public Set<E> next() {
if (nextRepresentative == null) {
throw new NoSuchElementException();
}
Set<E> result = nextRepresentative.asSet();
findNextRepresentative();
return result;
}
private void findNextRepresentative() {
while (iterator.hasNext()) {
Entry<E> candidate = iterator.next();
if (candidate.isRepresentative()) {
nextRepresentative = candidate;
return;
}
}
nextRepresentative = null;
}
@Override
public void remove() {
throw new UnsupportedOperationException();
}
};
}
private static class Entry<E> {
private int size = 1;
private final E value;
private Entry<E> parent = this;
private Entry<E> next = null;
private Entry<E> last = this;
Entry(E value) {
this.value = value;
}
void mergeWith(Entry<E> otherRepresentative) {
size += otherRepresentative.size;
last.next = otherRepresentative;
last = otherRepresentative.last;
otherRepresentative.parent = this;
}
Entry<E> findRepresentative() {
Entry<E> representative = parent;
while (representative.parent != representative) {
representative = representative.parent;
}
for (Entry<E> entry = this; entry != representative; ) {
Entry<E> nextEntry = entry.parent;
entry.parent = representative;
entry = nextEntry;
}
return representative;
}
boolean isRepresentative() {
return parent == this;
}
Set<E> asSet() {
return new AbstractSet<E>() {
@Override
public Iterator<E> iterator() {
return new Iterator<E>() {
private Entry<E> nextEntry = findRepresentative();
@Override
public boolean hasNext() {
return nextEntry != null;
}
@Override
public E next() {
if (nextEntry == null) {
throw new NoSuchElementException();
}
E result = nextEntry.value;
nextEntry = nextEntry.next;
return result;
}
@Override
public void remove() {
throw new UnsupportedOperationException();
}
};
}
@Override
public int size() {
return findRepresentative().size;
}
};
}
}
}
@@ -0,0 +1,199 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.util.Iterator;
import java.util.NoSuchElementException;
public class Histogram implements Iterable<Histogram.Bin> {
private static final double EPSILON = 1.0e-6;
private final double min;
private final double delta;
private final double resolution;
private double[] frequencies;
/**
* Constructs a new histogram for values in range [minValue, maxValue] with
* given resolution.
*
* @param minValue - minimum allowed value
* @param maxValue - maximum allowed value
* @param resolution - histogram's resolution
*/
public Histogram(double minValue, double maxValue, double resolution) {
this.min = minValue - EPSILON;
this.delta = maxValue - minValue + 2 * EPSILON;
int size = Math.max(1, (int) Math.round((maxValue - minValue) / resolution));
this.resolution = this.delta / size;
this.frequencies = new double[size];
}
public void kernelSmooth(double[] kernel) {
double[] newFrequencies = new double[frequencies.length];
int shift = (kernel.length - 1) / 2;
for (int i = 0; i < kernel.length; i++) {
int jStart = Math.max(0, i - shift);
int jEnd = Math.min(frequencies.length, frequencies.length + i - shift);
for (int j = jStart; j < jEnd; j++) {
newFrequencies[j - i + shift] += kernel[i] * frequencies[j];
}
}
frequencies = newFrequencies;
}
public void circularKernelSmooth(double[] kernel) {
double[] newFrequencies = new double[frequencies.length];
int shift = (kernel.length - 1) / 2;
for (int i = 0; i < frequencies.length; i++) {
for (int d = 0; d < kernel.length; d++) {
int j = i + d - shift;
if (j < 0) {
j += frequencies.length;
} else if (j >= frequencies.length) {
j -= frequencies.length;
}
newFrequencies[i] += kernel[d] * frequencies[j];
}
}
frequencies = newFrequencies;
}
public double[] createGaussianKernel(double length, double stdDeviation) {
int r = (int) Math.round(length / resolution) / 2;
stdDeviation /= resolution;
int size = 2 * r + 1;
double[] kernel = new double[size];
double sum = 0;
double b = 2 * stdDeviation * stdDeviation;
double a = 1 / Math.sqrt(Math.PI * b);
for (int i = 0; i < size; i++) {
kernel[i] = a * Math.exp(-(i - r) * (i - r) / b);
sum += kernel[i];
}
for (int i = 0; i < size; i++) {
kernel[i] /= sum;
}
return kernel;
}
public void circularGaussianSmooth(double windowLength, double stdDeviation) {
circularKernelSmooth(createGaussianKernel(windowLength, stdDeviation));
}
public void gaussianSmooth(double windowLength, double stdDeviation) {
kernelSmooth(createGaussianKernel(windowLength, stdDeviation));
}
/**
* Adds single occurrence of given value to the histogram.
*
* @param value inserted values
*/
public void add(double value) {
frequencies[(int) ((value - min) / resolution)] += 1.0;
}
/**
* Returns histogram's number of bins.
*
* @return number of bins
*/
public int getSize() {
return frequencies.length;
}
/**
* Finds the histogram's peak value.
*
* @return peak value
*/
public double getPeakValue() {
int peakIndex = 0;
for (int i = 1; i < frequencies.length; i++) {
if (frequencies[i] > frequencies[peakIndex]) {
peakIndex = i;
}
}
int peakEndIndex = peakIndex + 1;
final double EPS = 0.0001;
while (peakEndIndex < frequencies.length && Math.abs(frequencies[peakEndIndex] - frequencies[peakIndex]) < EPS) {
peakEndIndex++;
}
return ((double) peakIndex + peakEndIndex) / 2 * resolution + min;
}
@Override
public Iterator<Bin> iterator() {
return new Iterator() {
private int index = 0;
@Override
public boolean hasNext() {
return index < frequencies.length;
}
@Override
public Object next() {
if (index >= frequencies.length) {
throw new NoSuchElementException();
}
return new Bin(index++);
}
@Override
public void remove() {
throw new UnsupportedOperationException("Not supported yet.");
}
};
}
public final class Bin {
private final int index;
private Bin(int index) {
this.index = index;
}
public double getValue() {
return (index + 0.5) * resolution + min;
}
}
}
@@ -0,0 +1,167 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import com.knecon.fforesight.service.layoutparser.processor.model.text.TextPositionSequence;
import lombok.Data;
@Data
public class Line extends BoundingBox {
private static final double WORD_DISTANCE_MULTIPLIER = 0.2;
private final double x0;
private final double y0;
private final double x1;
private final double y1;
private final double height;
private final List<Character> characters;
private final List<TextPositionSequence> words = new ArrayList<>();
public Line(List<Character> characters, double wordSpacing) {
this.characters = characters;
if (characters.size() >= 2) {
// Simple linear regression
double sx = 0.0, sxx = 0.0, sxy = 0.0, sy = 0.0;
for (Character component : characters) {
sx += component.getX();
sxx += component.getX() * component.getX();
sxy += component.getX() * component.getY();
sy += component.getY();
}
double b = (characters.size() * sxy - sx * sy) / (characters.size() * sxx - sx * sx);
double a = (sy - b * sx) / characters.size();
this.x0 = characters.get(0).getX();
this.y0 = a + b * this.x0;
this.x1 = characters.get(characters.size() - 1).getX();
this.y1 = a + b * this.x1;
} else if (!characters.isEmpty()) {
Character component = characters.get(0);
double dx = component.getTextPosition().getWidthDirAdj() / 3;
double dy = dx * Math.tan(0);
this.x0 = component.getX() - dx;
this.x1 = component.getX() + dx;
this.y0 = component.getY() - dy;
this.y1 = component.getY() + dy;
} else {
throw new IllegalArgumentException("Component list must not be empty");
}
height = computeHeight();
computeWords(wordSpacing * WORD_DISTANCE_MULTIPLIER);
buildBox();
}
public double getAngle() {
return Math.atan2(y1 - y0, x1 - x0);
}
public double getLength() {
return Math.sqrt((x0 - x1) * (x0 - x1) + (y0 - y1) * (y0 - y1));
}
private double computeHeight() {
double sum = 0.0;
for (Character component : characters) {
sum += component.getHeight();
}
return sum / characters.size();
}
public double angularDifference(Line j) {
double diff = Math.abs(getAngle() - j.getAngle());
if (diff <= Math.PI / 2) {
return diff;
} else {
return Math.PI - diff;
}
}
public double horizontalDistance(Line other) {
double[] xs = new double[4];
xs[0] = x0;
xs[1] = x1;
xs[2] = other.x0;
xs[3] = other.x1;
boolean overlapping = xs[1] >= xs[2] && xs[3] >= xs[0];
Arrays.sort(xs);
return Math.abs(xs[2] - xs[1]) * (overlapping ? 1 : -1);
}
public double verticalDistance(Line other) {
double ym = (y0 + y1) / 2;
double yn = (other.y0 + other.y1) / 2;
return Math.abs(ym - yn) / Math.sqrt(1);
}
private void computeWords(double wordSpacing) {
TextPositionSequence word = new TextPositionSequence();
Character previous = null;
for (Character current : characters) {
if (previous != null) {
double dist = current.getTextPosition().getXDirAdj() - previous.getTextPosition().getXDirAdj() - previous.getTextPosition().getWidthDirAdj();
if (dist > wordSpacing) {
words.add(word);
word = new TextPositionSequence();
}
}
word.getTextPositions().add(current.getTextPosition());
previous = current;
}
words.add(word);
}
private void buildBox() {
double minX = Double.POSITIVE_INFINITY;
double minY = Double.POSITIVE_INFINITY;
double maxX = Double.NEGATIVE_INFINITY;
double maxY = Double.NEGATIVE_INFINITY;
for (Character character : characters) {
minX = Math.min(minX, character.getTextPosition().getXDirAdj());
minY = Math.min(minY, character.getTextPosition().getYDirAdj());
maxX = Math.max(maxX, character.getTextPosition().getXDirAdj() + character.getTextPosition().getWidthDirAdj());
maxY = Math.max(maxY, character.getTextPosition().getYDirAdj() + character.getTextPosition().getHeightDir());
}
this.setBBox(new Rectangle2D.Double(minX, minY, maxX - minX, maxY - minY));
}
public String toString() {
StringBuilder sb = new StringBuilder();
words.forEach(word -> sb.append(word.toString()).append(" "));
return sb.toString().trim();
}
}
@@ -0,0 +1,36 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import lombok.Getter;
public class Neighbor {
@Getter
private final double distance;
@Getter
private final double angle;
private final Character originCharacter;
@Getter
private final Character character;
public Neighbor(Character neighbor, Character origin) {
this.distance = neighbor.distance(origin);
this.angle = neighbor.angle(origin);
this.character = neighbor;
this.originCharacter = origin;
}
public double getHorizontalDistance() {
return character.horizontalDistance(originCharacter);
}
public double getVerticalDistance() {
return character.verticalDistance(originCharacter);
}
}
@@ -0,0 +1,50 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model;
import java.awt.geom.Rectangle2D;
import java.util.Comparator;
import java.util.List;
import lombok.Data;
@Data
public class Zone extends BoundingBox {
private List<Line> lines;
public Zone(List<Line> lines) {
lines.sort(Comparator.comparingDouble(Line::getY));
this.lines = lines;
buildBox();
}
public void buildBox() {
double minX = Double.POSITIVE_INFINITY;
double minY = Double.POSITIVE_INFINITY;
double maxX = Double.NEGATIVE_INFINITY;
double maxY = Double.NEGATIVE_INFINITY;
for (Line line : lines) {
minX = Math.min(minX, line.getX());
minY = Math.min(minY, line.getY());
maxX = Math.max(maxX, line.getX() + line.getWidth());
maxY = Math.max(maxY, line.getY() + line.getHeight());
}
this.setBBox(new Rectangle2D.Double(minX, minY, maxX - minX, maxY - minY));
}
public String toString() {
StringBuilder sb = new StringBuilder();
lines.forEach(line -> sb.append(line.toString()).append("\n"));
return sb.toString().trim();
}
}
@@ -0,0 +1,64 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder;
import java.awt.geom.Rectangle2D;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.BoundingBox;
public class BoundingBoxZoneGroup extends BoundingBox {
private BoundingBox leftChild;
private BoundingBox rightChild;
public BoundingBoxZoneGroup(BoundingBox child1, BoundingBox child2) {
this.leftChild = child1;
this.rightChild = child2;
setBounds(Math.min(child1.getX(), child2.getX()),
Math.min(child1.getY(), child2.getY()),
Math.max(child1.getX() + child1.getWidth(), child2.getX() + child2.getWidth()),
Math.max(child1.getY() + child1.getHeight(), child2.getY() + child2.getHeight()));
}
public void setbBox(Rectangle2D bBox) {
super.setBBox(bBox);
}
public BoundingBox getLeftChild() {
return leftChild;
}
public BoundingBox getRightChild() {
return rightChild;
}
public BoundingBoxZoneGroup setLeftChild(BoundingBox obj) {
this.leftChild = obj;
return this;
}
public BoundingBoxZoneGroup setRightChild(BoundingBox obj) {
this.rightChild = obj;
return this;
}
public BoundingBoxZoneGroup setBounds(double x0, double y0, double x1, double y1) {
assert x1 >= x0;
assert y1 >= y0;
this.setBBox(new Rectangle2D.Double(x0, y0, x1 - x0, y1 - y0));
return this;
}
}
@@ -0,0 +1,115 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.utils.DoubleUtils;
public class DistElem<E> implements Comparable<DistElem<E>> {
@Override
public int hashCode() {
final int prime = 31;
int result = 1;
result = prime * result + (c ? 1231 : 1237);
long temp;
temp = Double.doubleToLongBits(dist);
result = prime * result + (int) (temp ^ (temp >>> 32));
result = prime * result + ((obj1 == null) ? 0 : obj1.hashCode());
result = prime * result + ((obj2 == null) ? 0 : obj2.hashCode());
return result;
}
@Override
public boolean equals(Object obj) {
if (this == obj) {
return true;
}
if (obj == null) {
return false;
}
if (getClass() != obj.getClass()) {
return false;
}
DistElem other = (DistElem) obj;
if (c != other.c) {
return false;
}
if (Double.doubleToLongBits(dist) != Double.doubleToLongBits(other.dist)) {
return false;
}
if (obj1 == null) {
if (other.obj1 != null) {
return false;
}
} else if (!obj1.equals(other.obj1)) {
return false;
}
if (obj2 == null) {
if (other.obj2 != null) {
return false;
}
} else if (!obj2.equals(other.obj2)) {
return false;
}
return true;
}
boolean c;
double dist;
E obj1;
E obj2;
public boolean isC() {
return c;
}
public void setC(boolean c) {
this.c = c;
}
public double getDist() {
return dist;
}
public E getObj1() {
return obj1;
}
public E getObj2() {
return obj2;
}
public DistElem(boolean c, double dist, E obj1, E obj2) {
this.c = c;
this.dist = dist;
this.obj1 = obj1;
this.obj2 = obj2;
}
@Override
public int compareTo(DistElem<E> compareObject) {
double eps = 1E-3;
if (c == compareObject.c) {
return DoubleUtils.compareDouble(dist, compareObject.dist, eps);
} else {
return c ? -1 : 1;
}
}
}
@@ -0,0 +1,258 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.BoundingBox;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Zone;
/**
* A set-like data structure for objects placed on a plane. Can efficiently find objects in a certain rectangular area.
* It maintains two parallel lists of objects, each of which is sorted by its x or y coordinate.
*
* @author Pawel Szostek
*/
public class DocumentPlane {
/**
* List of objects on the plane. Stored in a random order
*/
private final List<BoundingBox> objs;
/**
* Size of a grid square. If gridSize=50, then the plane is divided into squares of size 50. Each square contains
* objects placed in a 50x50 area
*/
private final int gridSize;
/**
* Redundant dictionary of objects on the plane. Allows efficient 2D space search. Keys are X-Y coordinates of a
* grid square. Single object can be stored under several keys (depending on its physical size). Grid squares are
* lazy-initialized.
*/
private final Map<GridXY, List<BoundingBox>> grid;
/**
* Representation of XY coordinates
*/
private static class GridXY {
public int x;
public int y;
public GridXY(int x, int y) {
this.x = x;
this.y = y;
}
@Override
public int hashCode() {
return x * y;
}
@Override
public boolean equals(Object obj) {
if (obj == null || getClass() != obj.getClass()) {
return false;
}
GridXY comparedObj = (GridXY) obj;
return x == comparedObj.x && y == comparedObj.y;
}
@Override
public String toString() {
return "(" + x + "," + y + ")";
}
}
public List<BoundingBox> getObjects() {
return objs;
}
public DocumentPlane(List<Zone> objectList, int gridSize) {
this.grid = new HashMap<GridXY, List<BoundingBox>>();
this.objs = new ArrayList<BoundingBox>();
this.gridSize = gridSize;
for (Zone obj : objectList) {
add(obj);
}
}
/**
* Looks for objects placed between obj1 and obj2 excluding them
*
* @param obj1 object
* @param obj2 object
* @return object list
*/
public List<BoundingBox> findObjectsBetween(BoundingBox obj1, BoundingBox obj2) {
double x0 = Math.min(obj1.getX(), obj2.getX());
double y0 = Math.min(obj1.getY(), obj2.getY());
double x1 = Math.max(obj1.getX() + obj1.getWidth(), obj2.getX() + obj2.getWidth());
double y1 = Math.max(obj1.getY() + obj1.getHeight(), obj2.getY() + obj2.getHeight());
assert x1 >= x0 && y1 >= y0;
Rectangle2D searchBounds = new Rectangle2D.Double(x0, y0, x1 - x0, y1 - y0);
List<BoundingBox> objsBetween = find(searchBounds);
/*
* the rectangle area must contain at least obj1 and obj2
*/
objsBetween.remove(obj1);
objsBetween.remove(obj2);
return objsBetween;
}
/**
* Checks if there is any object placed between obj1 and obj2
*
* @param obj1 object
* @param obj2 object
* @return true if anything is placed between, false otherwise
*/
public boolean anyObjectsBetween(BoundingBox obj1, BoundingBox obj2) {
List<BoundingBox> lObjs = findObjectsBetween(obj1, obj2);
return !(lObjs.isEmpty());
}
/**
* Adds object to the plane
*
* @param obj object
* @return document plane
*/
public DocumentPlane add(BoundingBox obj) {
int objsBefore = this.objs.size();
/*
* iterate over grid squares
*/
for (int y = ((int) obj.getY()) / gridSize; y <= ((int) (obj.getY() + obj.getHeight() + gridSize - 1)) / gridSize; ++y) {
for (int x = ((int) obj.getX()) / gridSize; x <= ((int) (obj.getX() + obj.getWidth() + gridSize - 1)) / gridSize; ++x) {
GridXY xy = new GridXY(x, y);
if (!grid.keySet().contains(xy)) {
/*
* add the non-existing key
*/
grid.put(xy, new ArrayList<BoundingBox>());
grid.get(xy).add(obj);
assert grid.get(xy).size() == 1;
} else {
grid.get(xy).add(obj);
}
}
}
objs.add(obj);
/*
* size of the object list should be incremented
*/
assert objsBefore + 1 == objs.size();
/*
* object list must contain the same number of objects as object dictionary
*/
assert objs.size() == elementsInGrid();
return this;
}
public DocumentPlane remove(BoundingBox obj) {
/*
* iterate over grid squares
*/
for (int y = ((int) obj.getY()) / gridSize; y <= ((int) (obj.getY() + obj.getHeight() + gridSize - 1)) / gridSize; ++y) {
for (int x = ((int) obj.getX()) / gridSize; x <= ((int) (obj.getX() + obj.getWidth() + gridSize - 1)) / gridSize; ++x) {
GridXY xy = new GridXY(x, y);
if (grid.get(xy).contains(obj)) {
grid.get(xy).remove(obj);
}
}
}
objs.remove(obj);
assert objs.size() == elementsInGrid();
return this;
}
/**
* Find objects within search bounds
*
* @param searchBounds is a search rectangle
* @return list of objects in!side search rectangle
*/
public List<BoundingBox> find(Rectangle2D searchBounds) {
List<BoundingBox> done = new ArrayList<BoundingBox>(); //contains already considered objects (wrt. optimization)
List<BoundingBox> ret = new ArrayList<BoundingBox>();
double x0 = searchBounds.getX();
double y0 = searchBounds.getY();
double y1 = searchBounds.getY() + searchBounds.getHeight();
double x1 = searchBounds.getX() + searchBounds.getWidth();
/*
* iterate over grid squares
*/
for (int y = (int) y0 / gridSize; y < ((int) (y1 + gridSize - 1)) / gridSize; ++y) {
for (int x = (int) x0 / gridSize; x < ((int) (x1 + gridSize - 1)) / gridSize; ++x) {
GridXY xy = new GridXY(x, y);
if (!grid.containsKey(xy)) {
continue;
}
for (BoundingBox obj : grid.get(xy)) {
if (done.contains(obj)) /*
* omit if already checked
*/ {
continue;
}
/*
* add to the checked objects
*/
done.add(obj);
/*
* check if two objects overlap
*/
if (obj.getX() + obj.getWidth() <= x0 || x1 <= obj.getX() || obj.getY() + obj.getHeight() <= y0 || y1 <= obj.getY()) {
continue;
}
ret.add(obj);
}
}
}
return ret;
}
/**
* Count objects stored in objects dictionary
*
* @return number of elements
*/
protected int elementsInGrid() {
List<BoundingBox> objs_ = new ArrayList<BoundingBox>();
for (GridXY coord : grid.keySet()) {
for (BoundingBox obj : grid.get(coord)) {
if (!objs_.contains(obj)) {
objs_.add(obj);
}
}
}
return objs_.size();
}
}
@@ -0,0 +1,29 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder;
import java.util.ArrayList;
import java.util.List;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Zone;
public class TreeToListConverter {
public List<Zone> convertToList(BoundingBoxZoneGroup obj) {
List<Zone> ret = new ArrayList<>();
if (obj.getLeftChild() instanceof Zone) {
Zone zone = (Zone) obj.getLeftChild();
ret.add(zone);
} else { // obj.getLeftChild() instanceof BxZoneGroup
ret.addAll(convertToList((BoundingBoxZoneGroup) obj.getLeftChild()));
}
if (obj.getRightChild() instanceof Zone) {
Zone zone = (Zone) obj.getRightChild();
ret.add(zone);
} else { // obj.getRightChild() instanceof BxZoneGroup
ret.addAll(convertToList((BoundingBoxZoneGroup) obj.getRightChild()));
}
return ret;
}
}
@@ -0,0 +1,50 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.AngleFilter;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Character;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.DisjointSets;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Line;
@Service
public class LineBuilderService {
private static final double CHARACTER_SPACING_DISTANCE_MULTIPLIER = 3.5;
private static final double MAX_VERTICAL_CHARACTER_DISTANCE = 0.67;
private static final double ANGLE_TOLERANCE = Math.PI / 6;
public List<Line> buildLines(List<Character> characters, double characterSpacing, double lineSpacing) {
double maxHorizontalDistance = characterSpacing * CHARACTER_SPACING_DISTANCE_MULTIPLIER;
double maxVerticalDistance = lineSpacing * MAX_VERTICAL_CHARACTER_DISTANCE;
DisjointSets<Character> sets = new DisjointSets<>(characters);
AngleFilter filter = AngleFilter.newInstance(-ANGLE_TOLERANCE, ANGLE_TOLERANCE);
characters.forEach(character -> {
character.getNeighbors().forEach(neighbor -> {
double x = neighbor.getHorizontalDistance() / maxHorizontalDistance;
double y = neighbor.getVerticalDistance() / maxVerticalDistance;
if (filter.matches(neighbor) && Math.pow(x, 2) + Math.pow(y, 2) <= 1) {
sets.union(character, neighbor.getCharacter());
}
});
});
List<Line> lines = new ArrayList<>();
sets.forEach(group -> {
List<Character> lineComponents = new ArrayList<>(group);
lineComponents.sort(Comparator.comparingDouble(Character::getX));
lines.add(new Line(lineComponents, characterSpacing));
});
return lines;
}
}
@@ -0,0 +1,78 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Character;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Neighbor;
@Service
public class NearestNeighbourService {
private static final int NUMBER_OF_NEIGHBOURS = 8;
private static final double STEP = 16.0;
public void findNearestNeighbors(List<Character> characters) {
if (characters.isEmpty()) {
return;
}
characters.sort(Comparator.comparingDouble(Character::getX));
int maxNeighborCount = NUMBER_OF_NEIGHBOURS;
if (characters.size() <= NUMBER_OF_NEIGHBOURS) {
maxNeighborCount = characters.size() - 1;
}
for (int i = 0; i < characters.size(); i++) {
List<Neighbor> candidates = new ArrayList<>();
int start = i;
int end = i + 1;
double distance = Double.POSITIVE_INFINITY;
for (double searchDistance = 0; searchDistance < distance; ) {
searchDistance += STEP;
boolean newCandidatesFound = false;
while (start > 0 && characters.get(i).getX() - characters.get(start - 1).getX() < searchDistance) {
start--;
candidates.add(new Neighbor(characters.get(start), characters.get(i)));
clearLeastDistant(candidates, maxNeighborCount);
newCandidatesFound = true;
}
while (end < characters.size() && characters.get(end).getX() - characters.get(i).getX() < searchDistance) {
candidates.add(new Neighbor(characters.get(end), characters.get(i)));
clearLeastDistant(candidates, maxNeighborCount);
end++;
newCandidatesFound = true;
}
if (newCandidatesFound && candidates.size() >= maxNeighborCount) {
distance = candidates.get(maxNeighborCount - 1).getDistance();
}
}
clearLeastDistant(candidates, maxNeighborCount);
characters.get(i).setNeighbors(new ArrayList<>(candidates));
}
}
private void clearLeastDistant(List<Neighbor> candidates, int maxNeighborCount) {
if (candidates.size() > maxNeighborCount) {
candidates.sort(Comparator.comparingDouble(Neighbor::getDistance));
candidates.remove(candidates.remove(candidates.size() - 1));
}
}
}
@@ -0,0 +1,286 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service;
import java.util.ArrayList;
import java.util.Collection;
import java.util.Collections;
import java.util.Comparator;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.BoundingBox;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Zone;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder.BoundingBoxZoneGroup;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder.DistElem;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder.DocumentPlane;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.readingorder.TreeToListConverter;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.utils.DoubleUtils;
@Service
public class ReadingOrderService {
static final int GRIDSIZE = 50;
static final double EPS = 0.01;
static final int MAX_ZONES = 1000;
static final Comparator<BoundingBox> Y_ASCENDING_ORDER = new Comparator<BoundingBox>() {
@Override
public int compare(BoundingBox o1, BoundingBox o2) {
return DoubleUtils.compareDouble(o1.getY(), o2.getY(), EPS);
}
};
static final Comparator<BoundingBox> X_ASCENDING_ORDER = new Comparator<BoundingBox>() {
@Override
public int compare(BoundingBox o1, BoundingBox o2) {
return DoubleUtils.compareDouble(o1.getX(), o2.getX(), EPS);
}
};
static final Comparator<BoundingBox> YX_ASCENDING_ORDER = new Comparator<BoundingBox>() {
@Override
public int compare(BoundingBox o1, BoundingBox o2) {
int yCompare = Y_ASCENDING_ORDER.compare(o1, o2);
return yCompare == 0 ? X_ASCENDING_ORDER.compare(o1, o2) : yCompare;
}
};
public List<Zone> resolve(List<Zone> zones) {
List<Zone> orderedZones;
if (zones.size() > MAX_ZONES) {
orderedZones = new ArrayList<>(zones);
Collections.sort(orderedZones, YX_ASCENDING_ORDER);
} else {
orderedZones = reorderZones(zones);
}
return orderedZones;
}
private List<Zone> reorderZones(List<Zone> unorderedZones) {
if (unorderedZones.isEmpty()) {
return new ArrayList<>();
} else if (unorderedZones.size() == 1) {
List<Zone> ret = new ArrayList<>(1);
ret.add(unorderedZones.get(0));
return ret;
} else {
BoundingBoxZoneGroup bxZonesTree = groupZonesHierarchically(unorderedZones);
sortGroupedZones(bxZonesTree);
TreeToListConverter treeConverter = new TreeToListConverter();
List<Zone> orderedZones = treeConverter.convertToList(bxZonesTree);
assert unorderedZones.size() == orderedZones.size();
return orderedZones;
}
}
/**
* Builds a binary tree of zones and groups of zones from a list of unordered zones. This is done in hierarchical
* clustering by joining two least distant nodes. Distance is calculated in the distance() method.
*
* @param zones is a list of unordered zones
* @return root of the zones clustered in a tree
*/
private BoundingBoxZoneGroup groupZonesHierarchically(List<Zone> zones) {
/*
* Distance tuples are stored sorted by ascending distance value
*/
List<DistElem<BoundingBox>> dists = new ArrayList<DistElem<BoundingBox>>(zones.size() * zones.size() / 2);
for (int idx1 = 0; idx1 < zones.size(); ++idx1) {
for (int idx2 = idx1 + 1; idx2 < zones.size(); ++idx2) {
Zone zone1 = zones.get(idx1);
Zone zone2 = zones.get(idx2);
dists.add(new DistElem<BoundingBox>(false, distance(zone1, zone2), zone1, zone2));
}
}
Collections.sort(dists);
DocumentPlane plane = new DocumentPlane(zones, GRIDSIZE);
while (!dists.isEmpty()) {
DistElem<BoundingBox> distElem = dists.get(0);
dists.remove(0);
if (!distElem.isC() && plane.anyObjectsBetween(distElem.getObj1(), distElem.getObj2())) {
dists.add(new DistElem<BoundingBox>(true, distElem.getDist(), distElem.getObj1(), distElem.getObj2()));
continue;
}
BoundingBoxZoneGroup newGroup = new BoundingBoxZoneGroup(distElem.getObj1(), distElem.getObj2());
plane.remove(distElem.getObj1()).remove(distElem.getObj2());
dists = removeDistElementsContainingObject(dists, distElem.getObj1());
dists = removeDistElementsContainingObject(dists, distElem.getObj2());
for (BoundingBox other : plane.getObjects()) {
dists.add(new DistElem<BoundingBox>(false, distance(other, newGroup), newGroup, other));
}
Collections.sort(dists);
plane.add(newGroup);
}
assert plane.getObjects().size() == 1 : "There should be one object left at the plane after grouping";
return (BoundingBoxZoneGroup) plane.getObjects().get(0);
}
/**
* Removes all distance tuples containing obj
*/
private List<DistElem<BoundingBox>> removeDistElementsContainingObject(Collection<DistElem<BoundingBox>> list, BoundingBox obj) {
List<DistElem<BoundingBox>> ret = new ArrayList<DistElem<BoundingBox>>();
for (DistElem<BoundingBox> distElem : list) {
if (distElem.getObj1() != obj && distElem.getObj2() != obj) {
ret.add(distElem);
}
}
return ret;
}
/**
* Swaps children of BxZoneGroup if necessary. A group with smaller sort factor is placed to the left (leftChild).
* An object with greater sort factor is placed on the right (rightChild). This plays an important role when
* traversing the tree in conversion to a one dimensional list.
*
* @param group
*/
private void sortGroupedZones(BoundingBoxZoneGroup group) {
BoundingBox leftChild = group.getLeftChild();
BoundingBox rightChild = group.getRightChild();
if (shouldBeSwapped(leftChild, rightChild)) {
// swap
group.setLeftChild(rightChild);
group.setRightChild(leftChild);
}
if (leftChild instanceof BoundingBoxZoneGroup) // if the child is a tree node, then recurse
{
sortGroupedZones((BoundingBoxZoneGroup) leftChild);
}
if (rightChild instanceof BoundingBoxZoneGroup) // as above - recurse
{
sortGroupedZones((BoundingBoxZoneGroup) rightChild);
}
}
private boolean shouldBeSwapped(BoundingBox first, BoundingBox second) {
double cx, cy, cw, ch, ox, oy, ow, oh;
cx = first.getBBox().getX();
cy = first.getBBox().getY();
cw = first.getBBox().getWidth();
ch = first.getBBox().getHeight();
ox = second.getBBox().getX();
oy = second.getBBox().getY();
ow = second.getBBox().getWidth();
oh = second.getBBox().getHeight();
// Determine Octant
//
// 0 | 1 | 2
// __|___|__
// 7 | 9 | 3 First is placed in 9th square
// __|___|__
// 6 | 5 | 4
if (cx + cw <= ox) { //2,3,4
return false;
} else if (ox + ow <= cx) { //0,6,7
return true; //6
} else if (cy + ch <= oy) {
return false; //5
} else if (oy + oh <= cy) {
return true; //1
} else { //two zones
double xdiff = ox + ow / 2 - cx - cw / 2;
double ydiff = oy + oh / 2 - cy - ch / 2;
return xdiff + ydiff < 0;
}
}
/**
* A distance function between two TextBoxes.
* <p>
* Consider the bounding rectangle for obj1 and obj2. Return its area minus the areas of obj1 and obj2, shown as
* 'www' below. This value may be negative. (x0,y0) +------+..........+ | obj1 |wwwwwwwwww: +------+www+------+
* :wwwwwwwwww| obj2 | +..........+------+ (x1,y1)
*
* @return distance value based on objects' coordinates and physical size on a plane
*/
private double distance(BoundingBox obj1, BoundingBox obj2) {
double x0 = Math.min(obj1.getX(), obj2.getX());
double y0 = Math.min(obj1.getY(), obj2.getY());
double x1 = Math.max(obj1.getX() + obj1.getWidth(), obj2.getX() + obj2.getWidth());
double y1 = Math.max(obj1.getY() + obj1.getHeight(), obj2.getY() + obj2.getHeight());
double dist = ((x1 - x0) * (y1 - y0) - obj1.getArea() - obj2.getArea());
double factor = ((x1 - x0)/x1) / ((y1 - y0)/y1);
double obj1X = obj1.getX();
double obj1Y_2 = obj1.getBBox().getMaxY();
double obj1X_2 = obj1.getBBox().getMaxX();
double obj1CenterX = obj1.getBBox().getCenterX();
double obj1CenterY = obj1.getBBox().getCenterY();
double obj2X = obj2.getX();
double obj2Y_2 = obj2.getBBox().getMaxY();
double obj2X_2 = obj2.getBBox().getMaxX();
double obj2CenterX = obj2.getBBox().getCenterX();
double obj2CenterY = obj2.getBBox().getCenterY();
double obj1obj2VectorCosineAbsLeft = Math.abs((obj2X - obj1X) / Math.sqrt((obj2X - obj1X) * (obj2X - obj1X) + (obj2CenterY - obj1CenterY) * (obj2CenterY - obj1CenterY)));
double obj1obj2VectorCosineAbsRight = Math.abs((obj2X_2 - obj1X_2) / Math.sqrt((obj2X_2 - obj1X_2) * (obj2X_2 - obj1X_2) + (obj2CenterY - obj1CenterY) * (obj2CenterY - obj1CenterY)));
double obj1obj2VectorCosineAbsCenter = Math.abs((obj2CenterX - obj1CenterX) / Math.sqrt((obj2CenterX - obj1CenterX) * (obj2CenterX - obj1CenterX) + (obj2CenterY - obj1CenterY) * (obj2CenterY - obj1CenterY)));
double cosine = Math.min(obj1obj2VectorCosineAbsLeft, Math.min(obj1obj2VectorCosineAbsRight, obj1obj2VectorCosineAbsCenter));
final double MAGIC_COEFF = 0.85;
//return dist * (MAGIC_COEFF + cosine);
return Math.sqrt(Math.pow((obj1X - obj2X), 2) + Math.pow((obj1Y_2 - obj2Y_2) * MAGIC_COEFF, 2));
/**if (Math.abs(obj1CenterX - obj2CenterX) >= Math.abs(obj1CenterY - obj2CenterY)) {
return dist * 2;
} else {
return dist;
}**/
}
private double distanceNew(BoundingBox obj1, BoundingBox obj2) {
if(obj1.getBBox().intersects(obj2.getBBox()))
return -1;
double minX0 = Math.min(obj1.getX(), obj2.getX());
double maxX0 = Math.max(obj1.getX(), obj2.getX());
double minY0 = Math.min(obj1.getY(), obj2.getY());
double maxY0 = Math.max(obj1.getY(), obj2.getY());
double minX1 = Math.min(obj1.getX() + obj1.getWidth(), obj2.getX() + obj2.getWidth());
double maxX1 = Math.max(obj1.getX() + obj1.getWidth(), obj2.getX() + obj2.getWidth());
double minY1 = Math.min(obj1.getY() + obj1.getHeight(), obj2.getY() + obj2.getHeight());
double maxY1 = Math.max(obj1.getY() + obj1.getHeight(), obj2.getY() + obj2.getHeight());
List<Double> xValues = new ArrayList<>(List.of(minX0, maxX0, minX1, maxX1));
Collections.sort(xValues);
List<Double> yValues = new ArrayList<>(List.of(minY0, maxY0, minY1, maxY1));
Collections.sort(yValues);
double yArea = (xValues.get(2) - xValues.get(1)) * (yValues.get(3) - yValues.get(0));
double xArea = (yValues.get(2) - yValues.get(1)) * (xValues.get(3) - xValues.get(0));
return Math.min(10*yArea, xArea);
}
}
@@ -0,0 +1,56 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.AngleFilter;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Character;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Histogram;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Neighbor;
@Service
public class SpacingService {
private static final double SPACING_HISTOGRAM_RESOLUTION = 0.5;
private static final double SPACING_HISTOGRAM_SMOOTHING_LENGTH = 2.5;
private static final double SPACING_HIST_SMOOTHING_STANDARD_DEVIATION = 0.5;
private static final double ANGLE_TOLERANCE = Math.PI / 6;
public double computeCharacterSpacing(List<Character> components) {
return computeSpacing(components, 0);
}
public double computeLineSpacing(List<Character> components) {
return computeSpacing(components, Math.PI / 2);
}
private double computeSpacing(List<Character> components, double angle) {
double maxDistance = Double.NEGATIVE_INFINITY;
for (Character component : components) {
for (Neighbor neighbor : component.getNeighbors()) {
maxDistance = Math.max(maxDistance, neighbor.getDistance());
}
}
Histogram histogram = new Histogram(0, maxDistance, SPACING_HISTOGRAM_RESOLUTION);
AngleFilter filter = AngleFilter.newInstance(angle - ANGLE_TOLERANCE, angle + ANGLE_TOLERANCE);
for (Character component : components) {
for (Neighbor neighbor : component.getNeighbors()) {
if (filter.matches(neighbor)) {
histogram.add(neighbor.getDistance());
}
}
}
histogram.gaussianSmooth(SPACING_HISTOGRAM_SMOOTHING_LENGTH, SPACING_HIST_SMOOTHING_STANDARD_DEVIATION);
return histogram.getPeakValue();
}
}
@@ -0,0 +1,94 @@
package com.knecon.fforesight.service.layoutparser.processor.services.docstrum.service;
import java.util.ArrayList;
import java.util.List;
import org.springframework.stereotype.Service;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.DisjointSets;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Line;
import com.knecon.fforesight.service.layoutparser.processor.services.docstrum.model.Zone;
@Service
public class ZoneBuilderService {
private static final double MIN_HORIZONTAL_DISTANCE_MULTIPLIER = -0.5;
private static final double MAX_VERTICAL_DISTANCE_MULTIPLIER = 1.2;
private static final double MIN_HORIZONTAL_MERGE_DISTANCE_MULTIPLIER = -3.0;
private static final double MAX_VERTICAL_MERGE_DISTANCE_MULTIPLIER = 0.5;
private static final double MIN_LINE_SIZE_SCALE = 0.9;
private static final double MAX_LINE_SIZE_SCALE = 2.5;
private static final double ANGLE_TOLERANCE = Math.PI / 6;
public static final int MAX_ZONES = 300;
public List<Zone> buildZones(List<Line> lines, double characterSpacing, double lineSpacing) {
double minHorizontalDistance = characterSpacing * MIN_HORIZONTAL_DISTANCE_MULTIPLIER;
double maxVerticalDistance = lineSpacing * MAX_VERTICAL_DISTANCE_MULTIPLIER;
double minHorizontalMergeDistance = characterSpacing * MIN_HORIZONTAL_MERGE_DISTANCE_MULTIPLIER;
double maxVerticalMergeDistance = lineSpacing * MAX_VERTICAL_MERGE_DISTANCE_MULTIPLIER;
DisjointSets<Line> sets = new DisjointSets<>(lines);
double meanHeight = calculateMeanHeight(lines);
lines.forEach(outerLine -> //
lines.forEach(innerLine -> {
double scale = Math.min(outerLine.getHeight(), innerLine.getHeight()) / meanHeight;
scale = Math.max(MIN_LINE_SIZE_SCALE, Math.min(scale, MAX_LINE_SIZE_SCALE));
if (!sets.areTogether(outerLine, innerLine) && outerLine.angularDifference(innerLine) <= ANGLE_TOLERANCE) {
double horizontalDistance = outerLine.horizontalDistance(innerLine) / scale;
double verticalDistance = outerLine.verticalDistance(innerLine) / scale;
// Line over or above
if (minHorizontalDistance <= horizontalDistance && verticalDistance <= maxVerticalDistance) {
sets.union(outerLine, innerLine);
}
// Split line that needs later merging
else if (minHorizontalMergeDistance <= horizontalDistance && verticalDistance <= maxVerticalMergeDistance) {
sets.union(outerLine, innerLine);
}
}
}));
List<Zone> zones = new ArrayList<>();
sets.forEach(group -> {
zones.add(new Zone(new ArrayList<>(group)));
});
if (zones.size() > MAX_ZONES) {
List<Line> oneZoneLines = new ArrayList<>();
for (Zone zone : zones) {
oneZoneLines.addAll(zone.getLines());
}
return List.of(new Zone(oneZoneLines));
}
return zones;
}
private double calculateMeanHeight(List<Line> lines) {
double meanHeight = 0.0;
double weights = 0.0;
for (Line line : lines) {
double weight = line.getLength();
meanHeight += line.getHeight() * weight;
weights += weight;
}
meanHeight /= weights;
return meanHeight;
}
}

Some files were not shown because too many files have changed in this diff Show More