Compare commits

...
Author SHA1 Message Date
Kilian Schuettler d4e728350d RED-6093: Prototype document structure
*refactored File Structure
*started refactor of original rules
2023-03-07 14:21:21 +01:00
Kilian Schuettler b410067b8c RED-6093: Prototype document structure
*refactored Nodes
*added some tests
*wip
2023-03-02 18:45:59 +01:00
deiflaender 73d3ed625a RED-6093: Prototype document structurewip dependency upgrade 2023-02-27 12:29:29 +01:00
Kilian Schuettler e9176db88f RED-6093: Prototype document structure
wip
2023-02-27 12:00:51 +01:00
Kilian Schuettler ef72cc6861 RED-6093: Prototype find entities in rules
*added improved string to text position mapping
2023-02-17 17:06:00 +01:00
Kilian Schuettler 66b2d52d40 RED-6093: Prototype find entities in rules
*added a prototype paragraph rule
2023-02-15 16:51:16 +01:00
deiflaender dd1b838a5c RED-6093: Prototype find entities in rules 2023-02-13 15:49:20 +01:00
deiflaender 1fca62f578 RED-5664: Enabled to redact words that start or end with seperator, needed for japan documents 2023-02-13 10:33:22 +01:00
Timo Bejan 0e925f2f24 Pull request #512: RED-4609 - adjusted some metrics, added tests for metrics
Merge in RED/redaction-service from RED-4609 to master

* commit 'e23432096cb0ed486a294ed210646bbd8682350e':
  RED-4609 - adjusted some metrics, added tests for metrics
2023-02-09 09:54:19 +01:00
Timo Bejan e23432096c RED-4609 - adjusted some metrics, added tests for metrics 2023-02-08 19:12:47 +02:00
Dominique Eiflaender c1e2b8da29 Pull request #511: RED-5276: Fixed strange behavior of text parsing for tables example document
Merge in RED/redaction-service from RED-5276-1 to master

* commit '16b04b5918a3c8cac0070125d0e984bcd55b9b70':
  RED-5276: Fixed strange behavior of text parsing for tables example document
2023-01-31 11:16:07 +01:00
deiflaender 16b04b5918 RED-5276: Fixed strange behavior of text parsing for tables example document 2023-01-31 11:03:55 +01:00
Philipp Schramm c16b6d41d5 Pull request #510: RED-5248: Fix handling of temp files
Merge in RED/redaction-service from RED-5248 to master

* commit 'b839c4e3aea67c94fa29d476f63673fc232dcdab':
  RED-5248: Fix handling of temp files
2023-01-30 12:24:54 +01:00
Philipp Schramm b839c4e3ae RED-5248: Fix handling of temp files 2023-01-30 11:51:14 +01:00
Philipp Schramm 6fd6caa8ad Pull request #509: RED-5917: Wrong value set for Signature and Logo after Resize
Merge in RED/redaction-service from RED-5917 to master

* commit '41282c0edd1cbc98357ba04e33731d5b079064b8':
  RED-5917: Wrong value set for Signature and Logo after Resize
2023-01-19 13:00:30 +01:00
Philipp Schramm 41282c0edd RED-5917: Wrong value set for Signature and Logo after Resize 2023-01-19 11:25:26 +01:00
Timo Bejan 3c79e65345 Pull request #508: RED-5981 remove from dictionary pending analhysis
Merge in RED/redaction-service from RED-5981 to master

* commit 'd2eeaa91a6857b779996e5f5e9254bd4a460ac63':
  RED-5981 remove from dictionary pending analhysis
2023-01-15 09:15:54 +01:00
Timo Bejan d2eeaa91a6 RED-5981 remove from dictionary pending analhysis 2023-01-15 16:05:21 +08:00
Dominique Eiflaender 53a375b832 Pull request #507: RED-5276: Imporved table calculation, support spanned rows and colmns
Merge in RED/redaction-service from RED-5276-1 to master

* commit 'd233c18d335d17d3c79590dd3a295eaa89881de2':
  RED-5276: Imporved table calculation, support spanned rows and colmns
2022-12-23 11:25:43 +01:00
deiflaender d233c18d33 RED-5276: Imporved table calculation, support spanned rows and colmns 2022-12-23 11:20:05 +01:00
Philipp Schramm b74673ae63 Pull request #506: RED-5249: Marked utils classes with @UtilityClass
Merge in RED/redaction-service from RED-5249 to master

* commit 'faa702d3f499b89dfd9a7dedf8c16af89128c77d':
  RED-5249: Marked utils classes with @UtilityClass
2022-12-15 09:49:06 +01:00
Philipp Schramm faa702d3f4 RED-5249: Marked utils classes with @UtilityClass 2022-12-14 15:27:23 +01:00
Corina Olariu 05bf95d62b Pull request #505: RED-5748 - Redaction Service Error after helm upgrade
Merge in RED/redaction-service from RED-5748 to master

* commit '776de8392a78f7b26e864e37f4be607d1ff7e48e':
  RED-5748 - Redaction Service Error after helm upgrade - in case of deleted types throw a NotFoundException providing the info about the type not found instead of NullPointerException
2022-12-14 11:04:10 +01:00
devplant 776de8392a RED-5748 - Redaction Service Error after helm upgrade
- in case of deleted types throw a NotFoundException providing the info about the type not found instead of NullPointerException
2022-12-13 16:39:23 +02:00
Viktor Seifert 660abb318f Pull request #504: RED-5670: Update platform version to include security fixes from spring-boot
Merge in RED/redaction-service from RED-5670 to master

* commit '8b74142d319718cc448c9ddbbb40fa1b30c29b18':
  RED-5670: Update platform version to include security fixes from spring-boot
2022-11-25 16:24:54 +01:00
Viktor Seifert 8b74142d31 RED-5670: Update platform version to include security fixes from spring-boot 2022-11-25 16:14:34 +01:00
Corina Olariu 467a242a3d Pull request #503: RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED
Merge in RED/redaction-service from RED-5469 to master

* commit '87b11842a6606cd24155b8e9b0a82932267512bb':
  RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED - renamed from cvServiceEnabled to cvTableParsingEnabled
2022-11-18 11:25:10 +01:00
devplant 87b11842a6 RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED
- renamed from cvServiceEnabled to cvTableParsingEnabled
2022-11-17 14:23:06 +02:00
Timo Bejan 8f56d50322 Pull request #501: RED-5562 more symbols
Merge in RED/redaction-service from RED-5562-mst to master

* commit '38ce801f2d9e8d2cc83fea4e3ca56d73738f7bb1':
  RED-5562 more symbols
2022-11-15 11:10:01 +01:00
Timo Bejan 38ce801f2d RED-5562 more symbols 2022-11-15 11:29:17 +02:00
Timo Bejan 91e227248d Pull request #498: RED-5562 japanse space characters
Merge in RED/redaction-service from RED-5562-mst to master

* commit '623b8df5e6136d2ae6649e5016933abdc40454cc':
  RED-5562 improved pattern compile
  RED-5562 japanse space characters - fixed PMD
  RED-5562 japanse space characters
2022-11-15 10:05:33 +01:00
Timo Bejan 623b8df5e6 RED-5562 improved pattern compile 2022-11-15 01:38:33 +02:00
Timo Bejan 20ab65afd2 RED-5562 japanse space characters - fixed PMD 2022-11-15 00:57:54 +02:00
Timo Bejan 1ce47b7fbc RED-5562 japanse space characters 2022-11-14 23:31:31 +02:00
Dominique Eiflaender 5feb6891e2 Pull request #496: RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section
Merge in RED/redaction-service from RSS-164-2 to master

* commit '02bdbbc2d1f6bff0c907240f61726a78a0b8f318':
  RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section
2022-11-04 13:37:20 +01:00
deiflaender 02bdbbc2d1 RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section 2022-11-04 13:32:09 +01:00
Dominique Eiflaender 19e607e8a8 Pull request #495: RSS-177: Added required rule functions for scm poc
Merge in RED/redaction-service from RSS-177 to master

* commit '5c38150d34d8558f0e0e19ecd340d459a132426d':
  RSS-177: Added required rule functions for scm poc
2022-11-03 14:22:40 +01:00
deiflaender 5c38150d34 RSS-177: Added required rule functions for scm poc 2022-11-03 12:50:29 +01:00
Dominique Eiflaender 18487c639b Pull request #494: RSS-145: Added rules to add new FileAttributes
Merge in RED/redaction-service from RSS-145 to master

* commit 'e2234dc52a8a27a285c0e20b8326bd7129fc7dbf':
  RSS-145: Fixed Immutable list exception when merging existing and added fileattributes
  RSS-145: Added rules to add new FileAttributes
2022-10-28 13:07:20 +02:00
deiflaender e2234dc52a RSS-145: Fixed Immutable list exception when merging existing and added fileattributes 2022-10-28 12:55:52 +02:00
deiflaender 503df78f88 RSS-145: Added rules to add new FileAttributes 2022-10-28 12:33:45 +02:00
Philipp Schramm 5527eaec4e Pull request #492: RSS-118: Refactored method in Section for adding AI entries
Merge in RED/redaction-service from RSS-118 to master

* commit 'b4f079c3c2a5b26828a81b1c950d0f7e40c6314a':
  RSS-118: Refactored method in Section for adding AI entries
2022-10-27 14:20:49 +02:00
Philipp Schramm b4f079c3c2 RSS-118: Refactored method in Section for adding AI entries 2022-10-27 14:15:42 +02:00
Dominique Eiflaender 1e6e5e2154 Pull request #490: RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages
Merge in RED/redaction-service from RED-5381 to master

* commit '17bdcf8d2469429106234d9ccfdb9f2db17d9b84':
  RED-5381: Fixed pr findings
  RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages
2022-10-21 13:00:08 +02:00
deiflaender 17bdcf8d24 RED-5381: Fixed pr findings 2022-10-21 12:40:10 +02:00
deiflaender aa43453206 RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages 2022-10-21 11:42:39 +02:00
Philipp Schramm ddbf80e4a6 Pull request #486: RED-5232: Fixed NPE
Merge in RED/redaction-service from RED-5232 to master

* commit '2e3d4ad361ee9453ebcea3645014bce0fd4af22c':
  RED-5232: Fixed NPE
2022-10-17 15:55:30 +02:00
Philipp Schramm 2e3d4ad361 RED-5232: Fixed NPE 2022-10-17 15:51:21 +02:00
Ali Oezyetimoglu b8320bd000 Pull request #485: RED-5232: Code reformatting
Merge in RED/redaction-service from RED-5232 to master

* commit '76fda2b573d0fa1aaf73d4ccc7af9496ac257fa1':
  RED-5232: Code reformatting
2022-10-17 15:04:39 +02:00
Ali Oezyetimoglu 76fda2b573 RED-5232: Code reformatting 2022-10-17 14:58:26 +02:00
Dominique Eiflaender 69540bcd5e Pull request #482: RED-5295: Added redactWordPartByRegEx rule function
Merge in RED/redaction-service from RED-5295 to master

* commit 'e0dd06c6bf64bccff41f425cd934ebbe934384cf':
  RED-5295: Added redactWordPartByRegEx rule function
2022-10-12 10:49:11 +02:00
Dominique Eiflaender 8ab2738cd0 Pull request #484: RED-5275: Calculate precision and recall for headline detection
Merge in RED/redaction-service from RED-5275 to master

* commit '8d88b19915f0eaf83dc4174635585081f0468db2':
  RED-5275: Calculate precision and recall for headline detection
2022-10-12 10:48:56 +02:00
deiflaender 8d88b19915 RED-5275: Calculate precision and recall for headline detection 2022-10-11 11:05:19 +02:00
Philipp Schramm 9d88925ff1 Pull request #483: RED-5028: Integrated cv table service
Merge in RED/redaction-service from RED-5028 to master

* commit 'f6bc49d42c65a8580a5558891cabd4738af01d87':
  RED-5028: Integrated cv table service
2022-10-11 09:03:07 +02:00
Philipp Schramm f6bc49d42c RED-5028: Integrated cv table service 2022-10-11 08:07:55 +02:00
deiflaender e0dd06c6bf RED-5295: Added redactWordPartByRegEx rule function 2022-10-05 11:32:33 +02:00
Dominique Eiflaender 97209a3508 Pull request #481: RED-5275: Fixed not all headline were found because headline contains newlines
Merge in RED/redaction-service from RED-5275 to master

* commit '074205aa4d4532d5143463ab80199be08e1599a1':
  RED-5275: Fixed not all headline were found because headline contains newlines
2022-09-28 15:16:08 +02:00
deiflaender 074205aa4d RED-5275: Fixed not all headline were found because headline contains newlines 2022-09-28 15:12:49 +02:00
Dominique Eiflaender 2f88d8083c Pull request #480: RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule
Merge in RED/redaction-service from RED-5275 to master

* commit 'a510a8bb9f6796da4652b45fdefbf3c5f3f2ac92':
  RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule
2022-09-28 13:32:12 +02:00
deiflaender a510a8bb9f RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule 2022-09-28 13:25:03 +02:00
Dominique Eiflaender c027924b19 Pull request #478: RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco
Merge in RED/redaction-service from RED-4988 to master

* commit '313cf68044513f354d297162b1789ce9ce1f171c':
  RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco
2022-09-26 16:20:41 +02:00
deiflaender 313cf68044 RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco 2022-09-26 16:07:25 +02:00
Dominique Eiflaender 3c3666cd06 Pull request #476: RED-5139: Bugfix if false positives are added, for rule added entries
Merge in RED/redaction-service from RED-5139-d to master

* commit '5aba8fe88151ba6576e3ad17fd9f2ba78fc889f6':
  RED-5139: Bugfix if false positives are added, for rule added entries
2022-09-26 14:49:09 +02:00
deiflaender 5aba8fe881 RED-5139: Bugfix if false positives are added, for rule added entries 2022-09-26 14:39:09 +02:00
Dominique Eiflaender ec60d0eab4 Pull request #475: RED-5151: Do not retry messages on oom errors
Merge in RED/redaction-service from RED-5151 to master

* commit '0e71613ffcce147b2ae659e47ad6ab1c6405f0e7':
  RED-5151: Do not retry messages on oom errors
2022-09-26 14:01:22 +02:00
deiflaender 0e71613ffc RED-5151: Do not retry messages on oom errors 2022-09-26 12:52:39 +02:00
Dominique Eiflaender 142f5256ae Pull request #473: RSS-114: Do not skipOverride on same types
Merge in RED/redaction-service from RSS-114 to master

* commit 'ceee64ce59d4434f953ea220f9a733877ecde08d':
  RSS-114: Do not skipOverride on same types
2022-09-16 13:56:36 +02:00
deiflaender ceee64ce59 RSS-114: Do not skipOverride on same types 2022-09-16 10:57:54 +02:00
Dominique Eiflaender d2f2cd975c Pull request #472: RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes
Merge in RED/redaction-service from RSS-105 to master

* commit '5503dbfffc10ef9f51da7c37cb66e933d747a874':
  RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes
2022-09-15 13:08:42 +02:00
deiflaender 5503dbfffc RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes 2022-09-15 13:03:31 +02:00
Dominique Eiflaender 369710dbd4 Pull request #471: RSS-109: Added rule redactBetweenRegexes
Merge in RED/redaction-service from RSS-109 to master

* commit 'b9b3e8c8f5f11fbce6d0ae85f8e079e6285b70e0':
  RSS-109: Added rule redactBetweenRegexes
2022-09-15 11:47:22 +02:00
deiflaender b9b3e8c8f5 RSS-109: Added rule redactBetweenRegexes 2022-09-15 11:41:50 +02:00
Dominique Eiflaender c478935b86 Pull request #470: RSS-107: Added funktion to redact Headline
Merge in RED/redaction-service from RSS-107 to master

* commit 'd92757cda462d8af9ac852b88c867032c753a3b0':
  RSS-107: Added funktion to redact Headline
2022-09-13 12:12:35 +02:00
deiflaender d92757cda4 RSS-107: Added funktion to redact Headline 2022-09-13 12:09:09 +02:00
Dominique Eiflaender 5828e19422 Pull request #469: RSS-105: Added parameters to include/exclude start/stop in redactBetween rule
Merge in RED/redaction-service from RSS-105 to master

* commit 'b7bf84c323b1e730481dad16428db01227d3347e':
  RSS-105: Fixed checkstyle error
  RSS-105: Added parameters to include/exclude start/stop in redactBetween rule
2022-09-13 11:49:42 +02:00
deiflaender b7bf84c323 RSS-105: Fixed checkstyle error 2022-09-13 11:44:28 +02:00
deiflaender fe7e8a83df RSS-105: Added parameters to include/exclude start/stop in redactBetween rule 2022-09-13 11:35:36 +02:00
Dominique Eiflaender a71a7818c2 Pull request #468: RSS-106: Fixed not working redactions over multi pages
Merge in RED/redaction-service from RSS-106 to master

* commit 'ee5da81299425452171370ab609a0172858c4787':
  RSS-106: Fixed not working redactions over multi pages
2022-09-13 08:57:27 +02:00
deiflaender ee5da81299 RSS-106: Fixed not working redactions over multi pages 2022-09-12 17:40:16 +02:00
Dominique Eiflaender e486711755 Pull request #467: RSS-92: Fixed excludeHeadlines on tables
Merge in RED/redaction-service from RSS-92 to master

* commit '58c3d15b78e6ddf310b7288b4b703477c0a1218d':
  RSS-92: Fixed excludeHeadlines on tables
2022-09-12 13:45:19 +02:00
deiflaender 58c3d15b78 RSS-92: Fixed excludeHeadlines on tables 2022-09-12 12:58:15 +02:00
Christoph Schabert ce15faf072 Pull request #466: hotfix: remove push from nightly build
Merge in RED/redaction-service from fixNightly to master

* commit '430a08e611cb6ac99a31139f1a46bf36a66bb6ab':
  hotfix: remove push from nightly build
2022-09-12 11:29:26 +02:00
cschabert 430a08e611 hotfix: remove push from nightly build 2022-09-12 11:26:14 +02:00
Dominique Eiflaender 4cc40d8381 Pull request #465: RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions
Merge in RED/redaction-service from RSS-31 to master

* commit '317a8a9af9d9755befe66d82eef5f7b21fe49b33':
  RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions
2022-09-09 14:07:50 +02:00
deiflaender 317a8a9af9 RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions 2022-09-09 13:25:53 +02:00
Dominique Eiflaender 28e437a037 Pull request #464: RSS-86: Added new rule function redactLineAfterAcrossColumns
Merge in RED/redaction-service from RSS-86 to master

* commit 'a5f27cfa4cf81bfe1cb29016d18d3fc40e77624d':
  RSS-86: Added new rule function redactLineAfterAcrossColumns
2022-09-08 13:47:45 +02:00
deiflaender a5f27cfa4c RSS-86: Added new rule function redactLineAfterAcrossColumns 2022-09-08 13:05:13 +02:00
Philipp Schramm 34be42cd45 Pull request #463: RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag
Merge in RED/redaction-service from RSS-2 to master

* commit 'f1b5d605ccbe4984943e7452dd701004f64f7525':
  RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag
2022-09-08 12:17:20 +02:00
Philipp Schramm f1b5d605cc RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag 2022-09-08 12:10:07 +02:00
Dominique Eiflaender c3dadd6906 Pull request #462: RSS-42: Made redactBetween rule more generic
Merge in RED/redaction-service from RSS-42 to master

* commit 'c29c5eef0d05a86a18fca8978a42ffff3c42ea16':
  RSS-42: Made redactBetween rule more generic
2022-09-07 14:37:26 +02:00
deiflaender c29c5eef0d RSS-42: Made redactBetween rule more generic 2022-09-07 14:31:08 +02:00
Philipp Schramm 9867cd6848 Pull request #461: RED-5002: Fixed return http error code for testing rules
Merge in RED/redaction-service from red-5002 to master

* commit '64bd25a9000286c2c30400f1da02cc2ff54ea946':
  RED-5002: Fixed return http error code for testing rules
2022-08-29 09:43:46 +02:00
Philipp Schramm 64bd25a900 RED-5002: Fixed return http error code for testing rules 2022-08-29 09:38:13 +02:00
Ali Oezyetimoglu 9b9b0ab271 Pull request #459: RED-4178: Remove special behavior for tables with 2 columns
Merge in RED/redaction-service from RED-4178-rs1 to master

* commit '29451db72d7cf0b22acb27fd86d3198dab2534ca':
  RED-4178: Remove special behavior for tables with 2 columns
2022-08-26 16:50:26 +02:00
Ali Oezyetimoglu da21d7da4b Pull request #460: RED-5032: Upgrade services to new base image
Merge in RED/redaction-service from RED-5032-rs1 to master

* commit 'b6471f904c14e02a7f42cab0aa1dc07bb1aa503a':
  RED-5032: Upgrade services to new base image
2022-08-26 16:20:58 +02:00
Ali Oezyetimoglu b6471f904c RED-5032: Upgrade services to new base image 2022-08-25 17:01:51 +02:00
Ali Oezyetimoglu 29451db72d RED-4178: Remove special behavior for tables with 2 columns 2022-08-25 11:13:14 +02:00
Dominique Eiflaender 87b76cdae0 Pull request #457: RED-5022: Add figure detection values the same way as normal images
Merge in RED/redaction-service from RED-5022 to master

* commit '85cad66ade28fd482beee57b4b01951ec2b93dfd':
  RED-5022: Add figure detection values the same way as normal images
2022-08-19 14:38:27 +02:00
deiflaender 85cad66ade RED-5022: Add figure detection values the same way as normal images 2022-08-19 13:50:57 +02:00
Viktor Seifert 1cc93d3a57 Pull request #456: RED-4824: Added annotations to correctly handle json deserialization
Merge in RED/redaction-service from RED-5010 to master

* commit 'a1ef711bc32c365b2f276df5de7f3578f82f567f':
  RED-4824: Corrected annotation for json deserialization so that is unambiguously detected as a delegating creator method
  RED-4824: Added annotations to correctly handle json deserialization
2022-08-18 09:32:27 +02:00
Viktor Seifert a1ef711bc3 RED-4824: Corrected annotation for json deserialization so that is unambiguously detected as a delegating creator method 2022-08-17 19:12:28 +02:00
Viktor Seifert b6a95244d8 RED-4824: Added annotations to correctly handle json deserialization 2022-08-17 18:46:13 +02:00
Ali Oezyetimoglu 4c06ab958c Pull request #455: RED-4890
Merge in RED/redaction-service from RED-4890 to master

* commit 'f5fb8f7c074b3e7b38c6cfdd28ba4c8cc1edfeff':
  RED-4890: Refactored redaction log tests to show failed files
  RED-4890: Refactored RulesTests, will fail individually
2022-08-16 11:16:09 +02:00
Ali Oezyetimoglu f5fb8f7c07 RED-4890: Refactored redaction log tests to show failed files 2022-08-16 10:56:00 +02:00
Ali Oezyetimoglu e962992b79 Pull request #454: RED-4891: Rectangle redactions not listed in DELTA view
Merge in RED/redaction-service from RED-4891-rs1 to master

* commit 'a9ae01ab32f08d0dadb1b12b0cc4db8070071b42':
  RED-4891: Rectangle redactions not listed in DELTA view
2022-08-16 09:32:29 +02:00
Ali Oezyetimoglu a9ae01ab32 RED-4891: Rectangle redactions not listed in DELTA view 2022-08-15 13:49:03 +02:00
Dominique Eiflaender e77180f6ac Pull request #453: RED-3974: Regenerated redactionLog files
Merge in RED/redaction-service from RED-3974 to master

* commit '269716125c19f8a4029a39566332d20d5cc53a91':
  RED-3974: Regenerated redactionLog files
2022-08-12 12:29:02 +02:00
deiflaender 269716125c RED-3974: Regenerated redactionLog files 2022-08-12 12:09:46 +02:00
Dominique Eiflaender dc21e14921 Pull request #452: RED-3974: Do not add header to header row
Merge in RED/redaction-service from RED-3974 to master

* commit '6bc5a6a1351f39bd1815b4fa288faf7bce82b1eb':
  RED-3974: Do not add header to header row
2022-08-12 11:38:58 +02:00
deiflaender 6bc5a6a135 RED-3974: Do not add header to header row 2022-08-12 11:29:12 +02:00
Dominique Eiflaender 4f66f5acf7 Pull request #451: RED-3974: Fixed special case table header with diffent size of columns in rows
Merge in RED/redaction-service from RED-3974 to master

* commit 'a9ff46a3fd3363b47799e8dfe5dc99575c36a33b':
  RED-3974: Fixed special case table header with diffent size of columns in rows
2022-08-11 17:12:07 +02:00
deiflaender a9ff46a3fd RED-3974: Fixed special case table header with diffent size of columns in rows 2022-08-11 17:09:25 +02:00
Dominique Eiflaender 07aaa9722a Pull request #450: RED-3974: Use first row as header if header detection does not find a header
Merge in RED/redaction-service from RED-3974 to master

* commit 'cba81ce061df867936314dcbdbc565248c8db006':
  RED-3974: Refactored processTablePerRow
  RED-3974: Use first row as header if header detection does not find a header
2022-08-11 14:28:33 +02:00
deiflaender cba81ce061 RED-3974: Refactored processTablePerRow 2022-08-11 14:18:36 +02:00
deiflaender f84a366328 RED-3974: Use first row as header if header detection does not find a header 2022-08-11 11:54:25 +02:00
Viktor Seifert 85c44374d9 Pull request #449: RED-4824
Merge in RED/redaction-service from RED-4824 to master

* commit '1d39c150c7e9084ae6a3833149ad66a4f31da774':
  RED-4824: Corrected code style by removing parameter assignment
  RED-4824: Replaced type of the text dir with an enum that only accepts values that can be handled by our code
  RED-4824: Recreated log files to include changes in rectangle calculation
  RED-4824: Removed obsolete logging statement
  RED-4824: Finalized affine transformation by correcting the center of the rotation according to the text direction
  RED-4824: WIP: Transformed rectangle calculation from a if-else chain to an affine transformation
  RED-4824: Added test with a simple pdf that contains all combinations for rotation and text direction
  RED-4824: Cleaned up code to make method more readable
  RED-4824: Created base test for text-position rectangle creation
2022-08-11 11:09:51 +02:00
Viktor Seifert 1d39c150c7 RED-4824: Corrected code style by removing parameter assignment 2022-08-10 15:49:27 +02:00
Viktor Seifert b2f1201d92 RED-4824: Replaced type of the text dir with an enum that only accepts values that can be handled by our code 2022-08-10 15:40:20 +02:00
Viktor Seifert b592fee500 RED-4824: Recreated log files to include changes in rectangle calculation 2022-08-09 18:58:44 +02:00
Viktor Seifert 4f36b8b43e RED-4824: Removed obsolete logging statement 2022-08-09 17:12:51 +02:00
Viktor Seifert c7a789ada6 Merge branch 'master' into RED-4824 2022-08-09 16:08:19 +02:00
Viktor Seifert 6de3a3b043 RED-4824: Finalized affine transformation by correcting the center of the rotation according to the text direction 2022-08-09 15:12:10 +02:00
Viktor Seifert 43ff331a42 RED-4824: WIP: Transformed rectangle calculation from a if-else chain to an affine transformation 2022-08-08 19:31:55 +02:00
Timo Bejan 3013dc95f6 Pull request #448: RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive
Merge in RED/redaction-service from RED-4835 to master

* commit '6e46cad2c5466136c7191042e6834eb074fc5911':
  RED-4835 - cleanup
  RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive
2022-08-08 16:39:49 +02:00
Viktor Seifert 435f75996f RED-4824: Added test with a simple pdf that contains all combinations for rotation and text direction 2022-08-08 15:16:38 +02:00
Viktor Seifert ef04b7168a RED-4824: Cleaned up code to make method more readable 2022-08-08 14:05:25 +02:00
Timo Bejan 6e46cad2c5 RED-4835 - cleanup 2022-08-08 13:31:22 +03:00
Timo Bejan c41e230f85 RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive 2022-08-08 13:30:18 +03:00
Viktor Seifert cadbc3c7a4 RED-4824: Created base test for text-position rectangle creation 2022-08-05 16:56:26 +02:00
Philipp Schramm f4655c4050 RED-4890: Refactored RulesTests, will fail individually 2022-08-04 13:18:05 +02:00
Ali Oezyetimoglu 16ea8364df Pull request #447: RED-4867: Sonar issues: RedactionService
Merge in RED/redaction-service from RED-4867-rs1 to master

* commit '3ec1e20e417e65907448930cf7f8eb9713874fb9':
  RED-4867: Sonar issues: RedactionService
2022-08-01 16:00:29 +02:00
Ali Oezyetimoglu 3ec1e20e41 RED-4867: Sonar issues: RedactionService 2022-08-01 15:55:22 +02:00
Dominique Eiflaender ce1f8e9117 Pull request #434: RED-4753: Reduced size of TEXT File
Merge in RED/redaction-service from RED-4753 to master

* commit '55bebe8541cfded8a456b63dbd8d8b7e79bab33f':
  RED-4753: Reduced size of TEXT File a bit more
  RED-4753: Reduced size of TEXT File
2022-08-01 11:04:25 +02:00
Kresnadi Budisantoso b7e97890be Pull request #441: RED-4829 Fix wrong annotation
Merge in RED/redaction-service from kbudisantoso/Sectionjava-1659012500262 to master

* commit '47de526b4bd451d302317647cb05ac685e069622':
  RED-4829 Fix wrong annotation
2022-08-01 09:07:11 +02:00
Dominique Eiflaender ec49945682 Pull request #446: RED-4843: Disable jsondsl per default
Merge in RED/redaction-service from RED-4843 to master

* commit 'b64222938c64eeb006d62807a705918ffd2c585b':
  RED-4843: Disable jsondsl per default
2022-07-29 12:47:38 +02:00
deiflaender b64222938c RED-4843: Disable jsondsl per default 2022-07-29 12:43:35 +02:00
Timo Bejan e4823e20d4 Pull request #444: RED-4799 - remove FP entities from redactionlog
Merge in RED/redaction-service from RED-4799-3 to master

* commit '53428ec65296b9fac2c206e7bd89916c933fb5c8':
  RED-4799 - remove FP entities from redactionlog
2022-07-29 10:53:51 +02:00
Timo Bejan 53428ec652 RED-4799 - remove FP entities from redactionlog 2022-07-29 11:50:38 +03:00
Dominique Eiflaender ca270fb7de Pull request #443: RED-4843: Upgraded storage commons
Merge in RED/redaction-service from RED-4843 to master

* commit '06fe0dbd97cdf2011be17df07f8c409a221530a4':
  RED-4843: Upgraded storage commons
2022-07-29 10:41:24 +02:00
deiflaender 06fe0dbd97 RED-4843: Upgraded storage commons 2022-07-29 10:27:18 +02:00
Kresnadi Budisantoso 47de526b4b RED-4829 Fix wrong annotation 2022-07-28 14:48:39 +02:00
Timo Bejan 1d8e86e4f6 Pull request #439: RED-4799
Merge in RED/redaction-service from RED-4799-fix to master

* commit '2b7972315f655bd296a03cae8cc5e4846d47bbce':
  RED-4799
2022-07-27 11:58:11 +02:00
Timo Bejan 2b7972315f RED-4799 2022-07-27 12:54:51 +03:00
Philipp Schramm 0af853dbfc Pull request #436: RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo
Merge in RED/redaction-service from RED-4510-Rework to master

* commit '0544a42117a742de70325cd1a84e57d6829aaa25':
  RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo
2022-07-26 17:08:59 +02:00
Philipp Schramm 0544a42117 RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo 2022-07-26 17:06:04 +02:00
deiflaender 55bebe8541 RED-4753: Reduced size of TEXT File a bit more 2022-07-26 11:55:25 +02:00
Timo Bejan 8bdfd6747d Pull request #435: RED-4686
Merge in RED/redaction-service from tbejan/pomxml-1658825710723 to master

* commit '80c0c25c069995790d6bd043657f50dc8de0a451':
  RED-4686
2022-07-26 11:01:22 +02:00
Timo Bejan 80c0c25c06 RED-4686 2022-07-26 10:55:17 +02:00
deiflaender 1cc66a3092 RED-4753: Reduced size of TEXT File 2022-07-26 09:45:46 +02:00
Timo Bejan 17baf9c0eb Pull request #433: RED-4686 - updated cyclic deps
Merge in RED/redaction-service from RED-4686 to master

* commit '090a6196000dfcd7a447c54224fba8eda4605380':
  RED-4686 - updated cyclic deps
2022-07-26 09:25:58 +02:00
Timo Bejan 090a619600 RED-4686 - updated cyclic deps 2022-07-26 10:22:50 +03:00
Timo Bejan 9c8e2f047e Pull request #432: RED-4686 - storage commons update
Merge in RED/redaction-service from RED-4686 to master

* commit '24612e347171f27eb8747772b82d13c98f40344b':
  RED-4686 - storage commons update - added annotations and test-file
  RED-4686 - storage commons update
2022-07-25 21:19:43 +02:00
Timo Bejan 24612e3471 RED-4686 - storage commons update - added annotations and test-file 2022-07-25 22:08:36 +03:00
Timo Bejan 9be8d46f4e RED-4686 - storage commons update 2022-07-25 21:45:58 +03:00
Philipp Schramm 3e32540616 Pull request #431: RED-4771: Reformatted code
Merge in RED/redaction-service from RED-4771 to master

* commit '5403829fd874c04375bef39535c22f52f7a55853':
  RED-4771: Reformatted code
2022-07-25 16:11:37 +02:00
Philipp Schramm 5403829fd8 Merge branch 'master' into RED-4771
# Conflicts:
#	bamboo-specs/src/main/java/buildjob/PlanSpec.java
2022-07-25 16:08:23 +02:00
Philipp Schramm 3e53e156a5 RED-4771: Reformatted code 2022-07-25 16:04:51 +02:00
Philipp Schramm 4c73d2841b Pull request #430: RED-4510: Added bamboo build to execute tests nightly and generate RedactionLog for tests if it doesn't exist
Merge in RED/redaction-service from RED-4510 to master

* commit 'db8eedc9a36dfba22946940b31e28854ffdca26d':
  RED-4510: Added bamboo build to execute tests nightly and generate RedactionLog for tests if it doesn't exist
2022-07-25 14:52:53 +02:00
Philipp Schramm db8eedc9a3 RED-4510: Added bamboo build to execute tests nightly and generate RedactionLog for tests if it doesn't exist 2022-07-25 14:14:44 +02:00
Dominique Eiflaender 58e16fe39a Pull request #429: RED-4686: Removed test scope from jackson-datatype-jsr310
Merge in RED/redaction-service from RED-4686-2 to master

* commit '0fe2b627f071f6d8e4841cf08e3bcdc059b60567':
  RED-4686: Removed test scope from jackson-datatype-jsr310
2022-07-22 15:10:33 +02:00
deiflaender 0fe2b627f0 RED-4686: Removed test scope from jackson-datatype-jsr310 2022-07-22 15:06:57 +02:00
Dominique Eiflaender e4b5bb93e1 Pull request #426: RED-4686: Stream objects to storage
Merge in RED/redaction-service from RED-4686 to master

* commit '99a8b9788e1b1062c8d8a93d4b5d4b65f3d34ce2':
  RED-4686: Stream objects to storage
2022-07-22 13:54:23 +02:00
deiflaender 99a8b9788e RED-4686: Stream objects to storage 2022-07-22 13:29:22 +02:00
Philipp Schramm 822e1f096f Pull request #425: RED-4510
Merge in RED/redaction-service from RED-4510 to master

* commit '8c5ede3fde0fd1b2df9e8fca0908e9233d6d403c':
  RED-4510: Refactoring after review
  RED-4510: Implemented Test class which generates and compares RedactionLogs
2022-07-21 17:01:18 +02:00
Philipp Schramm 8c5ede3fde RED-4510: Refactoring after review 2022-07-21 16:58:36 +02:00
Philipp Schramm 40380a12a3 Merge branch 'master' into RED-4510 2022-07-21 12:38:27 +02:00
Philipp Schramm f48bdb6267 RED-4510: Implemented Test class which generates and compares RedactionLogs 2022-07-21 12:38:07 +02:00
Dominique Eiflaender d3a5265b7f Pull request #424: RED-4648: Fixed wrong field names for DSLJson
Merge in RED/redaction-service from RED-4648 to master

* commit '708b7f992cadba8779ac6d2f7a9e9e4a837eaf1b':
  RED-4648: Fixed wrong field names for DSLJson
2022-07-21 12:33:42 +02:00
deiflaender 708b7f992c RED-4648: Fixed wrong field names for DSLJson 2022-07-21 12:22:19 +02:00
Ali Oezyetimoglu 2b0a667424 Pull request #422: RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service)
Merge in RED/redaction-service from RED-4548-rs2 to master

* commit 'c330de86246b8cc10955401d307d7d1a20190e32':
  RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service)
2022-07-14 17:08:03 +02:00
Ali Oezyetimoglu c330de8624 RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service) 2022-07-14 16:57:09 +02:00
Timo Bejan 32e73dc9d9 Pull request #420: RED-4593
Merge in RED/redaction-service from tbejan/pomxml-1657693418623 to master

* commit '5e61f3370dcec8206b757dd59a1a7664921c8300':
  RED-4593
2022-07-13 08:32:03 +02:00
Timo Bejan 5e61f3370d RED-4593 2022-07-13 08:23:59 +02:00
Philipp Schramm bfadac7a3f Pull request #419: RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service)
Merge in RED/redaction-service from RED-4548 to master

* commit 'df80ff68cd21626679c107b056b275b67ce6bf74':
  RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service)
2022-07-12 16:41:06 +02:00
Philipp Schramm df80ff68cd RED-4548: Create SIMPLIFIED_TEXT.json from TEXT.json (for NER-service) 2022-07-12 16:38:25 +02:00
Timo Bejan 13352e7d16 Pull request #418: RED-4525 - Colors update
Merge in RED/redaction-service from RED-4525 to master

* commit '3c9d59f73ba2ca2b7500703b57df6d78b558d6a2':
  RED-4525 - Colors update
2022-07-11 20:58:27 +02:00
Timo Bejan 3c9d59f73b RED-4525 - Colors update 2022-07-11 21:55:29 +03:00
Dominique Eiflaender 333793021a Pull request #417: RED-4254: Added dirty hack in pdfbox classes to find words that contains uniqueCharacters with 2 chars like 'RA'
Merge in RED/redaction-service from RED-4254 to master

* commit '264d7e3a87d4a99aa60a4c7c140335ba8d3a7241':
  RED-4254: Added dirty hack in pdfbox classes to find words that contains uniqueCharacters with 2 chars like 'RA'
2022-06-28 15:30:02 +02:00
deiflaender 264d7e3a87 RED-4254: Added dirty hack in pdfbox classes to find words that contains uniqueCharacters with 2 chars like 'RA' 2022-06-28 15:11:33 +02:00
Dominique Eiflaender 6b89572c87 Pull request #416: RED-4407: Fixed merging request for dossierDictionarys in redactionLog
Merge in RED/redaction-service from RED-4407 to master

* commit 'a74d049cf8a52395771f9940471defd692cae0e8':
  RED-4407: Fixed merging request for dossierDictionarys in redactionLog
2022-06-28 11:55:12 +02:00
deiflaender a74d049cf8 RED-4407: Fixed merging request for dossierDictionarys in redactionLog 2022-06-28 11:52:06 +02:00
Ali Oezyetimoglu 76dc08d0e6 Pull request #415: RED-4350: Reason set to null after reject suggestion "Remove from Dict" for skipped types
Merge in RED/redaction-service from RED-4350-rs1 to master

* commit '663c361707678370149ed08cb6c33c8ac341dc26':
  RED-4350: Reason set to null after reject suggestion "Remove from Dict" for skipped types
2022-06-28 09:01:05 +02:00
aoezyetimoglu 663c361707 RED-4350: Reason set to null after reject suggestion "Remove from Dict" for skipped types 2022-06-27 17:43:28 +02:00
Dominique Eiflaender c81b062784 Pull request #413: RED-4301: Revert redaction log merge of duplicate annotations
Merge in RED/redaction-service from RED-4301 to master

* commit '57ef7ead72018a91d9c15530efce40f713fb6c86':
  RED-4301: Revert redaction log merge of duplicate annotations
2022-06-20 12:50:53 +02:00
deiflaender 57ef7ead72 RED-4301: Revert redaction log merge of duplicate annotations 2022-06-20 12:38:57 +02:00
Dominique Eiflaender 3269470446 Pull request #412: RED-4129: Fixed imported redactions annotations intersection calculation
Merge in RED/redaction-service from RED-4129 to master

* commit 'c7e10cefbacf8ca54c86ecba5a8598b7c9d55ce4':
  RED-4129: Fixed imported redactions annotations intersection calculation
2022-06-20 11:59:58 +02:00
deiflaender c7e10cefba RED-4129: Fixed imported redactions annotations intersection calculation 2022-06-20 11:57:10 +02:00
Timo Bejan 2301849815 Pull request #405: RED-3791 - redaction log merge of duplicate annotations
Merge in RED/redaction-service from RED-3791 to master

* commit 'ec8d3183dc119e3eddd45a653ca88a2666073f51':
  RED-3791 - redaction log merge of duplicate annotations
2022-06-14 18:25:35 +02:00
Dominique Eiflaender 2d5ba8f447 Pull request #411: RED-4129: Fixed calculation of imported redaction intersections for all rotation cases
Merge in RED/redaction-service from RED-4129 to master

* commit '2a6ee02f9f8f527546f01cf98542dfc7e39b259a':
  RED-4129: Fixed calculation of imported redaction intersections for all rotation cases
2022-06-14 12:15:01 +02:00
Timo Bejan ec8d3183dc RED-3791 - redaction log merge of duplicate annotations 2022-06-09 09:26:00 +03:00
413 changed files with 3928907 additions and 4179 deletions
+1
View File
@@ -1,4 +1,5 @@
# Changelog
All notable changes to this project will be documented in this file.
## [Unreleased]
+28 -28
View File
@@ -1,37 +1,37 @@
<project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/maven-v4_0_0.xsd">
<modelVersion>4.0.0</modelVersion>
<modelVersion>4.0.0</modelVersion>
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>8.1.3</version>
<relativePath/>
</parent>
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>8.1.3</version>
<relativePath/>
</parent>
<artifactId>bamboo-specs</artifactId>
<version>1.0.0-SNAPSHOT</version>
<packaging>jar</packaging>
<artifactId>bamboo-specs</artifactId>
<version>1.0.0-SNAPSHOT</version>
<packaging>jar</packaging>
<dependencies>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-api</artifactId>
</dependency>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs</artifactId>
</dependency>
<dependencies>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-api</artifactId>
</dependency>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs</artifactId>
</dependency>
<!-- Test dependencies -->
<dependency>
<groupId>junit</groupId>
<artifactId>junit</artifactId>
<scope>test</scope>
</dependency>
</dependencies>
<!-- Test dependencies -->
<dependency>
<groupId>junit</groupId>
<artifactId>junit</artifactId>
<scope>test</scope>
</dependency>
</dependencies>
<!-- run 'mvn test' to perform offline validation of the plan -->
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
<!-- run 'mvn test' to perform offline validation of the plan -->
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
</project>
+79 -100
View File
@@ -1,10 +1,12 @@
package buildjob;
import java.time.DayOfWeek;
import static com.atlassian.bamboo.specs.builders.task.TestParserTask.createJUnitParserTask;
import java.time.LocalTime;
import com.atlassian.bamboo.specs.api.BambooSpec;
import com.atlassian.bamboo.specs.api.builders.BambooKey;
import com.atlassian.bamboo.specs.api.builders.Variable;
import com.atlassian.bamboo.specs.api.builders.docker.DockerConfiguration;
import com.atlassian.bamboo.specs.api.builders.permission.PermissionType;
import com.atlassian.bamboo.specs.api.builders.permission.Permissions;
@@ -17,20 +19,16 @@ import com.atlassian.bamboo.specs.api.builders.plan.branches.BranchCleanup;
import com.atlassian.bamboo.specs.api.builders.plan.branches.PlanBranchManagement;
import com.atlassian.bamboo.specs.api.builders.project.Project;
import com.atlassian.bamboo.specs.builders.task.CheckoutItem;
import com.atlassian.bamboo.specs.api.builders.Variable;
import com.atlassian.bamboo.specs.builders.task.CleanWorkingDirectoryTask;
import com.atlassian.bamboo.specs.builders.task.InjectVariablesTask;
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
import com.atlassian.bamboo.specs.builders.task.VcsCheckoutTask;
import com.atlassian.bamboo.specs.builders.task.VcsTagTask;
import com.atlassian.bamboo.specs.builders.trigger.BitbucketServerTrigger;
import com.atlassian.bamboo.specs.builders.trigger.RepositoryPollingTrigger;
import com.atlassian.bamboo.specs.builders.trigger.ScheduledTrigger;
import com.atlassian.bamboo.specs.model.task.InjectVariablesScope;
import com.atlassian.bamboo.specs.util.BambooServer;
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
import com.atlassian.bamboo.specs.model.task.ScriptTaskProperties.Location;
import static com.atlassian.bamboo.specs.builders.task.TestParserTask.createJUnitParserTask;
import com.atlassian.bamboo.specs.util.BambooServer;
/**
* Plan configuration for Bamboo.
@@ -41,14 +39,15 @@ public class PlanSpec {
private static final String SERVICE_NAME = "redaction-service";
private static final String JVM_ARGS =" -Xmx4g -XX:+ExitOnOutOfMemoryError -XX:SurvivorRatio=2 -XX:NewRatio=1 -XX:InitialTenuringThreshold=16 -XX:MaxTenuringThreshold=16 -XX:InitiatingHeapOccupancyPercent=35 ";
private static final String JVM_ARGS = " -Xmx4g -XX:+ExitOnOutOfMemoryError -XX:SurvivorRatio=2 -XX:NewRatio=1 -XX:InitialTenuringThreshold=16 -XX:MaxTenuringThreshold=16 -XX:InitiatingHeapOccupancyPercent=35 ";
private static final String SERVICE_KEY = SERVICE_NAME.toUpperCase().replaceAll("-", "");
/**
* Run main to publish plan on Bamboo
*/
public static void main(final String[] args) throws Exception {
public static void main(final String[] args) {
//By default credentials are read from the '.credentials' file.
BambooServer bambooServer = new BambooServer("http://localhost:8085");
@@ -57,15 +56,26 @@ public class PlanSpec {
PlanPermissions planPermission = new PlanSpec().createPlanPermission(plan.getIdentifier());
bambooServer.publish(planPermission);
Plan nightPlan = new PlanSpec().createNightPlan();
bambooServer.publish(nightPlan);
PlanPermissions nightPlanPermission = new PlanSpec().createPlanPermission(nightPlan.getIdentifier());
bambooServer.publish(nightPlanPermission);
Plan secPlan = new PlanSpec().createSecBuild();
bambooServer.publish(secPlan);
PlanPermissions secPlanPermission = new PlanSpec().createPlanPermission(secPlan.getIdentifier());
bambooServer.publish(secPlanPermission);
}
private PlanPermissions createPlanPermission(PlanIdentifier planIdentifier) {
Permissions permission = new Permissions()
.userPermissions("atlbamboo", PermissionType.EDIT, PermissionType.VIEW, PermissionType.ADMIN, PermissionType.CLONE, PermissionType.BUILD)
Permissions permission = new Permissions().userPermissions("atlbamboo",
PermissionType.EDIT,
PermissionType.VIEW,
PermissionType.ADMIN,
PermissionType.CLONE,
PermissionType.BUILD)
.groupPermissions("development", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("devplant", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.loggedInUserPermissions(PermissionType.VIEW)
@@ -73,105 +83,74 @@ public class PlanSpec {
return new PlanPermissions(planIdentifier.getProjectKey(), planIdentifier.getPlanKey()).permissions(permission);
}
private Project project() {
return new Project()
.name("RED")
.key(new BambooKey("RED"));
return new Project().name("RED").key(new BambooKey("RED"));
}
public Plan createPlan() {
return new Plan(
project(),
SERVICE_NAME, new BambooKey(SERVICE_KEY))
.description("Plan created from (enter repository url of your plan)")
.variables(new Variable("maven_add_param", ""))
.stages(new Stage("Default Stage")
.jobs(new Job("Default Job",
new BambooKey("JOB1"))
.tasks(
new ScriptTask()
.description("Clean")
.inlineBody("#!/bin/bash\n" +
"set -e\n" +
"rm -rf ./*"),
new VcsCheckoutTask()
.description("Checkout Default Repository")
.cleanCheckout(true)
.checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask()
.description("Build")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
.argument(SERVICE_NAME),
createJUnitParserTask()
.description("Resultparser")
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
.enabled(true),
new InjectVariablesTask()
.description("Inject git Tag")
.path("git.tag")
.namespace("g")
.scope(InjectVariablesScope.LOCAL),
new VcsTagTask()
.description("${bamboo.g.gitTag}")
.tagName("${bamboo.g.gitTag}")
.defaultRepository())
.dockerConfiguration(
new DockerConfiguration()
.image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock")
)
)
)
.linkedRepositories("RED / " + SERVICE_NAME)
return new Plan(project(), SERVICE_NAME, new BambooKey(SERVICE_KEY)).description("Plan created from (enter repository url of your plan)")
.variables(new Variable("maven_add_param", ""))
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new CleanWorkingDirectoryTask().description("Clean working directory.")
.enabled(true),
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask().description("Build").location(Location.FILE).fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh").argument(SERVICE_NAME),
createJUnitParserTask().description("Resultparser")
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
.enabled(true),
new InjectVariablesTask().description("Inject git Tag").path("git.tag").namespace("g").scope(InjectVariablesScope.LOCAL),
new VcsTagTask().description("${bamboo.g.gitTag}").tagName("${bamboo.g.gitTag}").defaultRepository())
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(new BitbucketServerTrigger())
.planBranchManagement(new PlanBranchManagement()
.createForVcsBranch()
.delete(new BranchCleanup()
.whenInactiveInRepositoryAfterDays(14))
.planBranchManagement(new PlanBranchManagement().createForVcsBranch()
.delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14))
.notificationForCommitters());
}
public Plan createNightPlan() {
return new Plan(project(), SERVICE_NAME + "-Night", new BambooKey(SERVICE_KEY + "NIGHT")).description("Long running nightly Plan for tests")
.variables(new Variable("maven_add_param", "-Dtest-groups=rules-test"))
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new CleanWorkingDirectoryTask().description("Clean working directory.")
.enabled(true),
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask().description("Build")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
.argument(SERVICE_NAME + " verify"),
createJUnitParserTask().description("Resultparser")
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
.enabled(true))
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
.planBranchManagement(new PlanBranchManagement().delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14)).notificationForCommitters());
}
public Plan createSecBuild() {
return new Plan(
project(),
SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC"))
.description("Security Analysis Plan")
.stages(new Stage("Default Stage")
.jobs(new Job("Default Job",
new BambooKey("JOB1"))
.tasks(
new ScriptTask()
.description("Clean")
.inlineBody("#!/bin/bash\n" +
"set -e\n" +
"rm -rf ./*"),
new VcsCheckoutTask()
.description("Checkout Default Repository")
.cleanCheckout(true)
.checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask()
.description("Sonar")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh")
.argument(SERVICE_NAME))
.dockerConfiguration(
new DockerConfiguration()
.image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.dockerRunArguments("--net=host")
.volume("/etc/maven/settings.xml", "/usr/share/maven/conf/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock")
)
)
)
return new Plan(project(), SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC")).description("Security Analysis Plan")
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new ScriptTask().description("Clean")
.inlineBody("#!/bin/bash\n" + "set -e\n" + "rm -rf ./*"),
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask().description("Sonar").location(Location.FILE).fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh").argument(SERVICE_NAME))
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.dockerRunArguments("--net=host")
.volume("/etc/maven/settings.xml", "/usr/share/maven/conf/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(
new ScheduledTrigger()
.scheduleOnceDaily(LocalTime.of(23, 00)))
.planBranchManagement(new PlanBranchManagement()
.createForVcsBranchMatching("release.*")
.notificationForCommitters());
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
.planBranchManagement(new PlanBranchManagement().createForVcsBranchMatching("release.*").notificationForCommitters());
}
}
@@ -2,6 +2,7 @@
set -e
SERVICE_NAME=$1
MVN_TARGET=${2:-deploy}
if [[ "$bamboo_planRepository_branchName" == "master" ]]
then
@@ -46,7 +47,7 @@ mvn --no-transfer-progress \
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
--no-transfer-progress \
clean deploy \
clean $MVN_TARGET \
${bamboo_maven_add_param} \
-e \
-DdeployAtEnd=true \
@@ -1,6 +1,5 @@
package buildjob;
import org.junit.Test;
import com.atlassian.bamboo.specs.api.builders.plan.Plan;
@@ -8,12 +7,18 @@ import com.atlassian.bamboo.specs.api.exceptions.PropertiesValidationException;
import com.atlassian.bamboo.specs.api.util.EntityPropertiesBuilders;
public class PlanSpecTest {
@Test
public void checkYourPlanOffline() throws PropertiesValidationException {
Plan plan = new PlanSpec().createPlan();
EntityPropertiesBuilders.build(plan);
Plan nightPlan = new PlanSpec().createNightPlan();
EntityPropertiesBuilders.build(nightPlan);
Plan secPlan = new PlanSpec().createSecBuild();
EntityPropertiesBuilders.build(secPlan);
}
}
+1 -1
View File
@@ -6,7 +6,7 @@
<groupId>com.iqser.red</groupId>
<artifactId>platform-docker-dependency</artifactId>
<version>1.2.0</version>
<relativePath />
<relativePath/>
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -1,4 +1,4 @@
FROM red/redaction-service-base-v1:1.0.0
FROM red/redaction-service-base-v1:2.0.0
ARG PLATFORM_JAR
+10 -9
View File
@@ -5,8 +5,8 @@
<parent>
<artifactId>platform-dependency</artifactId>
<groupId>com.iqser.red</groupId>
<version>1.10.0</version>
<relativePath />
<version>1.17.0</version>
<relativePath/>
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -29,15 +29,10 @@
<dependencyManagement>
<dependencies>
<dependency>
<groupId>com.dslplatform</groupId>
<artifactId>dsl-json-java8</artifactId>
<version>${dsljson.version}</version>
</dependency>
<dependency>
<groupId>com.iqser.red</groupId>
<artifactId>platform-commons-dependency</artifactId>
<version>1.13.0</version>
<version>1.21.0</version>
<scope>import</scope>
<type>pom</type>
</dependency>
@@ -60,7 +55,7 @@
<plugin>
<groupId>org.sonarsource.scanner.maven</groupId>
<artifactId>sonar-maven-plugin</artifactId>
</plugin>
</plugin>
<plugin>
<groupId>org.owasp</groupId>
<artifactId>dependency-check-maven</artifactId>
@@ -71,6 +66,12 @@
<plugin>
<groupId>org.jacoco</groupId>
<artifactId>jacoco-maven-plugin</artifactId>
<version>0.8.8</version>
<configuration>
<excludes>
<exclude>org/drools/**/*</exclude>
</excludes>
</configuration>
<executions>
<execution>
<id>prepare-agent</id>
@@ -12,7 +12,7 @@
<artifactId>redaction-service-api-v1</artifactId>
<properties>
<persistence-service.version>1.166.0</persistence-service.version>
<persistence-service.version>1.299.0</persistence-service.version>
</properties>
<dependencies>
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -1,6 +1,9 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.Set;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -31,6 +34,7 @@ public class AnalyzeResult {
private ManualRedactions manualRedactions;
private Set<FileAttribute> addedFileAttributes;
}
@@ -2,6 +2,14 @@ package com.iqser.red.service.redaction.v1.model;
public enum ArgumentType {
INTEGER, BOOLEAN, STRING, FILE_ATTRIBUTE, REGEX, TYPE, RULE_NUMBER, LEGAL_BASIS, REFERENCE_TYPE
INTEGER,
BOOLEAN,
STRING,
FILE_ATTRIBUTE,
REGEX,
TYPE,
RULE_NUMBER,
LEGAL_BASIS,
REFERENCE_TYPE
}
@@ -16,4 +16,5 @@ public class Change {
private int analysisNumber;
private ChangeType type;
private OffsetDateTime dateTime;
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
public enum ChangeType {
ADDED, REMOVED, CHANGED
ADDED,
REMOVED,
CHANGED
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
public enum Engine {
DICTIONARY, NER, RULE
DICTIONARY,
NER,
RULE
}
@@ -18,4 +18,5 @@ public class ImportedRedaction {
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -11,10 +13,12 @@ import java.util.Map;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class ImportedRedactions {
@Builder.Default
private Map<Integer, List<ImportedRedaction>> importedRedactions = new HashMap<>();
}
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.BaseAnnotation;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -24,7 +25,9 @@ public class ManualChange {
private String userId;
private Map<String, String> propertyChanges = new HashMap<>();
public static ManualChange from(BaseAnnotation baseAnnotation) {
ManualChange manualChange = new ManualChange();
manualChange.annotationStatus = baseAnnotation.getStatus();
manualChange.processedDate = baseAnnotation.getProcessedDate();
@@ -33,16 +36,22 @@ public class ManualChange {
return manualChange;
}
public boolean isProcessed() {
return processedDate != null;
}
public ManualChange withManualRedactionType(ManualRedactionType manualRedactionType) {
this.manualRedactionType = manualRedactionType;
return this;
}
public ManualChange withChange(String property, String value) {
this.propertyChanges.put(property, value);
return this;
}
@@ -2,6 +2,9 @@ package com.iqser.red.service.redaction.v1.model;
public enum MessageType {
ANALYSE, REANALYSE, STRUCTURE_ANALYSE, SURROUNDING_TEXT
ANALYSE,
REANALYSE,
STRUCTURE_ANALYSE,
SURROUNDING_TEXT
}
@@ -12,4 +12,5 @@ import lombok.NoArgsConstructor;
public class ReanalyzeResult {
private RedactionLog redactionLog;
}
@@ -14,4 +14,5 @@ public class Rectangle {
private float height;
private int page;
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -8,14 +9,12 @@ import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@CompiledJson
@AllArgsConstructor
@NoArgsConstructor
public class RedactionLog {
/**
* Version 0 Redaction Logs have manual redactions merged inside them
* Version 1 Redaction Logs only contain system ( rule/dictionary ) redactions. Manual Redactions are merged in at runtime.
@@ -35,5 +34,4 @@ public class RedactionLog {
private long rulesVersion = -1;
private long legalBasisVersion = -1;
}
@@ -1,11 +1,11 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import lombok.*;
import java.util.*;
@Data
@Builder
@NoArgsConstructor
@@ -35,7 +35,6 @@ public class RedactionLogEntry {
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private String textBefore;
private String textAfter;
@@ -70,21 +69,27 @@ public class RedactionLogEntry {
@Builder.Default
private Set<String> importedRedactionIntersections = new HashSet<>();
public boolean lastChangeIsRemoved() {
return last(changes).map(c -> c.getType() == ChangeType.REMOVED).orElse(false);
}
public boolean isLocalManualRedaction() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
public boolean isManuallyRemoved() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
private <T> Optional<T> last(List<T> list) {
return list.isEmpty() ? Optional.empty() : Optional.of(list.get(list.size() - 1));
}
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.Type;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -29,4 +30,5 @@ public class RedactionRequest {
private List<Type> types;
private boolean includeFalsePositives;
}
@@ -15,13 +15,20 @@ public class SectionArea {
private int page;
private String header;
public boolean contains(Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft().getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft().getY() + other.getHeight();
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft()
.getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft()
.getY() + other.getHeight();
}
// TODO we should only use one rectangle class.
public boolean contains(com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft()
.getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
}
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -28,4 +29,5 @@ public class SectionGrid {
private List<SectionArea> sectionAreas;
}
}
@@ -4,6 +4,7 @@ import com.iqser.red.service.persistence.service.v1.api.model.annotations.Manual
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
@@ -32,8 +33,6 @@ public interface RedactionResource {
@PostMapping(value = "/manual/surrounding-text/{dossierId}/{fileId}", consumes = MediaType.APPLICATION_JSON_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions);
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions);
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
@@ -12,8 +12,8 @@
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<drools.version>7.68.0.Final</drools.version>
<kie.version>7.68.0.Final</kie.version>
<drools.version>7.73.0.Final</drools.version>
<kie.version>7.73.0.Final</kie.version>
<locationtech.version>1.18.2</locationtech.version>
<javaassist.version>3.28.0-GA</javaassist.version>
<ahocorasick.version>0.6.3</ahocorasick.version>
@@ -38,6 +38,12 @@
<version>${jackson.version}</version>
</dependency>
<dependency>
<groupId>com.fasterxml.jackson.datatype</groupId>
<artifactId>jackson-datatype-jsr310</artifactId>
<version>${jackson.version}</version>
</dependency>
<dependency>
<groupId>org.ahocorasick</groupId>
<artifactId>ahocorasick</artifactId>
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server;
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
@@ -22,14 +23,16 @@ import io.micrometer.core.instrument.MeterRegistry;
public class Application {
public static void main(String[] args) {
System.setProperty("org.apache.pdfbox.rendering.UsePureJavaCMYKConversion", "true");
SpringApplication.run(Application.class, args);
}
@Bean
public TimedAspect timedAspect(MeterRegistry registry) {
return new TimedAspect(registry);
}
}
@@ -14,7 +14,7 @@ import lombok.NoArgsConstructor;
public class Document {
private List<Page> pages = new ArrayList<>();
private List<Paragraph> paragraphs = new ArrayList<>();
private List<Section> sections = new ArrayList<>();
private List<Header> headers = new ArrayList<>();
private List<Footer> footers = new ArrayList<>();
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
@@ -14,7 +14,9 @@ public class FloatFrequencyCounter {
@Getter
Map<Float, Integer> countPerValue = new HashMap<>();
public void add(float value) {
if (!countPerValue.containsKey(value)) {
countPerValue.put(value, 1);
} else {
@@ -22,7 +24,9 @@ public class FloatFrequencyCounter {
}
}
public void addAll(Map<Float, Integer> otherCounter) {
for (Map.Entry<Float, Integer> entry : otherCounter.entrySet()) {
if (countPerValue.containsKey(entry.getKey())) {
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
@@ -32,12 +36,12 @@ public class FloatFrequencyCounter {
}
}
public Float getMostPopular() {
Map.Entry<Float, Integer> mostPopular = null;
for (Map.Entry<Float, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
mostPopular = entry;
} else if (entry.getValue() >= mostPopular.getValue()) {
if (mostPopular == null || entry.getValue() >= mostPopular.getValue()) {
mostPopular = entry;
}
}
@@ -46,6 +50,7 @@ public class FloatFrequencyCounter {
public List<Float> getHighterThanMostPopular() {
Float mostPopular = getMostPopular();
List<Float> higher = new ArrayList<>();
for (Float value : countPerValue.keySet()) {
@@ -59,11 +64,10 @@ public class FloatFrequencyCounter {
public Float getHighest() {
Float highest = null;
for (Float value : countPerValue.keySet()) {
if (highest == null) {
highest = value;
} else if (value > highest) {
if (highest == null || value > highest) {
highest = value;
}
}
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@@ -14,6 +15,7 @@ public class Footer {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@@ -14,6 +15,7 @@ public class Header {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
@@ -2,5 +2,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
public enum Orientation {
NONE, LEFT, RIGHT
NONE,
LEFT,
RIGHT
}
@@ -1,15 +1,18 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.Data;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
import java.util.ArrayList;
import java.util.List;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import lombok.Data;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
@Data
@RequiredArgsConstructor
public class Page {
@@ -31,7 +34,8 @@ public class Page {
private StringFrequencyCounter fontCounter = new StringFrequencyCounter();
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private double cropBoxArea;
private float pageWidth;
private float pageHeight;
public boolean isRotated() {
@@ -1,18 +1,26 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entities;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Collectors;
@Data
@NoArgsConstructor
public class Paragraph implements Comparable {
public class Section implements Comparable {
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
private List<PdfImage> images = new ArrayList<>();
@@ -61,4 +69,9 @@ public class Paragraph implements Comparable {
return 0;
}
}
@@ -1,19 +1,26 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.dslplatform.json.CompiledJson;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import com.iqser.red.service.redaction.v1.server.redaction.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.*;
@Data
@Builder
@CompiledJson
@@ -26,20 +33,27 @@ public class SectionText {
private boolean isTable;
private String headline;
List<Paragraph> paragraphs;
@Builder.Default
private List<SectionArea> sectionAreas = new ArrayList<>();
@Builder.Default
private Set<Image> images = new HashSet<>();
@Builder.Default
private List<TextBlock> textBlocks = new ArrayList<>();
@Builder.Default
private Map<String, CellValue> tabularData = new HashMap<>();
@Builder.Default
private List<Integer> cellStarts = new ArrayList<>();
public void setTabularData(Map<String, CellValue> tabularData) {
tabularData.remove(null);
this.tabularData = tabularData;
}
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
@@ -0,0 +1,20 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedSectionText {
private int sectionNumber;
private String text;
}
@@ -0,0 +1,23 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.ArrayList;
import java.util.List;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedText {
private int numberOfPages;
private List<SimplifiedSectionText> sectionTexts = new ArrayList<>();
}
@@ -37,9 +37,7 @@ public class StringFrequencyCounter {
Map.Entry<String, Integer> mostPopular = null;
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
mostPopular = entry;
} else if (entry.getValue() > mostPopular.getValue()) {
if (mostPopular == null || entry.getValue() > mostPopular.getValue()) {
mostPopular = entry;
}
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -1,19 +1,21 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.ArrayList;
import java.util.List;
import com.dslplatform.json.CompiledJson;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextDirection;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@AllArgsConstructor
@Builder
@Data
@@ -23,20 +25,169 @@ public class TextBlock extends AbstractTextContainer {
@Builder.Default
private List<TextPositionSequence> sequences = new ArrayList<>();
@JsonIgnore
private int rotation;
private int indexOnPage;
@JsonIgnore
private String mostPopularWordFont;
@JsonIgnore
private String mostPopularWordStyle;
@JsonIgnore
private float mostPopularWordFontSize;
@JsonIgnore
private float mostPopularWordHeight;
@JsonIgnore
private float mostPopularWordSpaceWidth;
@JsonIgnore
private float highestFontSize;
@JsonIgnore
private String classification;
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation) {
@JsonIgnore
@JsonAttribute(ignore = true)
public TextDirection getDir() {
return sequences.get(0).getDir();
}
@JsonIgnore
@JsonAttribute(ignore = true)
private float getPageHeight() {
return sequences.get(0).getPageHeight();
}
@JsonIgnore
@JsonAttribute(ignore = true)
private float getPageWidth() {
return sequences.get(0).getPageWidth();
}
/**
* Returns the minX value in pdf coordinate system.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
*
* @return the minX value in pdf coordinate system
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getPdfMinX() {
if (getDir().getDegrees() == 90) {
return minY;
} else if (getDir().getDegrees() == 180) {
return getPageWidth() - maxX;
} else if (getDir().getDegrees() == 270) {
return getPageWidth() - maxY;
} else {
return minX;
}
}
/**
* Returns the maxX value in pdf coordinate system.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
*
* @return the maxX value in pdf coordinate system
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getPdfMaxX() {
if (getDir().getDegrees() == 90) {
return maxY;
} else if (getDir().getDegrees() == 180) {
return getPageWidth() - minX;
} else if (getDir().getDegrees() == 270) {
return getPageWidth() - minY;
} else {
return maxX;
}
}
/**
* Returns the minY value in pdf coordinate system.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
*
* @return the minY value in pdf coordinate system
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getPdfMinY() {
if (getDir().getDegrees() == 90) {
return minX;
} else if (getDir().getDegrees() == 180) {
return maxY;
} else if (getDir().getDegrees() == 270) {
return getPageHeight() - maxX;
} else {
return getPageHeight() - maxY;
}
}
/**
* Returns the maxY value in pdf coordinate system.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
*
* @return the maxY value in pdf coordinate system
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getPdfMaxY() {
if (getDir().getDegrees() == 90) {
return maxX;
} else if (getDir().getDegrees() == 180) {
return minY;
} else if (getDir().getDegrees() == 270) {
return getPageHeight() - minX;
} else {
return getPageHeight() - minY;
}
}
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation, int indexOnPage) {
this.indexOnPage = indexOnPage;
this.minX = minX;
this.maxX = maxX;
this.minY = minY;
@@ -45,19 +196,25 @@ public class TextBlock extends AbstractTextContainer {
this.rotation = rotation;
}
public TextBlock union(TextPositionSequence r) {
TextBlock union = this.copy();
union.add(r);
return union;
}
public TextBlock union(TextBlock r) {
TextBlock union = this.copy();
union.add(r);
return union;
}
public void add(TextBlock r) {
if (r.getMinX() < minX) {
minX = r.getMinX();
}
@@ -73,30 +230,38 @@ public class TextBlock extends AbstractTextContainer {
sequences.addAll(r.getSequences());
}
public void add(TextPositionSequence r) {
if (r.getX1() < minX) {
minX = r.getX1();
if (r.getMinXDirAdj() < minX) {
minX = r.getMinXDirAdj();
}
if (r.getX2() > maxX) {
maxX = r.getX2();
if (r.getMaxXDirAdj() > maxX) {
maxX = r.getMaxXDirAdj();
}
if (r.getY1() < minY) {
minY = r.getY1();
if (r.getMinYDirAdj() < minY) {
minY = r.getMinYDirAdj();
}
if (r.getY2() > maxY) {
maxY = r.getY2();
if (r.getMaxYDirAdj() > maxY) {
maxY = r.getMaxYDirAdj();
}
}
public TextBlock copy() {
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation);
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation, indexOnPage);
}
public void resize(float x1, float y1, float width, float height) {
set(x1, y1, x1 + width, y1 + height);
}
public void set(float x1, float y1, float x2, float y2) {
this.minX = Math.min(x1, x2);
this.maxX = Math.max(x1, x2);
this.minY = Math.min(y1, y2);
@@ -122,6 +287,7 @@ public class TextBlock extends AbstractTextContainer {
}
@Override
@JsonIgnore
@JsonAttribute(ignore = true)
@@ -132,7 +298,7 @@ public class TextBlock extends AbstractTextContainer {
TextPositionSequence previous = null;
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
if (Math.abs(previous.getMaxYDirAdj() - word.getMaxYDirAdj()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@@ -14,6 +15,7 @@ public class UnclassifiedText {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
@@ -2,25 +2,22 @@ package com.iqser.red.service.redaction.v1.server.classification.service;
import static java.util.stream.Collectors.toSet;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.Iterator;
import java.util.List;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.Orientation;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.StringFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.classification.utils.RulingTextDirAdjustUtil;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import org.springframework.stereotype.Service;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.Iterator;
import java.util.List;
@Service
@SuppressWarnings("all")
@@ -29,11 +26,19 @@ public class BlockificationService {
static final float THRESHOLD = 1f;
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
sortRotatedSequences(textPositions);
/**
* This method is building blocks by expanding the minX/maxX and minY/maxY value on each word that is not split by the conditions.
* This method must use text direction adjusted postions (DirAdj). Where {0,0} is on the upper left. Never try to change this!
* Rulings (Table lines) must be adjusted to the text directions as well, when checking if a block is split by a ruling.
*
* @param textPositions The words of a page.
* @param horizontalRulingLines Horizontal table lines.
* @param verticalRulingLines Vertical table lines.
* @return Page object that contains the Textblock and text statistics.
*/
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
int indexOnPage = 0;
List<TextPositionSequence> chunkWords = new ArrayList<>();
List<AbstractTextContainer> chunkBlockList1 = new ArrayList<>();
@@ -44,35 +49,36 @@ public class BlockificationService {
Float splitX1 = null;
for (TextPositionSequence word : textPositions) {
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
boolean startFromTop = word.getY1() > maxY + word.getHeight();
boolean splitByX = prev != null && maxX + 50 < word.getX1() && prev.getY1() == word.getY1();
boolean newLineAfterSplit = prev != null && word.getY1() != prev.getY1() && wasSplitted && splitX1 != word.getX1();
boolean splittedByRuling = word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines) || word
.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines) || word
.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines) || word
.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines);
boolean lineSeparation = word.getMinYDirAdj() - maxY > word.getHeight() * 1.25;
boolean startFromTop = prev != null && word.getMinYDirAdj() < prev.getMinYDirAdj() - prev.getTextHeight();
boolean splitByX = prev != null && maxX + 50 < word.getMinXDirAdj() && prev.getMinYDirAdj() == word.getMinYDirAdj();
boolean xIsBeforeFirstX = prev != null && word.getMinXDirAdj() < minX;
boolean newLineAfterSplit = prev != null && word.getMinYDirAdj() != prev.getMinYDirAdj() && wasSplitted && splitX1 != word.getMinXDirAdj();
boolean isSplitByRuling = isSplitByRuling(minX, minY, maxX, maxY, word, horizontalRulingLines, verticalRulingLines);
boolean splitByDir = prev != null && !prev.getDir().equals(word.getDir());
if (prev != null && (lineSeparation || startFromTop || splitByX || newLineAfterSplit || splittedByRuling)) {
if (prev != null && (lineSeparation || startFromTop || splitByX || splitByDir || isSplitByRuling)) {
Orientation prevOrientation = null;
if (!chunkBlockList1.isEmpty()) {
prevOrientation = chunkBlockList1.get(chunkBlockList1.size() - 1).getOrientation();
}
TextBlock cb1 = buildTextBlock(chunkWords);
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
indexOnPage++;
chunkBlockList1.add(cb1);
chunkWords = new ArrayList<>();
if (splitByX && !splittedByRuling) {
if (splitByX && !isSplitByRuling) {
wasSplitted = true;
cb1.setOrientation(Orientation.LEFT);
splitX1 = word.getX1();
} else if (newLineAfterSplit && !splittedByRuling) {
splitX1 = word.getMinXDirAdj();
} else if (newLineAfterSplit && !isSplitByRuling) {
wasSplitted = false;
cb1.setOrientation(Orientation.RIGHT);
splitX1 = null;
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !splittedByRuling)) {
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !isSplitByRuling)) {
cb1.setOrientation(Orientation.LEFT);
}
@@ -86,21 +92,21 @@ public class BlockificationService {
chunkWords.add(word);
prev = word;
if (word.getX1() < minX) {
minX = word.getX1();
if (word.getMinXDirAdj() < minX) {
minX = word.getMinXDirAdj();
}
if (word.getX2() > maxX) {
maxX = word.getX2();
if (word.getMaxXDirAdj() > maxX) {
maxX = word.getMaxXDirAdj();
}
if (word.getY1() < minY) {
minY = word.getY1();
if (word.getMinYDirAdj() < minY) {
minY = word.getMinYDirAdj();
}
if (word.getY2() > maxY) {
maxY = word.getY2();
if (word.getMaxYDirAdj() > maxY) {
maxY = word.getMaxYDirAdj();
}
}
TextBlock cb1 = buildTextBlock(chunkWords);
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
if (cb1 != null) {
chunkBlockList1.add(cb1);
}
@@ -113,8 +119,7 @@ public class BlockificationService {
TextBlock block = (TextBlock) itty.next();
if (previousLeft != null && block.getOrientation().equals(Orientation.LEFT)) {
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft
.getMinY()) {
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft.getMinY()) {
previousLeft.add(block);
itty.remove();
continue;
@@ -122,8 +127,7 @@ public class BlockificationService {
}
if (previousRight != null && block.getOrientation().equals(Orientation.RIGHT)) {
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight
.getMinY()) {
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight.getMinY()) {
previousRight.add(block);
itty.remove();
continue;
@@ -142,10 +146,8 @@ public class BlockificationService {
while (itty.hasNext()) {
TextBlock block = (TextBlock) itty.next();
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY()) || previous != null && previous
.getOrientation()
.equals(Orientation.LEFT) && block.getOrientation()
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation().equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(),
previous.getMaxY()) || previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.RIGHT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY())) {
previous.add(block);
itty.remove();
@@ -165,7 +167,7 @@ public class BlockificationService {
}
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList) {
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList, int indexOnPage) {
TextBlock textBlock = null;
@@ -184,12 +186,16 @@ public class BlockificationService {
styleFrequencyCounter.add(wordBlock.getFontStyle());
if (textBlock == null) {
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
.getRotation());
textBlock = new TextBlock(wordBlock.getMinXDirAdj(),
wordBlock.getMaxXDirAdj(),
wordBlock.getMinYDirAdj(),
wordBlock.getMaxYDirAdj(),
wordBlockList,
wordBlock.getRotation(),
indexOnPage);
} else {
TextBlock spatialEntity = textBlock.union(wordBlock);
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
.getHeight());
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity.getHeight());
}
}
@@ -202,22 +208,61 @@ public class BlockificationService {
textBlock.setHighestFontSize(fontSizeFrequencyCounter.getHighest());
}
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences()
.stream()
.map(t -> round(t.getY1(), 3))
.collect(toSet())
.size() == 1) {
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getX1));
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences().stream().map(t -> round(t.getMinYDirAdj(), 3)).collect(toSet()).size() == 1) {
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getMinXDirAdj));
}
return textBlock;
}
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1,
List<Ruling> rulingLines) {
private boolean isSplitByRuling(float minX,
float minY,
float maxX,
float maxY,
TextPositionSequence word,
List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
return isSplitByRuling(maxX,
minY,
word.getMinXDirAdj(),
word.getMinYDirAdj(),
verticalRulingLines,
word.getDir().getDegrees(),
word.getPageWidth(),
word.getPageHeight()) //
|| isSplitByRuling(minX,
minY,
word.getMinXDirAdj(),
word.getMaxYDirAdj(),
horizontalRulingLines,
word.getDir().getDegrees(),
word.getPageWidth(),
word.getPageHeight()) //
|| isSplitByRuling(maxX,
minY,
word.getMinXDirAdj(),
word.getMinYDirAdj(),
horizontalRulingLines,
word.getDir().getDegrees(),
word.getPageWidth(),
word.getPageHeight()) //
|| isSplitByRuling(minX,
minY,
word.getMinXDirAdj(),
word.getMaxYDirAdj(),
verticalRulingLines,
word.getDir().getDegrees(),
word.getPageWidth(),
word.getPageHeight()); //
}
private boolean isSplitByRuling(float previousX2, float previousY1, float currentX1, float currentY1, List<Ruling> rulingLines, float dir, float pageWidth, float pageHeight) {
for (Ruling ruling : rulingLines) {
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
var line = RulingTextDirAdjustUtil.convertToDirAdj(ruling, dir, pageWidth, pageHeight);
if (line.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
return true;
}
}
@@ -225,104 +270,6 @@ public class BlockificationService {
}
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter,
boolean landscape) {
float minX = 10000;
float maxX = -100;
float minY = 10000;
float maxY = -100;
for (Page page : pages) {
if (page.getTextBlocks().isEmpty() || landscape != page.isLandscape()) {
continue;
}
for (AbstractTextContainer container : page.getTextBlocks()) {
if (container instanceof TextBlock) {
TextBlock textBlock = (TextBlock) container;
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
continue;
}
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
if (approxLineCount < 2.9f) {
continue;
}
if (documentFontSizeCounter.getMostPopular() != null) {
if (textBlock.getMostPopularWordFontSize() >= documentFontSizeCounter.getMostPopular()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
if (textBlock.getMaxX() > maxX) {
maxX = textBlock.getMaxX();
}
if (textBlock.getMinY() < minY) {
minY = textBlock.getMinY();
}
if (textBlock.getMaxY() > maxY) {
maxY = textBlock.getMaxY();
}
}
}
}
if (container instanceof Table) {
Table table = (Table) container;
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (cell == null || cell.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : cell.getTextBlocks()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
if (textBlock.getMaxX() > maxX) {
maxX = textBlock.getMaxX();
}
if (textBlock.getMinY() < minY) {
minY = textBlock.getMinY();
}
if (textBlock.getMaxY() > maxY) {
maxY = textBlock.getMaxY();
}
}
}
}
}
}
}
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
}
private void sortRotatedSequences(List<TextPositionSequence> sequences) {
List<TextPositionSequence> rotatedWords = new ArrayList<>();
Iterator<TextPositionSequence> itty = sequences.iterator();
while (itty.hasNext()) {
var pos = itty.next();
if (pos.getTextPositions().get(0).getDir() == 270) {
rotatedWords.add(pos);
itty.remove();
}
}
if (!rotatedWords.isEmpty() && !sequences.isEmpty()) {
rotatedWords.sort(Comparator.comparing(TextPositionSequence::getX1));
}
sequences.addAll(rotatedWords);
}
private double round(float value, int decimalPoints) {
var d = Math.pow(10, decimalPoints);
@@ -0,0 +1,161 @@
package com.iqser.red.service.redaction.v1.server.classification.service;
import java.util.List;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@Service
public class BodyTextFrameService {
/**
* Adjusts and sets the body text frame to a page.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
* The aspect ratio of the page is also regarded.
*
* @param page The page
* @param bodyTextFrame frame that contains the main text on portrait pages
* @param landscapeBodyTextFrame frame that contains the main text on landscape pages
*/
public void setBodyTextFrameAdjustedToPage(Page page, Rectangle bodyTextFrame, Rectangle landscapeBodyTextFrame) {
Rectangle textFrame = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
if (page.getPageWidth() > page.getPageHeight() && page.getRotation() == 270) {
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getY(), page.getPageHeight() - textFrame.getTopLeft().getX() - textFrame.getWidth()),
textFrame.getHeight(),
textFrame.getWidth(),
0);
} else if (page.getPageWidth() > page.getPageHeight() && page.getRotation() != 0) {
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getY(), textFrame.getTopLeft().getX()), textFrame.getHeight(), textFrame.getWidth(), page.getPageNumber());
} else if (page.getRotation() == 180) {
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getX(), page.getPageHeight() - textFrame.getTopLeft().getY() - textFrame.getHeight()),
textFrame.getWidth(),
textFrame.getHeight(),
0);
}
page.setBodyTextFrame(textFrame);
}
/**
* Calculates the frame that contains the main text, text outside the frame will be e.g. headers or footers.
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
* The aspect ratio of the page is also regarded.
*
* @param pages List of all pages
* @param documentFontSizeCounter Statistics of the document
* @param landscape Calculate for landscape or portrait
* @return Rectangle of the text frame
*/
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape) {
BodyTextFrameExpansionsRectangle expansionsRectangle = new BodyTextFrameExpansionsRectangle();
for (Page page : pages) {
if (page.getTextBlocks().isEmpty() || landscape != page.isLandscape()) {
continue;
}
for (AbstractTextContainer container : page.getTextBlocks()) {
if (container instanceof TextBlock) {
TextBlock textBlock = (TextBlock) container;
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
continue;
}
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
if (approxLineCount < 2.9f) {
continue;
}
if (documentFontSizeCounter.getMostPopular() != null && textBlock.getMostPopularWordFontSize() >= documentFontSizeCounter.getMostPopular()) {
expandRectangle(textBlock, page, expansionsRectangle);
}
}
if (container instanceof Table) {
Table table = (Table) container;
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (cell == null || cell.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : cell.getTextBlocks()) {
expandRectangle(textBlock, page, expansionsRectangle);
}
}
}
}
}
}
return new Rectangle(new Point(expansionsRectangle.minX, expansionsRectangle.minY),
expansionsRectangle.maxX - expansionsRectangle.minX,
expansionsRectangle.maxY - expansionsRectangle.minY,
0);
}
private void expandRectangle(TextBlock textBlock, Page page, BodyTextFrameExpansionsRectangle expansionsRectangle) {
if (page.getPageWidth() > page.getPageHeight() && page.getRotation() != 0) {
if (textBlock.getPdfMinY() < expansionsRectangle.minX) {
expansionsRectangle.minX = textBlock.getPdfMinY();
}
if (textBlock.getPdfMaxY() > expansionsRectangle.maxX) {
expansionsRectangle.maxX = textBlock.getPdfMaxY();
}
if (textBlock.getPdfMinX() < expansionsRectangle.minY) {
expansionsRectangle.minY = textBlock.getPdfMinX();
}
if (textBlock.getPdfMaxX() > expansionsRectangle.maxY) {
expansionsRectangle.maxY = textBlock.getPdfMaxX();
}
} else {
if (textBlock.getPdfMinX() < expansionsRectangle.minX) {
expansionsRectangle.minX = textBlock.getPdfMinX();
}
if (textBlock.getPdfMaxX() > expansionsRectangle.maxX) {
expansionsRectangle.maxX = textBlock.getPdfMaxX();
}
if (textBlock.getPdfMinY() < expansionsRectangle.minY) {
expansionsRectangle.minY = textBlock.getPdfMinY();
}
if (textBlock.getPdfMaxY() > expansionsRectangle.maxY) {
expansionsRectangle.maxY = textBlock.getPdfMaxY();
}
}
}
private class BodyTextFrameExpansionsRectangle {
float minX = 10000;
float maxX = -100;
float minY = 10000;
float maxY = -100;
}
}
@@ -1,82 +1,82 @@
package com.iqser.red.service.redaction.v1.server.classification.service;
import java.util.List;
import java.util.regex.Pattern;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import java.util.List;
import java.util.regex.Pattern;
@Slf4j
@Service
@RequiredArgsConstructor
public class ClassificationService {
private final BlockificationService blockificationService;
private final BodyTextFrameService bodyTextFrameService;
public void classifyDocument(Document document) {
Rectangle bodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false);
Rectangle landscapeBodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true);
Rectangle bodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false);
Rectangle landscapeBodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true);
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
for (Page page : document.getPages()) {
Rectangle btf = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
page.setBodyTextFrame(btf);
classifyPage(btf, page, document, headlineFontSizes);
bodyTextFrameService.setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
classifyPage(page, document, headlineFontSizes);
}
}
public void classifyPage(Rectangle bodyTextFrame, Page page, Document document, List<Float> headlineFontSizes) {
public void classifyPage(Page page, Document document, List<Float> headlineFontSizes) {
for (AbstractTextContainer textBlock : page.getTextBlocks()) {
if (textBlock instanceof TextBlock) {
classifyBlock((TextBlock) textBlock, bodyTextFrame, page, document, headlineFontSizes);
classifyBlock((TextBlock) textBlock, page, document, headlineFontSizes);
}
}
}
public void classifyBlock(TextBlock textBlock, Rectangle bodyTextFrame, Page page, Document document,
List<Float> headlineFontSizes) {
public void classifyBlock(TextBlock textBlock, Page page, Document document, List<Float> headlineFontSizes) {
var bodyTextFrame = page.getBodyTextFrame();
if (document.getFontSizeCounter().getMostPopular() == null) {
textBlock.setClassification("Other");
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification("Header");
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification("Footer");
} else if (page.getPageNumber() == 1 && (!PositionUtils.isTouchingUnderBodyTextFrame(bodyTextFrame, textBlock) && PositionUtils
.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock, document.getTextHeightCounter()
.getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter()
.getMostPopular() || page.getTextBlocks().size() == 1)) {
} else if (page.getPageNumber() == 1 && (PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock,
document.getTextHeightCounter().getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter().getMostPopular() || page.getTextBlocks()
.size() == 1)) {
if (!Pattern.matches("[0-9]+", textBlock.toString())) {
textBlock.setClassification("Title");
}
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle()
.equals("bold") || !document.getFontStyleCounter().getCountPerValue().containsKey("bold") && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() + 1) && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
} else if (textBlock.getMostPopularWordFontSize() > document.getFontSizeCounter()
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle().equals("bold") || !document.getFontStyleCounter()
.getCountPerValue()
.containsKey("bold") && textBlock.getMostPopularWordFontSize() > document.getFontSizeCounter().getMostPopular() + 1) && textBlock.getSequences()
.get(0)
.getTextPositions()
.get(0)
.getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
for (int i = 1; i <= headlineFontSizes.size(); i++) {
if (textBlock.getMostPopularWordFontSize() == headlineFontSizes.get(i - 1)) {
@@ -84,28 +84,25 @@ public class ClassificationService {
document.setHeadlines(true);
}
}
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText()
.startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter()
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText().startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame,
textBlock) && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter()
.getMostPopular()
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences()
.get(0)
.getTextPositions()
.get(0)
.getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
textBlock.setClassification("H " + (headlineFontSizes.size() + 1));
document.setHeadlines(true);
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
textBlock.setClassification("TextBlock Bold");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFont()
.equals(document.getFontCounter().getMostPopular()) && textBlock.getMostPopularWordStyle()
.equals(document.getFontStyleCounter()
.getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
.getMostPopular()) {
.equals(document.getFontStyleCounter().getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular()) {
textBlock.setClassification("TextBlock");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("italic") && !document.getFontStyleCounter()
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle().equals("italic") && !document.getFontStyleCounter()
.getMostPopular()
.equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
textBlock.setClassification("TextBlock Italic");
@@ -1,28 +1,27 @@
package com.iqser.red.service.redaction.v1.server.classification.utils;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.experimental.UtilityClass;
@UtilityClass
@SuppressWarnings("all")
public class PositionUtils {
public final class PositionUtils {
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
public boolean isWithinBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
if (btf == null || textBlock == null) {
return false;
}
double threshold = textBlock.getMostPopularWordHeight() * 3;
if (textBlock.getMinX() + threshold > btf.getX() &&
textBlock.getMaxX() - threshold < btf.getX() + btf.getWidth() &&
textBlock.getMinY() + threshold > btf.getY() &&
textBlock.getMaxY() - threshold < btf.getY() + btf.getHeight()) {
if (textBlock.getPdfMinX() + threshold > btf.getTopLeft().getX() && textBlock.getPdfMaxX() - threshold < btf.getTopLeft()
.getX() + btf.getWidth() && textBlock.getPdfMinY() + threshold > btf.getTopLeft().getY() && textBlock.getPdfMaxY() - threshold < btf.getTopLeft()
.getY() + btf.getHeight()) {
return true;
} else {
return false;
@@ -31,16 +30,27 @@ public class PositionUtils {
}
public boolean isOverBodyTextFrame(Rectangle btf, TextBlock textBlock, boolean rotated) {
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
public boolean isOverBodyTextFrame(Rectangle btf, TextBlock textBlock, int rotation) {
if (btf == null || textBlock == null) {
return false;
}
if (rotated && textBlock.getMinX() < btf.getX()) {
// Its very strange, P{0,0} is on top left in this case, instead of lower left.
if (rotation == 90 && textBlock.getPdfMaxX() < btf.getTopLeft().getX()) {
return true;
} else if (!rotated && textBlock.getMinY() > btf.getY() + btf.getHeight()) {
}
if (rotation == 180 && textBlock.getPdfMaxY() < btf.getTopLeft().getY()) {
return true;
}
if (rotation == 270 && textBlock.getPdfMinX() > btf.getTopLeft().getX() + btf.getWidth()) {
return true;
}
if (rotation == 0 && textBlock.getPdfMinY() > btf.getTopLeft().getY() + btf.getHeight()) {
return true;
} else {
return false;
@@ -48,16 +58,27 @@ public class PositionUtils {
}
public boolean isUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
public boolean isUnderBodyTextFrame(Rectangle btf, TextBlock textBlock, int rotation) {
if (btf == null || textBlock == null) {
return false;
}
if (textBlock.getMaxY() < btf.getY()) {
if (rotation == 90 && textBlock.getPdfMinX() > btf.getTopLeft().getX() + btf.getWidth()) {
return true;
}
if (rotation == 180 && textBlock.getPdfMinY() > btf.getTopLeft().getY() + btf.getHeight()) {
return true;
}
if (rotation == 270 && textBlock.getPdfMaxX() < btf.getTopLeft().getX()) {
return true;
}
if (rotation == 0 && textBlock.getPdfMaxY() < btf.getTopLeft().getY()) {
return true;
} else {
return false;
@@ -65,7 +86,8 @@ public class PositionUtils {
}
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
public boolean isTouchingUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
@@ -74,7 +96,7 @@ public class PositionUtils {
return false;
}
if (textBlock.getMinY() < btf.getY()) {
if (textBlock.getMinY() < btf.getTopLeft().getY()) {
return true;
} else {
return false;
@@ -84,11 +106,14 @@ public class PositionUtils {
public float getHeightDifferenceBetweenChunkWordAndDocumentWord(TextBlock textBlock, Float documentMostPopularWordHeight) {
return textBlock.getMostPopularWordHeight() - documentMostPopularWordHeight;
}
public Float getApproxLineCount(TextBlock textBlock) {
return textBlock.getHeight() / textBlock.getMostPopularWordHeight();
}
}
@@ -0,0 +1,67 @@
package com.iqser.red.service.redaction.v1.server.classification.utils;
import java.awt.geom.Line2D;
import java.awt.geom.Point2D;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import lombok.experimental.UtilityClass;
@UtilityClass
public final class RulingTextDirAdjustUtil {
/**
* Converts a ruling (line of a table) the same way TextPositions are converted in PDFBox.
* This will get the y position of the text, adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
*
* See org.apache.pdfbox.text.TextPosition
*/
public Line2D.Float convertToDirAdj(Ruling ruling, float dir, float pageWidth, float pageHeight) {
return new Line2D.Float(convertPoint(ruling.x1, ruling.y1, dir, pageWidth, pageHeight), convertPoint(ruling.x2, ruling.y2, dir, pageWidth, pageHeight));
}
private Point2D convertPoint(float x, float y, float dir, float pageWidth, float pageHeight) {
var xAdj = getXRot(x, y, dir, pageWidth, pageHeight);
var yAdj = 0f;
if (dir == 0 || dir == 180) {
yAdj = pageHeight - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
} else {
yAdj = pageWidth - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
}
return new Point2D.Float(xAdj, yAdj);
}
private float getXRot(float x, float y, float dir, float pageWidth, float pageHeight) {
if (dir == 0) {
return x;
} else if (dir == 90) {
return y;
} else if (dir == 180) {
return pageWidth - x;
} else if (dir == 270) {
return pageHeight - y;
}
return 0;
}
private float getYLowerLeftRot(float x, float y, float dir, float pageWidth, float pageHeight) {
if (dir == 0) {
return y;
} else if (dir == 90) {
return pageWidth - x;
} else if (dir == 180) {
return pageHeight - y;
} else if (dir == 270) {
return x;
}
return 0;
}
}
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.DictionaryReso
@FeignClient(name = "DictionaryResource", url = "${persistence-service.url}")
public interface DictionaryClient extends DictionaryResource {
}
@@ -1,10 +1,10 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.persistence.service.v1.api.resources.FileStatusProcessingUpdateResource;
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${persistence-service.url}")
public interface FileStatusProcessingUpdateClient extends FileStatusProcessingUpdateResource {
}
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.LegalBasisMapp
@FeignClient(name = "LegalBasisMappingResource", url = "${persistence-service.url}")
public interface LegalBasisClient extends LegalBasisMappingResource {
}
@@ -1,16 +1,16 @@
package com.iqser.red.service.redaction.v1.server.client;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.IOException;
import java.io.InputStream;
import org.springframework.lang.NonNull;
import org.springframework.lang.Nullable;
import org.springframework.util.Assert;
import org.springframework.util.FileCopyUtils;
import org.springframework.web.multipart.MultipartFile;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.IOException;
import java.io.InputStream;
public class MockMultipartFile implements MultipartFile {
private final String name;
@@ -32,8 +32,7 @@ public class MockMultipartFile implements MultipartFile {
}
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
@Nullable byte[] content) {
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, @Nullable byte[] content) {
Assert.hasLength(name, "Name must not be empty");
this.name = name;
@@ -43,8 +42,7 @@ public class MockMultipartFile implements MultipartFile {
}
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
InputStream contentStream) throws IOException {
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, InputStream contentStream) throws IOException {
this(name, originalFilename, contentType, FileCopyUtils.copyToByteArray(contentStream));
}
@@ -82,13 +80,13 @@ public class MockMultipartFile implements MultipartFile {
}
public byte[] getBytes() throws IOException {
public byte[] getBytes() {
return this.content;
}
public InputStream getInputStream() throws IOException {
public InputStream getInputStream() {
return new ByteArrayInputStream(this.content);
}
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.RulesResource;
@FeignClient(name = "RulesResource", url = "${persistence-service.url}")
public interface RulesClient extends RulesResource {
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -13,9 +14,9 @@ import lombok.NoArgsConstructor;
@NoArgsConstructor
public class EntityRecogintionEntity {
private String value;
private int startOffset;
private int endOffset;
private String type;
private String value;
private int startOffset;
private int endOffset;
private String type;
}
@@ -13,6 +13,6 @@ import lombok.NoArgsConstructor;
@NoArgsConstructor
public class EntityRecognitionRequest {
private List<EntityRecognitionSection> data;
private List<EntityRecognitionSection> data;
}
@@ -17,4 +17,5 @@ public class EntityRecognitionResult {
@Builder.Default
private Map<Integer, List<EntityRecogintionEntity>> entities = new HashMap<>();
}
@@ -13,4 +13,5 @@ public class EntityRecognitionSection {
private int sectionNumber;
private String text;
}
@@ -5,8 +5,8 @@ import java.util.List;
import java.util.Map;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -16,6 +16,6 @@ import lombok.NoArgsConstructor;
@AllArgsConstructor
public class NerEntities {
private Map<Integer, List<EntityRecogintionEntity>> data = new HashMap<>();
private Map<Integer, List<EntityRecogintionEntity>> data = new HashMap<>();
}
@@ -3,7 +3,9 @@ package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.commons.spring.ErrorMessage;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import lombok.extern.slf4j.Slf4j;
import org.springframework.http.HttpStatus;
import org.springframework.web.bind.annotation.ExceptionHandler;
import org.springframework.web.bind.annotation.ResponseBody;
@@ -18,10 +20,12 @@ public class ControllerAdvice {
/* error handling */
@ResponseBody
@ResponseStatus(value = HttpStatus.INTERNAL_SERVER_ERROR)
@ExceptionHandler(value = NullPointerException.class)
public ErrorMessage handleContentNotFoundException(NullPointerException e) {
if (e != null) {
log.error(e.getMessage(), e);
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
@@ -30,17 +34,21 @@ public class ControllerAdvice {
return new ErrorMessage(OffsetDateTime.now(), "Nullpointer exception");
}
@ResponseBody
@ResponseStatus(value = HttpStatus.BAD_REQUEST)
@ExceptionHandler(value = RulesValidationException.class)
public ErrorMessage handleRulesValidationException(RulesValidationException e) {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
@ResponseBody
@ResponseStatus(value = HttpStatus.NOT_FOUND)
@ExceptionHandler(value = NotFoundException.class)
public ErrorMessage handleFileNotFoundException(NotFoundException e) {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
@@ -1,29 +1,34 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.redaction.v1.model.*;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.*;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.stream.Collectors;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogMergeService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RestController
@@ -37,14 +42,19 @@ public class RedactionController implements RedactionResource {
private final RedactionLogMergeService redactionLogMergeService;
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
@Override
public RedactionResult classify(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
redactionRequest.getFileId(),
FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
redactionRequest.getFileId(),
FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
pdDocument.setAllSecurityToBeRemoved(true);
@@ -66,11 +76,15 @@ public class RedactionController implements RedactionResource {
@Override
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
redactionRequest.getFileId(),
FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
redactionRequest.getFileId(),
FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
pdDocument.setAllSecurityToBeRemoved(true);
@@ -94,8 +108,10 @@ public class RedactionController implements RedactionResource {
Document classifiedDoc;
try {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
redactionRequest.getFileId(),
FileType.ORIGIN));
classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
} catch (Exception e) {
throw new RedactionException(e);
}
@@ -118,14 +134,17 @@ public class RedactionController implements RedactionResource {
@Override
public void testRules(@RequestBody String rules) {
droolsExecutionService.testRules(rules);
try {
droolsExecutionService.testRules(rules);
} catch (Exception e) {
throw new RulesValidationException("Could not test rules: " + e.getMessage(), e);
}
}
@Override
public RedactionLog getRedactionLog(RedactionRequest redactionRequest) {
return redactionLogMergeService.provideRedactionLog(redactionRequest);
}
@@ -134,19 +153,14 @@ public class RedactionController implements RedactionResource {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
return RedactionResult.builder()
.document(byteArrayOutputStream.toByteArray())
.numberOfPages(numberOfPages)
.build();
return RedactionResult.builder().document(byteArrayOutputStream.toByteArray()).numberOfPages(numberOfPages).build();
}
}
@Override
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions) {
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions) {
var result = manualRedactionSurroundingTextService.addSurroundingText(dossierId, fileId, manualRedactions);
log.info("Added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", dossierId, fileId, result.getDuration());
@@ -1,10 +1,11 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import com.iqser.red.service.redaction.v1.resources.RuleBuilderResource;
import com.iqser.red.service.redaction.v1.server.redaction.rulebuilder.RuleBuilderModelService;
import lombok.RequiredArgsConstructor;
import org.springframework.web.bind.annotation.RestController;
@RestController
@@ -13,8 +14,10 @@ public class RuleBuilderController implements RuleBuilderResource {
private final RuleBuilderModelService ruleBuilderModelService;
@Override
public RuleBuilderModel getRuleBuilderModel() {
return ruleBuilderModelService.getRuleBuilderModel();
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.server.document.data;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class AtomicTextBlockData {
Long id;
String searchText;
int start;
int end;
int[] lineBreaks;
int[] stringIdxToPositionIdx;
float[][] positions;
}
@@ -0,0 +1,19 @@
package com.iqser.red.service.redaction.v1.server.document.data;
import java.util.List;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class DocumentData {
List<PageData> pages;
List<AtomicTextBlockData> atomicTextBlocks;
TableOfContentsData tableOfContents;
}
@@ -0,0 +1,20 @@
package com.iqser.red.service.redaction.v1.server.document.data;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class PageData {
int number;
int height;
int width;
Long header;
Long footer;
}
@@ -0,0 +1,52 @@
package com.iqser.red.service.redaction.v1.server.document.data;
import static java.lang.String.format;
import java.util.Arrays;
import java.util.List;
import javax.management.openmbean.InvalidKeyException;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class TableOfContentsData {
List<EntryData> entries;
public EntryData get(String tocId) {
List<Integer> ids = getIds(tocId);
if (ids.size() < 1) {
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", tocId));
}
EntryData entry = entries.get(ids.get(0));
for (int id : ids.subList(1, ids.size())) {
entry = entry.subEntries().get(id);
}
return entry;
}
private static List<Integer> getIds(String idsAsString) {
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
}
@Builder
public record EntryData(String tocId, List<EntryData> subEntries, NodeType type, Long atomicTextBlock, Long page, int numberOnPage) {
}
}
@@ -0,0 +1,88 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import static java.lang.String.format;
import lombok.Setter;
@Setter
public class Boundary {
private int start;
private int end;
public Boundary(int start, int end) {
assert start <= end;
this.start = start;
this.end = end;
}
public int length() {
return end - start;
}
public int start() {
return start;
}
public int end() {
return end;
}
public boolean contains(Boundary boundary) {
return start <= boundary.start() && boundary.end() <= end;
}
public boolean containedBy(Boundary boundary) {
return boundary.start() <= start && end <= boundary.end();
}
public boolean contains(int start, int end) {
if (start > end) {
throw new UnsupportedOperationException("start > end");
}
return this.start <= start && end <= this.end;
}
public boolean containedBy(int start, int end) {
if (start > end) {
throw new UnsupportedOperationException("start > end");
}
return start <= this.start && this.end <= end;
}
public boolean contains(int index) {
return start <= index && index < end;
}
public boolean intersects(Boundary boundary) {
return contains(boundary.start()) || contains(boundary.end());
}
@Override
public String toString() {
return format("Boundary [%d|%d)", start, end);
}
}
@@ -0,0 +1,96 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DocumentGraph {
List<SectionNode> sections;
List<PageNode> pages;
TableOfContents tableOfContents;
Integer numberOfPages;
TextBlock text;
public ConcatenatedTextBlock buildTextBlock() {
return streamAtomicTextBlocksInOrder().collect(new TextBlockCollector());
}
public Stream<AtomicTextBlock> streamAtomicTextBlocksInOrder() {
return Stream.concat(//
streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock),//
Stream.concat(//
pages.stream().map(PageNode::getHeader),//
pages.stream().map(PageNode::getFooter)));
}
public EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
addEntityToGraphAndSetFields(entity);
return entity;
}
public void addEntityToGraphAndSetFields(EntityNode entity) {
try {
boolean inserted = streamAllNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
} catch (NotFoundException e) {
enrichEntity(entity, text);
log.warn("Entity \"{}\" with {} is in between two main sections and will be removed!", entity.getValue(), entity.getBoundary());
entity.removeFromGraph();
}
}
public Set<EntityNode> getEntities() {
return streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getEntities).flatMap(List::stream).collect(Collectors.toSet());
}
private Stream<DocumentGraphNode> streamAllNodes() {
return tableOfContents.streamEntriesInOrder().map(TableOfContents.Entry::node);
}
@Override
public String toString() {
return text.toString();
}
}
@@ -0,0 +1,113 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import static java.lang.String.format;
import java.nio.charset.StandardCharsets;
import java.util.Arrays;
import java.util.LinkedList;
import java.util.List;
import java.util.stream.Stream;
import javax.management.openmbean.InvalidKeyException;
import com.google.common.hash.Hashing;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
import lombok.Data;
@Data
public class TableOfContents {
List<Entry> entries;
public TableOfContents() {
entries = new LinkedList<>();
}
public String createNewEntryAndReturnId(NodeType nodeType, String summary, DocumentGraphNode node) {
String id = String.format("%d", entries.size());
entries.add(new Entry(nodeType, id, summary, new LinkedList<>(), node));
return id;
}
public String createNewChildEntryAndReturnId(String parentId, NodeType nodeType, String summary, DocumentGraphNode node) {
Entry parent = getEntryById(parentId);
String childId = parentId + String.format(".%d", parent.children().size());
parent.children().add(new Entry(nodeType, childId, summary, new LinkedList<>(), node));
return childId;
}
public Entry getEntryById(String parentId) {
List<Integer> ids = getIds(parentId);
if (ids.size() < 1) {
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", parentId));
}
Entry entry = entries.get(ids.get(0));
for (int id : ids.subList(1, ids.size())) {
entry = entry.children().get(id);
}
return entry;
}
@Override
public String toString() {
return String.join("\n", streamEntriesInOrder().map(Entry::toString).toList());
}
public String toString(String id) {
return String.join("\n", streamSubEntriesInOrder(id).map(Entry::toString).toList());
}
public Stream<Entry> streamEntriesInOrder() {
return entries.stream().flatMap(TableOfContents::flatten);
}
public Stream<Entry> streamSubEntriesInOrder(String parentId) {
return Stream.of(getEntryById(parentId)).flatMap(TableOfContents::flatten);
}
private static List<Integer> getIds(String idsAsString) {
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
}
private static Stream<Entry> flatten(Entry entry) {
return Stream.concat(Stream.of(entry), entry.children().stream().flatMap(TableOfContents::flatten));
}
public record Entry(NodeType type, String id, String summary, List<Entry> children, DocumentGraphNode node) {
@Override
public String toString() {
return id + ": " + type + ".: " + summary;
}
@Override
public int hashCode() {
return Hashing.murmur3_32_fixed().hashString(type + id + summary + children.hashCode(), StandardCharsets.UTF_8).hashCode();
}
}
}
@@ -0,0 +1,219 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
import static java.lang.String.format;
import java.util.Set;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
public interface DocumentGraphNode {
/**
* Searches all Nodes located underneath this Node in the TableOfContents and concatenates their AtomicTextBlocks into a single TextBlockEntity.
* So, for a Section all AtomicTextBlocks of Subsections, Paragraphs, and Tables are concatenated into a single TextBlockEntity
*
* @return TextBlock containing all AtomicTextBlocks that are located under this Node.
*/
TextBlock buildTextBlock();
/**
* Any Node maintains its own Set of Entities.
* This Set contains all Entities, whose first index is located in any of the AtomicTextBlocks underneath this Node.
*
* @return Set of all Entities associated with this Node
*/
Set<EntityNode> getEntities();
/**
* Returns the PageNode associated with this Node.
* If the node has more than one PageNode associated, it returns the PageNode with the lowest number.
* For example a section might span multiple pages, it then returns the page where the section starts.
*
* @return PageNode representing the first page on which the Node is located in the document
*/
PageNode getPage();
/**
* Any Node except the First level of Sections, Header, Footer, and Pages have a direct Parent.
* For example a Paragraph has a parent Section, a Table Cell has a parent Table, etc...
* hasParent() may be used to check whether a parent is present.
*
* @return Node that represents the Parent or null, if no parent is present.
*/
DocumentGraphNode getParent();
Stream<DocumentGraphNode> streamAllSubNodes();
/**
* Each AtomicTextBlock has a number assigned per page, this returns the number of the first AtomicTextBlock underneath this node
*
* @return Integer representing the number on the page
*/
Integer getNumberOnPage();
/**
*
* @return the fist headline whent traversing the tree upwards
*/
default CharSequence getHeadline() {
return getParent().getHeadline();
}
/**
* By default, a parent is always present, this needs to be overwritten for Headers, Footers, Pages, and Sections.
*
* @return boolean, indicating whether a parent is present.
*/
default boolean hasParent() {
return true;
}
/**
* by default a Node does not have direct access to an AtomicTextBlock
*
* @return boolean, indicating if a Node has direct access to an AtomicTextBlock
*/
default boolean isTerminal() {
return false;
}
/**
* by default a Node does not have direct access to an AtomicTextBlock, this method throws a UnsupportedOperationException if not overridden.
*
* @return AtomicTextBlock
*/
default AtomicTextBlock getAtomicTextBlock() {
throw new UnsupportedOperationException("Only terminal Nodes have access to AtomicTextBlocks!");
}
/**
* creates an EntityNode with only initial values set, inserts it into the subgraph contained by this node and sets the inferrable fields.
* Throws NotFoundException and removes the entity if the provided boundary could not be found in the subgraph.
*
* @param boundary start and end indices in String coordinates of the entity to be created
* @param type type of the entity to be created
* @param entityType entityType of the entity to be created
* @return the newly created and inserted EntityNode with all fields set.
*/
default EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
addEntityToNodeAndSetFields(entity);
return entity;
}
/**
* searches for the first terminal Node containing the start index of the entity to be inserted.
* Catches NotFoundException to remove the EntityNode from the graph, then rethrows it
*
* @param entity newly created EntityNode with only initial values set
*/
default void addEntityToNodeAndSetFields(EntityNode entity) {
try {
streamAllSubNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
} catch (NotFoundException e) {
entity.removeFromGraph();
throw new RuntimeException(e);
}
}
/**
* If this Node's AtomicTextBlock contains the start index of the entity, the entity's position is read from the AtomicTextBlock.
* If the position can not be fully read from the AtomicTextBlock, it recursively looks in the TextBlocks of the parents until all positions are found.
* Further, the function throws NotFoundException if no parent contains all positions.
* This occurs, when the Entity is in between Nodes that do not share a parent, e.g. main sections.
* Finally, the function adds the Entity to its own list of Entities and to every parents' list recursively.
*
* @param entity The entity to be added to the graph
* @return true, if the entity has been added successfully.
* false, if the entity's start index is not contained or the Node doesn't have an AtomicTextBlock
*/
default boolean addEntityAndSetFieldsIfStartIndexContained(EntityNode entity) {
if (!isTerminal()) {
return false;
}
AtomicTextBlock atomicTextBlock = getAtomicTextBlock();
if (atomicTextBlock.containsIndex(entity.getBoundary().start())) {
entity.addContainingNode(this);
getEntities().add(entity);
addEntityToPage(entity);
addEntityToParents(entity);
setFields(entity, atomicTextBlock);
return true;
}
return false;
}
private void addEntityToPage(EntityNode entity) {
getPage().getEntities().add(entity);
entity.setPage(getPage());
}
private void setFields(EntityNode entity, AtomicTextBlock atomicTextBlock) {
if (atomicTextBlock.containsBoundary(entity.getBoundary())) {
enrichEntity(entity, atomicTextBlock);
} else {
this.setFieldsFromParents(this, entity);
}
}
private void addEntityToParents(EntityNode entity) {
DocumentGraphNode node = this;
while (node.hasParent()) {
node = node.getParent();
node.getEntities().add(entity);
entity.addContainingNode(node);
}
}
private void setFieldsFromParents(DocumentGraphNode node, EntityNode entity) {
if (node.hasParent()) {
DocumentGraphNode parent = node.getParent();
TextBlock textBlock = parent.buildTextBlock();
if (textBlock.containsBoundary(entity.getBoundary())) {
enrichEntity(entity, textBlock);
return;
} else {
setFieldsFromParents(parent, entity);
}
}
throw new NotFoundException(format("Position could not be found for Entity %s", entity.toString()));
}
}
@@ -0,0 +1,103 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.awt.geom.Rectangle2D;
import java.nio.charset.StandardCharsets;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import com.google.common.hash.Hashing;
import com.iqser.red.service.redaction.v1.model.Engine;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class EntityNode {
public static EntityNode initialEntityNode(Boundary boundary, String type, EntityType entityType) {
return EntityNode.builder().type(type).entityType(entityType).boundary(boundary).build();
}
// initial values
Boundary boundary;
String type;
EntityType entityType;
@Builder.Default
boolean redaction = false;
@Builder.Default
boolean falsePositive = false;
@Builder.Default
boolean removed = false;
@Builder.Default
boolean ignored = false;
@Builder.Default
boolean resized = false;
@Builder.Default
boolean skipRemoveEntitiesContainedInLarger = false;
@Builder.Default
boolean isDictionaryEntry = false;
@Builder.Default
Set<Engine> engines = new HashSet<>();
@Builder.Default
Set<Entity> references = new HashSet<>();
@Builder.Default
int matchedRule = -1;
@Builder.Default
String redactionReason = "";
@Builder.Default
String legalBasis = "";
// inferrable from graph
String value;
CharSequence textBefore;
CharSequence textAfter;
PageNode page;
List<Rectangle2D> positions;
@Builder.Default
Set<DocumentGraphNode> containingNodes = new HashSet<>();
public void addContainingNode(DocumentGraphNode containingNode) {
containingNodes.add(containingNode);
}
public void removeFromGraph() {
getContainingNodes().forEach(node -> node.getEntities().remove(this));
getPage().getEntities().remove(this);
setRemoved(true);
}
@Override
public int hashCode() {
var sb = new StringBuilder();
sb.append(value);
sb.append(boundary.start());
sb.append(page.getNumber());
positions.forEach(r -> {
sb.append(r.getMinX());
sb.append(r.getMinY());
sb.append(r.getWidth());
sb.append(r.getHeight());
});
return Hashing.murmur3_128().hashString(sb.toString(), StandardCharsets.UTF_8).hashCode();
}
}
@@ -0,0 +1,28 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
public enum NodeType {
SECTION {
public String toString() {
return "Section";
}
},
PARAGRAPH {
public String toString() {
return "Paragraph";
}
},
TABLE {
public String toString() {
return "Table";
}
},
TABLE_CELL {
public String toString() {
return "Cell";
}
}
}
@@ -0,0 +1,85 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class PageNode implements DocumentGraphNode{
Integer number;
Integer height;
Integer width;
List<DocumentGraphNode> mainBody;
AtomicTextBlock header;
AtomicTextBlock footer;
@Builder.Default
@EqualsAndHashCode.Exclude
Set<EntityNode> entities = new HashSet<>();
public ConcatenatedTextBlock buildTextBlock() {
return mainBody.stream().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
}
@Override
public PageNode getPage() {
return this;
}
@Override
public DocumentGraphNode getParent() {
return null;
}
@Override
public boolean hasParent() {
return false;
}
@Override
public Stream<DocumentGraphNode> streamAllSubNodes() {
return mainBody.stream();
}
@Override
public Integer getNumberOnPage() {
return 0;
}
@Override
public String toString() {
return header.getSearchText() + buildTextBlock().toString() + footer.getSearchText();
}
}
@@ -0,0 +1,70 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.util.HashSet;
import java.util.Set;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class ParagraphNode implements DocumentGraphNode {
String tocId;
Integer numberOnPage;
Integer numberInSection;
@EqualsAndHashCode.Exclude
SectionNode parentSection;
@EqualsAndHashCode.Exclude
PageNode page;
AtomicTextBlock atomicTextBlock;
@Builder.Default
@EqualsAndHashCode.Exclude
Set<EntityNode> entities = new HashSet<>();
@Override
public AtomicTextBlock buildTextBlock() {
return atomicTextBlock;
}
@Override
public DocumentGraphNode getParent() {
return parentSection;
}
@Override
public boolean isTerminal() {
return true;
}
@Override
public String toString() {
return tocId + ": " + atomicTextBlock.toString();
}
@Override
public Stream<DocumentGraphNode> streamAllSubNodes() {
return Stream.of(this);
}
}
@@ -0,0 +1,108 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class SectionNode implements DocumentGraphNode {
String tocId;
Integer numberOnPage;
@EqualsAndHashCode.Exclude
TableOfContents tableOfContents;
@EqualsAndHashCode.Exclude
DocumentGraphNode parentSection;
AtomicTextBlock headline;
List<SectionNode> subSections;
List<ParagraphNode> paragraphs;
List<TableNode> tables;
@EqualsAndHashCode.Exclude
List<PageNode> pages;
@Builder.Default
@EqualsAndHashCode.Exclude
Set<EntityNode> entities = new HashSet<>();
@Override
public String toString() {
return tocId + ": " + headline.toString();
}
@Override
public ConcatenatedTextBlock buildTextBlock() {
return streamAllSubNodes().map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
}
@Override
public DocumentGraphNode getParent() {
if (hasParent()) {
return parentSection;
} else {
throw new UnsupportedOperationException("This section has no parent Section!");
}
}
@Override
public Stream<DocumentGraphNode> streamAllSubNodes() {
return tableOfContents.streamSubEntriesInOrder(tocId).map(TableOfContents.Entry::node);
}
@Override
public boolean hasParent() {
return parentSection != null;
}
@Override
public boolean isTerminal() {
return true;
}
@Override
public AtomicTextBlock getAtomicTextBlock() {
return headline;
}
@Override
public PageNode getPage() {
return pages.get(0);
}
}
@@ -0,0 +1,59 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.util.LinkedList;
import java.util.List;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class TableCellNode implements DocumentGraphNode {
@EqualsAndHashCode.Exclude
TableNode parentTable;
Integer numberOnPage;
AtomicTextBlock atomicTextBlock;
PageNode page;
@Builder.Default
@EqualsAndHashCode.Exclude
List<EntityNode> entities = new LinkedList<>();
@Override
public AtomicTextBlock buildTextBlock() {
return atomicTextBlock;
}
@Override
public DocumentGraphNode getParent() {
return parentTable;
}
@Override
public boolean isTerminal() {
return true;
}
@Override
public Stream<DocumentGraphNode> streamAllSubNodes() {
return Stream.of(this);
}
}
@@ -0,0 +1,88 @@
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
import java.util.LinkedList;
import java.util.List;
import java.util.function.Function;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class TableNode implements DocumentGraphNode {
Integer id;
String tocId;
Integer numberOfRows;
Integer numberOfCols;
Integer numberOnPage;
List<TableCellNode> tableHeaders;
List<List<TableCellNode>> tableCells;
TableOfContents tableOfContents;
@EqualsAndHashCode.Exclude
SectionNode parentSection;
@EqualsAndHashCode.Exclude
List<PageNode> pages;
@Builder.Default
@EqualsAndHashCode.Exclude
List<EntityNode> entities = new LinkedList<>();
private Stream<TableCellNode> streamTableCells() {
return tableCells.stream().flatMap(List::stream);
}
private Stream<TableCellNode> streamTableRow(int row) {
return tableCells.get(row).stream();
}
private Stream<TableCellNode> streamTableCol(int col) {
return tableCells.stream().map(row -> row.get(col));
}
@Override
public ConcatenatedTextBlock buildTextBlock() {
return streamTableCells().map(TableCellNode::getAtomicTextBlock).collect(new TextBlockCollector());
}
@Override
public Stream<DocumentGraphNode> streamAllSubNodes() {
return streamTableCells().map(Function.identity());
}
@Override
public DocumentGraphNode getParent() {
return parentSection;
}
@Override
public PageNode getPage() {
return pages.get(0);
}
}
@@ -0,0 +1,126 @@
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class AtomicTextBlock implements TextBlock {
Long id;
//string coordinates
Boundary boundary;
String searchText;
List<Integer> lineBreaks;
//position coordinates
List<Integer> stringIdxToPositionIdx;
List<Rectangle2D> positions;
@EqualsAndHashCode.Exclude
DocumentGraphNode parent;
public int indexOf(String searchTerm) {
int pos = searchText.indexOf(searchTerm);
return pos == -1 ? -1 : pos + boundary.start();
}
public int numberOfLines() {
return lineBreaks.size();
}
@Override
public List<AtomicTextBlock> getAtomicTextBlocks() {
return List.of(this);
}
public int getNextLinebreak(int fromIndex) {
return lineBreaks.stream()//
.filter(linebreak -> linebreak > fromIndex) //
.findFirst() //
.orElse(searchText.length()) + boundary.start();
}
public int getPreviousLinebreak(int fromIndex) {
return lineBreaks.stream()//
.filter(linebreak -> linebreak <= fromIndex)//
.reduce((a, b) -> b)//
.orElse(0) + boundary.start();
}
public Rectangle2D getPosition(int stringIdx) {
return positions.get(stringIdxToPositionIdx.get(stringIdx - boundary.start()));
}
public List<Rectangle2D> getPositions(Boundary boundary) {
if (!containsBoundary(boundary)) {
throw new IndexOutOfBoundsException(format("%s is out of bounds for %s",
boundary,
this.boundary));
}
if (boundary.end() == this.boundary.end()) {
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), positions.size());
}
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), stringIdxToPositionIdx.get(boundary.end() - this.boundary.start()));
}
@Override
public int length() {
return searchText.length();
}
@Override
public char charAt(int index) {
return searchText.charAt(index - boundary.start());
}
@Override
public CharSequence subSequence(int start, int end) {
return searchText.substring(start - boundary.start(), end - boundary.start());
}
@Override
public String toString() {
return searchText;
}
}
@@ -0,0 +1,155 @@
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.LinkedList;
import java.util.List;
import java.util.function.Supplier;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import lombok.AccessLevel;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@FieldDefaults(level = AccessLevel.PRIVATE)
public class ConcatenatedTextBlock implements TextBlock, Supplier<ConcatenatedTextBlock> {
List<AtomicTextBlock> atomicTextBlocks;
StringBuilder searchText;
Boundary boundary;
public ConcatenatedTextBlock(List<AtomicTextBlock> atomicTextBlocks) {
this.atomicTextBlocks = new LinkedList<>();
this.searchText = new StringBuilder();
if (atomicTextBlocks.isEmpty()) {
boundary = new Boundary(-1, -1);
return;
}
var firstTextBlock = atomicTextBlocks.get(0);
this.atomicTextBlocks.add(firstTextBlock);
this.searchText.append(firstTextBlock.getSearchText());
boundary = new Boundary(firstTextBlock.getBoundary().start(), firstTextBlock.getBoundary().end());
atomicTextBlocks.subList(1, atomicTextBlocks.size()).forEach(this::concat);
}
public ConcatenatedTextBlock(AtomicTextBlock atomicTextBlocks) {
new ConcatenatedTextBlock(List.of(atomicTextBlocks));
}
public ConcatenatedTextBlock concat(TextBlock textBlock) {
if (this.atomicTextBlocks.isEmpty()) {
boundary.setStart(textBlock.getBoundary().start());
boundary.setEnd(textBlock.getBoundary().end());
} else if (boundary.end() != textBlock.getBoundary().start()) {
throw new UnsupportedOperationException(format("Can only concat consecutive TextBlocks, trying to concat %s to %s", textBlock.getBoundary(), boundary));
}
this.searchText.append(textBlock.getSearchText());
this.atomicTextBlocks.addAll(textBlock.getAtomicTextBlocks());
boundary.setEnd(textBlock.getBoundary().end());
return this;
}
public int indexOf(String searchTerm) {
int pos = this.searchText.indexOf(searchTerm);
return pos == -1 ? -1 : pos + boundary.start();
}
public int numberOfLines() {
return atomicTextBlocks.stream().map(AtomicTextBlock::getLineBreaks).mapToInt(List::size).sum();
}
public int getNextLinebreak(int fromIndex) {
return getAtomicTextBlockByStringIndex(fromIndex).getNextLinebreak(fromIndex);
}
public int getPreviousLinebreak(int fromIndex) {
return getAtomicTextBlockByStringIndex(fromIndex).getPreviousLinebreak(fromIndex);
}
public Rectangle2D getPosition(int stringIdx) {
return getAtomicTextBlockByStringIndex(stringIdx).getPosition(stringIdx);
}
public List<Rectangle2D> getPositions(Boundary boundary) {
List<AtomicTextBlock> textBlocks = getAllAtomicTextBlocksPartiallyInStringIdxRange(boundary);
if (textBlocks.size() == 1) {
return textBlocks.get(0).getPositions(boundary);
}
AtomicTextBlock firstTextBlock = textBlocks.get(0);
List<Rectangle2D> positions = new LinkedList<>(firstTextBlock.getPositions(new Boundary(boundary.start(), firstTextBlock.getBoundary().end())));
for (AtomicTextBlock textBlock : textBlocks.subList(1, textBlocks.size() - 1)) {
positions.addAll(textBlock.getPositions());
}
var lastTextBlock = textBlocks.get(textBlocks.size() - 1);
positions.addAll(lastTextBlock.getPositions(new Boundary(lastTextBlock.getBoundary().start(), boundary.end())));
return positions;
}
private AtomicTextBlock getAtomicTextBlockByStringIndex(int stringIdx) {
return atomicTextBlocks.stream().filter(textBlock -> (textBlock.getBoundary().end()) > stringIdx).findFirst().orElseThrow(IndexOutOfBoundsException::new);
}
private List<AtomicTextBlock> getAllAtomicTextBlocksPartiallyInStringIdxRange(Boundary boundary) {
return atomicTextBlocks.stream().filter(tb -> tb.getBoundary().intersects(boundary)).toList();
}
@Override
public int length() {
return this.searchText.length();
}
@Override
public char charAt(int index) {
return searchText.charAt(index - boundary.start());
}
@Override
public CharSequence subSequence(int start, int end) {
return searchText.subSequence(start - boundary.start(), end - boundary.start());
}
@Override
public ConcatenatedTextBlock get() {
return this;
}
}
@@ -0,0 +1,65 @@
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
public interface TextBlock extends CharSequence {
CharSequence getSearchText();
List<AtomicTextBlock> getAtomicTextBlocks();
Boundary getBoundary();
int getNextLinebreak(int fromIndex);
int getPreviousLinebreak(int fromIndex);
Rectangle2D getPosition(int stringIdx);
List<Rectangle2D> getPositions(Boundary range);
int numberOfLines();
int indexOf(String searchTerm);
default CharSequence getFirstLine() {
return subSequence(getBoundary().start(), getNextLinebreak(getBoundary().start()));
}
default boolean containsBoundary(Boundary boundary) {
if (boundary.end() < boundary.start()) {
throw new IllegalArgumentException(format("Invalid %s, StartIndex must be smaller than EndIndex", boundary));
}
return getBoundary().contains(boundary);
}
default boolean containsIndex(int stringIndex) {
return getBoundary().contains(stringIndex);
}
default CharSequence subSequence(Boundary boundary) {
return subSequence(boundary.start(), boundary.end());
}
}
@@ -0,0 +1,51 @@
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
import java.util.Collections;
import java.util.Set;
import java.util.function.BiConsumer;
import java.util.function.BinaryOperator;
import java.util.function.Function;
import java.util.function.Supplier;
import java.util.stream.Collector;
import lombok.NoArgsConstructor;
@NoArgsConstructor
public class TextBlockCollector implements Collector<AtomicTextBlock, ConcatenatedTextBlock, ConcatenatedTextBlock> {
@Override
public Supplier<ConcatenatedTextBlock> supplier() {
return new ConcatenatedTextBlock(Collections.emptyList());
}
@Override
public BiConsumer<ConcatenatedTextBlock, AtomicTextBlock> accumulator() {
return ConcatenatedTextBlock::concat;
}
@Override
public BinaryOperator<ConcatenatedTextBlock> combiner() {
return ConcatenatedTextBlock::concat;
}
@Override
public Function<ConcatenatedTextBlock, ConcatenatedTextBlock> finisher() {
return Function.identity();
}
@Override
public Set<Characteristics> characteristics() {
return Set.of(Characteristics.IDENTITY_FINISH, Characteristics.CONCURRENT);
}
}
@@ -0,0 +1,98 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import java.awt.geom.Rectangle2D;
import java.util.List;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
@Service
public class DocumentDataMapper {
public DocumentData toDocumentData(DocumentGraph documentGraph) {
List<AtomicTextBlockData> atomicTextBlockData = documentGraph.streamAtomicTextBlocksInOrder().map(this::toAtomicTextBlockData).toList();
List<PageData> pageData = documentGraph.getPages().stream().map(this::toPageData).toList();
TableOfContentsData tableOfContentsData = toTableOfContentsData(documentGraph.getTableOfContents());
return DocumentData.builder().atomicTextBlocks(atomicTextBlockData).pages(pageData).tableOfContents(tableOfContentsData).build();
}
private TableOfContentsData toTableOfContentsData(TableOfContents tableOfContents) {
return new TableOfContentsData(tableOfContents.getEntries().stream().map(this::toEntryData).toList());
}
private TableOfContentsData.EntryData toEntryData(TableOfContents.Entry entry) {
return TableOfContentsData.EntryData.builder()
.tocId(entry.id())
.subEntries(entry.children().stream().map(this::toEntryData).toList())
.type(entry.type())
.atomicTextBlock(entry.node().isTerminal() ? entry.node().getAtomicTextBlock().getId() : -1L)
.page(Long.valueOf(entry.node().getPage().getNumber()))
.numberOnPage(entry.node().getNumberOnPage())
.build();
}
private PageData toPageData(PageNode pageNode) {
return PageData.builder()
.height(pageNode.getHeight())
.width(pageNode.getWidth())
.number(pageNode.getNumber())
.footer(pageNode.getFooter().getId())
.header(pageNode.getHeader().getId())
.build();
}
private AtomicTextBlockData toAtomicTextBlockData(AtomicTextBlock atomicTextBlock) {
return AtomicTextBlockData.builder()
.id(atomicTextBlock.getId())
.searchText(atomicTextBlock.getSearchText())
.start(atomicTextBlock.getBoundary().start())
.end(atomicTextBlock.getBoundary().end())
.lineBreaks(toPrimitiveIntArray(atomicTextBlock.getLineBreaks()))
.stringIdxToPositionIdx(toPrimitiveIntArray(atomicTextBlock.getStringIdxToPositionIdx()))
.positions(toPrimitiveFloatMatrix(atomicTextBlock.getPositions()))
.build();
}
private float[][] toPrimitiveFloatMatrix(List<Rectangle2D> positions) {
float[][] positionMatrix = new float[positions.size()][];
for (int i = 0; i < positions.size(); i++) {
float[] singlePositions = new float[4];
singlePositions[0] = (float) positions.get(i).getMinX();
singlePositions[1] = (float) positions.get(i).getMinY();
singlePositions[2] = (float) positions.get(i).getWidth();
singlePositions[3] = (float) positions.get(i).getHeight();
positionMatrix[i] = singlePositions;
}
return positionMatrix;
}
private int[] toPrimitiveIntArray(List<Integer> list) {
int[] array = new int[list.size()];
for (int i = 0; i < list.size(); i++) {
array[i] = list.get(i);
}
return array;
}
}
@@ -0,0 +1,304 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Collections;
import java.util.Comparator;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.concurrent.atomic.AtomicInteger;
import java.util.concurrent.atomic.AtomicLong;
import java.util.stream.Collectors;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Footer;
import com.iqser.red.service.redaction.v1.server.classification.model.Header;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.Section;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableNode;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.RedRectangle2D;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchTextWithTextPositionModel;
import com.iqser.red.service.redaction.v1.server.redaction.service.SearchTextWithTextPositionFactory;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class DocumentGraphFactory {
private final SearchTextWithTextPositionFactory searchTextWithTextPositionFactory;
public DocumentGraph buildDocumentGraph(Document document) {
Context context = new Context(new TableOfContents(), new LinkedList<>(), new LinkedList<>(), new AtomicInteger(0), new AtomicLong(0));
context.pages.addAll(document.getPages().stream().map(this::buildPage).toList());
// is tracked by Table of Contents
addSections(document, context);
// not tracked by Table of Contents
addHeaderAndFooterToEachPage(document, context);
DocumentGraph documentGraph = DocumentGraph.builder()
.numberOfPages(context.pages.size())
.pages(context.pages)
.sections(context.sections)
.tableOfContents(context.tableOfContents)
.build();
documentGraph.setText(documentGraph.buildTextBlock());
return documentGraph;
}
private void addSections(Document document, Context context) {
for (var section : document.getSections()) {
addSection(section, context);
}
}
private void addSection(Section section, Context context) {
SectionNode sectionEntity = SectionNode.builder()
.entities(new LinkedList<>())
.pages(new LinkedList<>())
.paragraphs(new LinkedList<>())
.tables(new LinkedList<>())
.subSections(new LinkedList<>())
.tableOfContents(context.tableOfContents())
.build();
context.sections().add(sectionEntity);
List<AbstractTextContainer> pageBlocks = new ArrayList<>(section.getPageBlocks());
PageNode page = getPage(section.getPageBlocks().get(0).getPage(), context);
sectionEntity.getPages().add(page);
page.getMainBody().add(sectionEntity);
if (pageBlocks.get(0) instanceof TextBlock) {
sectionEntity.setHeadline(buildAtomicTextBlock(((TextBlock) pageBlocks.get(0)).getSequences(), sectionEntity, context));
sectionEntity.setNumberOnPage(((TextBlock) pageBlocks.get(0)).getIndexOnPage());
pageBlocks.remove(0);
} else {
sectionEntity.setNumberOnPage(1);
sectionEntity.setHeadline(emptyTextBlock(sectionEntity, context));
}
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(sectionEntity.getHeadline()), sectionEntity);
sectionEntity.setTocId(sectionId);
int paragraphIdx = 0;
int tableIdx = 0;
for (AbstractTextContainer abstractTextContainer : pageBlocks) {
if (abstractTextContainer instanceof TextBlock) {
addParagraph(sectionEntity, (TextBlock) abstractTextContainer, paragraphIdx, context);
paragraphIdx++;
} else if (abstractTextContainer instanceof Table) {
//addTable(sectionEntity, (Table) abstractTextContainer, tableIdx, context);
tableIdx++;
}
}
}
private void addTable(SectionNode sectionEntity, Table table, int tableIdx, Context context) {
PageNode page = getPage(table.getPage(), context);
TableNode tableEntity = TableNode.builder().id(tableIdx).tableOfContents(context.tableOfContents()).pages(new LinkedList<>()).parentSection(sectionEntity).build();
sectionEntity.getTables().add(tableEntity);
if (!page.getMainBody().contains(sectionEntity)) {
sectionEntity.getPages().add(page);
}
page.getMainBody().add(tableEntity);
}
private void addParagraph(SectionNode sectionEntity, TextBlock originalTextBlock, int paragraphIdx, Context context) {
PageNode page = getPage(originalTextBlock.getPage(), context);
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(originalTextBlock.getIndexOnPage()).page(page).parentSection(sectionEntity).build();
sectionEntity.getParagraphs().add(paragraph);
if (!page.getMainBody().contains(sectionEntity)) {
sectionEntity.getPages().add(page);
}
page.getMainBody().add(paragraph);
var textBlock = buildAtomicTextBlock(originalTextBlock.getSequences(), paragraph, context);
paragraph.setAtomicTextBlock(textBlock);
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(sectionEntity.getTocId(), NodeType.PARAGRAPH, buildSummary(textBlock), paragraph);
paragraph.setTocId(tocId);
}
private void addHeaderAndFooterToEachPage(Document document, Context context) {
Map<Integer, List<TextBlock>> headers = document.getHeaders()
.stream()
.map(Header::getTextBlocks)
.flatMap(List::stream)
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
Map<Integer, List<TextBlock>> footers = document.getFooters()
.stream()
.map(Footer::getTextBlocks)
.flatMap(List::stream)
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
if (headers.containsKey(pageIndex)) {
addHeader(headers.get(pageIndex), context);
} else {
addEmptyHeader(pageIndex, context);
}
}
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
if (footers.containsKey(pageIndex)) {
addFooter(footers.get(pageIndex), context);
} else {
addEmptyFooter(pageIndex, context);
}
}
}
private void addFooter(List<TextBlock> textBlocks, Context context) {
PageNode page = getPage(textBlocks.get(0).getPage(), context);
AtomicTextBlock footer = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
page.setFooter(footer);
}
public void addHeader(List<TextBlock> textBlocks, Context context) {
PageNode page = getPage(textBlocks.get(0).getPage(), context);
AtomicTextBlock header = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
page.setHeader(header);
}
private void addEmptyFooter(int pageIndex, Context context) {
PageNode page = getPage(pageIndex, context);
page.setFooter(emptyTextBlock(page, context));
}
private void addEmptyHeader(int pageIndex, Context context) {
PageNode page = getPage(pageIndex, context);
page.setHeader(emptyTextBlock(page, context));
}
private AtomicTextBlock emptyTextBlock(DocumentGraphNode parent, Context context) {
return AtomicTextBlock.builder()
.id(context.textBlockIdx.getAndIncrement())
.boundary(new Boundary(context.stringOffset.get(), context.stringOffset.get()))
.searchText("")
.lineBreaks(Collections.emptyList())
.stringIdxToPositionIdx(Collections.emptyList())
.positions(Collections.emptyList())
.parent(parent)
.build();
}
private static String buildSummary(AtomicTextBlock textBlock) {
if (textBlock == null) {
return " probably a table";
}
String[] words = textBlock.getFirstLine().toString().split(" ");
int bound = Math.min(words.length, 4);
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
return String.join(" ", list);
}
private PageNode buildPage(Page p) {
return PageNode.builder().height((int) p.getPageHeight()).width((int) p.getPageWidth()).number(p.getPageNumber()).mainBody(new LinkedList<>()).build();
}
private List<TextPositionSequence> mergeAndSortTextPositionSequences(List<TextBlock> textBlocks) {
Comparator<TextPositionSequence> sortByX = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[0] - sequence2.getTextPositions()
.get(0)
.getPosition()[0]);
Comparator<TextPositionSequence> sortByY = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[1] - sequence2.getTextPositions()
.get(0)
.getPosition()[1]);
return textBlocks.stream().map(TextBlock::getSequences).flatMap(List::stream).sorted(sortByX.thenComparing(sortByY)).toList();
}
private AtomicTextBlock buildAtomicTextBlock(List<TextPositionSequence> sequences, DocumentGraphNode parent, Context context) {
SearchTextWithTextPositionModel searchTextWithTextPositionModel = searchTextWithTextPositionFactory.buildSearchTextToTextPositionModel(sequences);
int offset = context.stringOffset().getAndAdd(searchTextWithTextPositionModel.getSearchText().length());
return AtomicTextBlock.builder()
.id(context.textBlockIdx.getAndIncrement())
.parent(parent)
.searchText(searchTextWithTextPositionModel.getSearchText())
.lineBreaks(searchTextWithTextPositionModel.getLineBreaks())
.positions(toRectangle2D(searchTextWithTextPositionModel.getPositions()))
.stringIdxToPositionIdx(searchTextWithTextPositionModel.getStringCoordsToPositionCoords())
.boundary(new Boundary(offset, offset + searchTextWithTextPositionModel.getSearchText().length()))
.build();
}
private List<Rectangle2D> toRectangle2D(List<RedRectangle2D> positions) {
return positions.stream().map(r -> (Rectangle2D) new Rectangle2D.Double(r.getX(), r.getY(), r.getWidth(), r.getHeight())).toList();
}
private PageNode getPage(int pageIndex, Context context) {
return context.pages.stream()
.filter(page -> page.getNumber() == pageIndex)
.findFirst()
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
}
record Context(
TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, AtomicInteger stringOffset, AtomicLong textBlockIdx) {
}
}
@@ -0,0 +1,171 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import static java.lang.Math.toIntExact;
import static java.lang.String.format;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.LinkedList;
import java.util.List;
import org.apache.commons.lang3.NotImplementedException;
import org.springframework.stereotype.Service;
import com.google.common.primitives.Ints;
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
@Service
public class DocumentGraphMapper {
public DocumentGraph toDocumentGraph(DocumentData documentData) {
Context context = new Context(documentData, new TableOfContents(), new LinkedList<>(), new LinkedList<>(), documentData.getAtomicTextBlocks());
context.pages.addAll(documentData.getPages().stream().map(pageData -> buildPage(pageData, context)).toList());
buildNodesFromTableOfContents("", context);
DocumentGraph documentGraph= DocumentGraph.builder()
.numberOfPages(documentData.getPages().size())
.pages(context.pages)
.sections(context.sections)
.tableOfContents(context.tableOfContents)
.build();
documentGraph.setText(documentGraph.buildTextBlock());
return documentGraph;
}
private void buildNodesFromTableOfContents(String currentTocId, Context context) {
List <TableOfContentsData.EntryData> entries;
if(currentTocId.equals("")) {
entries = context.documentData().getTableOfContents().getEntries();
} else {
entries = context.documentData().getTableOfContents().get(currentTocId).subEntries();
}
for (TableOfContentsData.EntryData entryData : entries) {
switch (entryData.type()) {
case SECTION -> buildSection(entryData, currentTocId, context);
case PARAGRAPH -> buildParagraph(entryData, currentTocId, context);
default -> throw new NotImplementedException("Not yet implemented for type " + entryData.type());
}
}
}
private void buildSection(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
SectionNode section = SectionNode.builder()
.entities(new LinkedList<>())
.pages(new LinkedList<>())
.paragraphs(new LinkedList<>())
.tables(new LinkedList<>())
.subSections(new LinkedList<>())
.tableOfContents(context.tableOfContents())
.numberOnPage(entryData.numberOnPage())
.build();
context.sections().add(section);
section.setHeadline(toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(entryData.atomicTextBlock())), section));
if (!currentTocId.equals("")) {
SectionNode parent = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
section.setParentSection(parent);
parent.getSubSections().add(section);
}
PageNode page = getPage(entryData.page(), context);
page.getMainBody().add(section);
section.getPages().add(page);
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(section.getHeadline()), section);
section.setTocId(sectionId);
buildNodesFromTableOfContents(sectionId, context);
}
private void buildParagraph(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
PageNode page = getPage(entryData.page(), context);
SectionNode parentSection = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(entryData.numberOnPage()).page(page).parentSection(parentSection).build();
AtomicTextBlock atomicTextBlock = toAtomicTextBlock(context.atomicTextBlockData.get(toIntExact(entryData.atomicTextBlock())), paragraph);
paragraph.setAtomicTextBlock(atomicTextBlock);
if (!page.getMainBody().contains(parentSection)) {
parentSection.getPages().add(page);
}
page.getMainBody().add(paragraph);
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(currentTocId, NodeType.PARAGRAPH, buildSummary(atomicTextBlock), paragraph);
paragraph.setTocId(tocId);
}
private PageNode buildPage(PageData p, Context context) {
PageNode page = PageNode.builder().height(p.getHeight()).width(p.getWidth()).number(p.getNumber()).mainBody(new LinkedList<>()).build();
AtomicTextBlock header = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getHeader())), page);
AtomicTextBlock footer = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getFooter())), page);
page.setHeader(header);
page.setFooter(footer);
return page;
}
private static String buildSummary(com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock textBlock) {
if (textBlock == null) {
return " probably a table";
}
String[] words = textBlock.getFirstLine().toString().split(" ");
int bound = Math.min(words.length, 4);
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
return String.join(" ", list);
}
private AtomicTextBlock toAtomicTextBlock(AtomicTextBlockData atomicTextBlockData, DocumentGraphNode parent) {
return AtomicTextBlock.builder()
.id(atomicTextBlockData.getId())
.searchText(atomicTextBlockData.getSearchText())
.boundary(new Boundary(atomicTextBlockData.getStart(), atomicTextBlockData.getEnd()))
.lineBreaks(Ints.asList(atomicTextBlockData.getLineBreaks()))
.positions(Arrays.stream(atomicTextBlockData.getPositions())
.map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3]))
.toList())
.stringIdxToPositionIdx(Ints.asList(atomicTextBlockData.getStringIdxToPositionIdx()))
.parent(parent)
.build();
}
private PageNode getPage(Long pageIndex, Context context) {
return context.pages.stream()
.filter(page -> page.getNumber() == toIntExact(pageIndex))
.findFirst()
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
}
record Context(DocumentData documentData, TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, List<AtomicTextBlockData> atomicTextBlockData) {
}
}
@@ -0,0 +1,31 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
public class EntityEnrichmentUtility {
public static EntityNode enrichEntity(EntityNode entity, TextBlock textBlock) {
entity.setPositions(textBlock.getPositions(entity.getBoundary()));
entity.setTextAfter(findTextAfter(entity.getBoundary().end(), textBlock));
entity.setTextBefore(findTextBefore(entity.getBoundary().start(), textBlock));
entity.setValue(textBlock.subSequence(entity.getBoundary()).toString());
return entity;
}
private static CharSequence findTextAfter(int index, TextBlock textBlock) {
int nextLineBreak = textBlock.getNextLinebreak(index);
return textBlock.subSequence(index, nextLineBreak);
}
private static CharSequence findTextBefore(int index, TextBlock textBlock) {
int previousLinebreak = textBlock.getPreviousLinebreak(index);
return textBlock.subSequence(previousLinebreak, index);
}
}
@@ -0,0 +1,38 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import java.util.Comparator;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
public class RangeComparators {
public static Comparator<Boundary> contained() {
return (range1, range2) -> {
if (contained(range1, range2)) {
return -1;
} else if (contained(range2, range1)) {
return 1;
} else {
return 0;
}
};
}
/**
* @param range1 A Range
* @param range2 Also Range
* @return true, if range1 contains range2
* false, otherwise
*/
public static boolean contained(Boundary range1, Boundary range2) {
return range1.start() <= range2.start() && range2.end() <= range1.end();
}
public static boolean contained(Boundary range, int index) {
return range.start() <= index && index < range.end();
}
}
@@ -0,0 +1,37 @@
package com.iqser.red.service.redaction.v1.server.document.services;
import java.util.LinkedList;
import java.util.List;
import java.util.regex.Matcher;
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
public class RegexMatcher {
public static boolean anyMatch(CharSequence searchText, String regexPattern) {
var pattern = Patterns.getCompiledPattern(regexPattern, false);
return pattern.matcher(searchText).find();
}
public static Boundary findFirstBoundary(String regexPattern, CharSequence searchText) {
var pattern = Patterns.getCompiledPattern(regexPattern, false);
Matcher matcher = pattern.matcher(searchText);
return new Boundary(matcher.start(), matcher.end());
}
public static List<Boundary> findBoundaries(String regexPattern, CharSequence searchText) {
var pattern = Patterns.getCompiledPattern(regexPattern, false);
Matcher matcher = pattern.matcher(searchText);
List<Boundary> boundaries = new LinkedList<>();
while (matcher.find()) {
boundaries.add(new Boundary(matcher.start(), matcher.end()));
}
return boundaries;
}
}
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.exception;
public class NotFoundException extends RuntimeException {
public NotFoundException(String message) {
super(message);
}
@@ -3,10 +3,13 @@ package com.iqser.red.service.redaction.v1.server.exception;
public class RedactionException extends RuntimeException {
public RedactionException(Throwable cause) {
super("Could not parse document", cause);
}
public RedactionException() {
super("Could not parse document");
}
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.exception;
public class RulesValidationException extends RuntimeException {
public RulesValidationException(String message, Throwable t) {
super(message, t);
}
@@ -69,17 +69,17 @@ import org.apache.pdfbox.pdmodel.font.PDFontDescriptor;
/**
* LEGACY text calculations which are known to be incorrect but are depended on by PDFTextStripper.
*
* <p>
* This class exists only so that we don't break the code of users who have their own subclasses of
* PDFTextStripper. It replaces the mostly empty implementation of showGlyph() in PDFStreamEngine
* with a heuristic implementation which is backwards compatible.
*
* <p>
* DO NOT USE THIS CODE UNLESS YOU ARE WORKING WITH PDFTextStripper.
* THIS CODE IS DELIBERATELY INCORRECT, USE PDFStreamEngine INSTEAD.
*/
@SuppressWarnings({"PMD", "checkstyle:all"})
class LegacyPDFStreamEngine extends PDFStreamEngine
{
class LegacyPDFStreamEngine extends PDFStreamEngine {
private static final Log LOG = LogFactory.getLog(LegacyPDFStreamEngine.class);
private int pageRotation;
@@ -88,11 +88,12 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
private final GlyphList glyphList;
private final Map<COSDictionary, Float> fontHeightMap = new WeakHashMap<COSDictionary, Float>();
/**
* Constructor.
*/
LegacyPDFStreamEngine() throws IOException
{
LegacyPDFStreamEngine() throws IOException {
addOperator(new BeginText());
addOperator(new Concatenate());
addOperator(new DrawObject()); // special text version
@@ -122,6 +123,7 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
glyphList = new GlyphList(GlyphList.getAdobeGlyphList(), input);
}
/**
* This will initialize and process the contents of the stream.
*
@@ -129,33 +131,27 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
* @throws java.io.IOException if there is an error accessing the stream.
*/
@Override
public void processPage(PDPage page) throws IOException
{
public void processPage(PDPage page) throws IOException {
this.pageRotation = page.getRotation();
this.pageSize = page.getCropBox();
if (pageSize.getLowerLeftX() == 0 && pageSize.getLowerLeftY() == 0)
{
if (pageSize.getLowerLeftX() == 0 && pageSize.getLowerLeftY() == 0) {
translateMatrix = null;
}
else
{
} else {
// translation matrix for cropbox
translateMatrix = Matrix.getTranslateInstance(-pageSize.getLowerLeftX(), -pageSize.getLowerLeftY());
}
super.processPage(page);
}
/**
* Called when a glyph is to be processed. The heuristic calculations here were originally
* written by Ben Litchfield for PDFStreamEngine.
*/
@Override
protected void showGlyph(Matrix textRenderingMatrix, PDFont font, int code,
String unicode,
Vector displacement)
throws IOException
{
protected void showGlyph(Matrix textRenderingMatrix, PDFont font, int code, String unicode, Vector displacement) throws IOException {
//
// legacy calculations which were previously in PDFStreamEngine
//
@@ -173,25 +169,19 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
// the sorting algorithm is based on the width of the character. As the displacement
// for vertical characters doesn't provide any suitable value for it, we have to
// calculate our own
if (font.isVertical())
{
if (font.isVertical()) {
displacementX = font.getWidth(code) / 1000;
// there may be an additional scaling factor for true type fonts
TrueTypeFont ttf = null;
if (font instanceof PDTrueTypeFont)
{
ttf = ((PDTrueTypeFont)font).getTrueTypeFont();
}
else if (font instanceof PDType0Font)
{
PDCIDFont cidFont = ((PDType0Font)font).getDescendantFont();
if (cidFont instanceof PDCIDFontType2)
{
ttf = ((PDCIDFontType2)cidFont).getTrueTypeFont();
if (font instanceof PDTrueTypeFont) {
ttf = ((PDTrueTypeFont) font).getTrueTypeFont();
} else if (font instanceof PDType0Font) {
PDCIDFont cidFont = ((PDType0Font) font).getDescendantFont();
if (cidFont instanceof PDCIDFontType2) {
ttf = ((PDCIDFontType2) cidFont).getTrueTypeFont();
}
}
if (ttf != null && ttf.getUnitsPerEm() != 1000)
{
if (ttf != null && ttf.getUnitsPerEm() != 1000) {
displacementX *= 1000f / ttf.getUnitsPerEm();
}
}
@@ -219,8 +209,7 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
// (modified) width and height calculations
float dxDisplay = nextX - textRenderingMatrix.getTranslateX();
Float fontHeight = fontHeightMap.get(font.getCOSObject());
if (fontHeight == null)
{
if (fontHeight == null) {
fontHeight = computeFontHeight(font);
fontHeightMap.put(font.getCOSObject(), fontHeight);
}
@@ -237,30 +226,24 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
// saved).
float glyphSpaceToTextSpaceFactor = 1 / 1000f;
if (font instanceof PDType3Font)
{
if (font instanceof PDType3Font) {
glyphSpaceToTextSpaceFactor = font.getFontMatrix().getScaleX();
}
float spaceWidthText = 0;
try
{
try {
// to avoid crash as described in PDFBOX-614, see what the space displacement should be
spaceWidthText = font.getSpaceWidth() * glyphSpaceToTextSpaceFactor;
}
catch (Throwable exception)
{
} catch (Throwable exception) {
LOG.warn(exception, exception);
}
if (spaceWidthText == 0)
{
if (spaceWidthText == 0) {
spaceWidthText = font.getAverageFontWidth() * glyphSpaceToTextSpaceFactor;
// the average space width appears to be higher than necessary so make it smaller
spaceWidthText *= .80f;
}
if (spaceWidthText == 0)
{
if (spaceWidthText == 0) {
spaceWidthText = 1.0f; // if could not find font, use a generic value
}
@@ -273,15 +256,11 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
// when there is no Unicode mapping available, Acrobat simply coerces the character code
// into Unicode, so we do the same. Subclasses of PDFStreamEngine don't necessarily want
// this, which is why we leave it until this point in PDFTextStreamEngine.
if (unicodeMapping == null)
{
if (font instanceof PDSimpleFont)
{
if (unicodeMapping == null) {
if (font instanceof PDSimpleFont) {
char c = (char) code;
unicodeMapping = new String(new char[] { c });
}
else
{
unicodeMapping = new String(new char[]{c});
} else {
// Acrobat doesn't seem to coerce composite font's character codes, instead it
// skips them. See the "allah2.pdf" TestTextStripper file.
return;
@@ -290,88 +269,118 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
// adjust for cropbox if needed
Matrix translatedTextRenderingMatrix;
if (translateMatrix == null)
{
if (translateMatrix == null) {
translatedTextRenderingMatrix = textRenderingMatrix;
}
else
{
} else {
translatedTextRenderingMatrix = Matrix.concatenate(translateMatrix, textRenderingMatrix);
nextX -= pageSize.getLowerLeftX();
nextY -= pageSize.getLowerLeftY();
}
processTextPosition(new TextPosition(pageRotation, pageSize.getWidth(),
pageSize.getHeight(), translatedTextRenderingMatrix, nextX, nextY,
Math.abs(dyDisplay), dxDisplay,
Math.abs(spaceWidthDisplay), unicodeMapping, new int[] { code }, font,
fontSize,
(int)(fontSize * textMatrix.getScalingFactorX())));
// This is a hack for unicode letter with 2 chars e.g. RA see unicodeProblem.pdf
if (unicodeMapping.length() == 2) {
processTextPosition(new TextPosition(pageRotation,
pageSize.getWidth(),
pageSize.getHeight(),
translatedTextRenderingMatrix,
nextX,
nextY,
Math.abs(dyDisplay),
dxDisplay,
Math.abs(spaceWidthDisplay),
Character.toString(unicodeMapping.charAt(0)),
new int[]{code},
font,
fontSize,
(int) (fontSize * textMatrix.getScalingFactorX())));
processTextPosition(new TextPosition(pageRotation,
pageSize.getWidth(),
pageSize.getHeight(),
translatedTextRenderingMatrix,
nextX,
nextY,
Math.abs(dyDisplay),
dxDisplay,
Math.abs(spaceWidthDisplay),
Character.toString(unicodeMapping.charAt(1)),
new int[]{code},
font,
fontSize,
(int) (fontSize * textMatrix.getScalingFactorX())));
} else {
processTextPosition(new TextPosition(pageRotation,
pageSize.getWidth(),
pageSize.getHeight(),
translatedTextRenderingMatrix,
nextX,
nextY,
Math.abs(dyDisplay),
dxDisplay,
Math.abs(spaceWidthDisplay),
unicodeMapping,
new int[]{code},
font,
fontSize,
(int) (fontSize * textMatrix.getScalingFactorX())));
}
}
/**
* Compute the font height. Override this if you want to use own calculations.
*
*
* @param font the font.
* @return the font height.
*
* @throws IOException if there is an error while getting the font bounding box.
*/
protected float computeFontHeight(PDFont font) throws IOException
{
protected float computeFontHeight(PDFont font) throws IOException {
BoundingBox bbox = font.getBoundingBox();
if (bbox.getLowerLeftY() < Short.MIN_VALUE)
{
if (bbox.getLowerLeftY() < Short.MIN_VALUE) {
// PDFBOX-2158 and PDFBOX-3130
// files by Salmat eSolutions / ClibPDF Library
bbox.setLowerLeftY(- (bbox.getLowerLeftY() + 65536));
bbox.setLowerLeftY(-(bbox.getLowerLeftY() + 65536));
}
// 1/2 the bbox is used as the height todo: why?
float glyphHeight = bbox.getHeight() / 2;
// sometimes the bbox has very high values, but CapHeight is OK
PDFontDescriptor fontDescriptor = font.getFontDescriptor();
if (fontDescriptor != null)
{
if (fontDescriptor != null) {
float capHeight = fontDescriptor.getCapHeight();
if (Float.compare(capHeight, 0) != 0 &&
(capHeight < glyphHeight || Float.compare(glyphHeight, 0) == 0))
{
if (Float.compare(capHeight, 0) != 0 && (capHeight < glyphHeight || Float.compare(glyphHeight, 0) == 0)) {
glyphHeight = capHeight;
}
// PDFBOX-3464, PDFBOX-4480, PDFBOX-4553:
// sometimes even CapHeight has very high value, but Ascent and Descent are ok
float ascent = fontDescriptor.getAscent();
float descent = fontDescriptor.getDescent();
if (capHeight > ascent && ascent > 0 && descent < 0 &&
((ascent - descent) / 2 < glyphHeight || Float.compare(glyphHeight, 0) == 0))
{
if (capHeight > ascent && ascent > 0 && descent < 0 && ((ascent - descent) / 2 < glyphHeight || Float.compare(glyphHeight, 0) == 0)) {
glyphHeight = (ascent - descent) / 2;
}
}
// transformPoint from glyph space -> text space
float height;
if (font instanceof PDType3Font)
{
if (font instanceof PDType3Font) {
height = font.getFontMatrix().transformPoint(0, glyphHeight).y;
}
else
{
} else {
height = glyphHeight / 1000;
}
return height;
}
/**
* A method provided as an event interface to allow a subclass to perform some specific
* functionality when text needs to be processed.
*
* @param text The text to be processed.
*/
protected void processTextPosition(TextPosition text)
{
protected void processTextPosition(TextPosition text) {
// subclasses can override to provide specific functionality
}
}
@@ -1,8 +1,10 @@
package com.iqser.red.service.redaction.v1.server.parsing;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.Getter;
import lombok.Setter;
import org.apache.pdfbox.text.PDFTextStripperByArea;
import org.apache.pdfbox.text.TextPosition;
@@ -18,19 +20,19 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
@Setter
private int pageNumber;
public PDFAreaTextStripper() throws IOException {
}
@Override
public void writeString(String text, List<TextPosition> textPositions) throws IOException {
int startIndex = 0;
for (int i = 0; i <= textPositions.size() - 1; i++) {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0"))) {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0"))) {
startIndex++;
continue;
}
@@ -38,32 +40,23 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i).getX() > textPositions.get(i - 1).getEndX() + 1) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0")) && i <= textPositions.size() - 2) {
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0")) && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i + 1;
@@ -71,14 +64,10 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
}
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
.getUnicode()
.equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
.getUnicode()
.equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
super.writeString(text);
@@ -86,6 +75,7 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
public void clearPositions() {
textPositionSequences = new ArrayList<>();
}
@@ -3,9 +3,11 @@ package com.iqser.red.service.redaction.v1.server.parsing;
import com.iqser.red.service.redaction.v1.server.parsing.model.RedTextPosition;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import lombok.Getter;
import lombok.Setter;
import lombok.extern.slf4j.Slf4j;
import org.apache.pdfbox.contentstream.operator.Operator;
import org.apache.pdfbox.contentstream.operator.OperatorName;
import org.apache.pdfbox.contentstream.operator.color.*;
@@ -20,6 +22,7 @@ import org.apache.pdfbox.text.TextPosition;
import java.awt.geom.Point2D;
import java.io.IOException;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
@Slf4j
@@ -155,7 +158,6 @@ public class PDFLinesTextStripper extends PDFTextStripper {
graphicsPath.clear();
break;
}
super.processOperator(operator, arguments);
@@ -182,14 +184,11 @@ public class PDFLinesTextStripper extends PDFTextStripper {
try {
if (stroke && !getGraphicsState().getStrokingColor().isPattern() && getGraphicsState().getStrokingColor()
.toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor()
.isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
.toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor().isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
rulings.addAll(path);
}
} catch (UnsupportedOperationException e) {
log.debug("UnsupportedOperationException: " + getGraphicsState().getStrokingColor()
.getColorSpace()
.getName() + " or " + getGraphicsState().getNonStrokingColor()
log.debug("UnsupportedOperationException: " + getGraphicsState().getStrokingColor().getColorSpace().getName() + " or " + getGraphicsState().getNonStrokingColor()
.getColorSpace()
.getName() + " does not support toRGB");
}
@@ -202,6 +201,8 @@ public class PDFLinesTextStripper extends PDFTextStripper {
int startIndex = 0;
RedTextPosition previous = null;
textPositions.sort(Comparator.comparing(TextPosition::getXDirAdj));
for (int i = 0; i <= textPositions.size() - 1; i++) {
if (!textPositionSequences.isEmpty()) {
@@ -226,9 +227,7 @@ public class PDFLinesTextStripper extends PDFTextStripper {
maxCharHeight = charHeight;
}
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0") || textPositions.get(i).getUnicode().equals("\t"))) {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0") || textPositions.get(i).getUnicode().equals("\t"))) {
startIndex++;
continue;
}
@@ -236,9 +235,7 @@ public class PDFLinesTextStripper extends PDFTextStripper {
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getXDirAdj() < textPositions.get(i - 1).getXDirAdj()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
@@ -246,12 +243,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
startIndex = i;
}
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i)
.getX() > textPositions.get(i - 1).getEndX() + 1) {
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i).getX() > textPositions.get(i - 1).getEndX() + 1) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
@@ -259,15 +253,11 @@ public class PDFLinesTextStripper extends PDFTextStripper {
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0") || textPositions.get(i)
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0") || textPositions.get(i)
.getUnicode()
.equals("\t")) && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
@@ -286,17 +276,15 @@ public class PDFLinesTextStripper extends PDFTextStripper {
}
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
.getUnicode()
.equals(" ") || sublist.get(sublist.size() - 1)
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1)
.getUnicode()
.equals("\u00A0") || sublist.get(sublist.size() - 1).getUnicode().equals("\t"))) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\u00A0") || sublist.get(0).getUnicode().equals("\t")))) {
.equals("\t")))) {
if (previous != null && sublist.get(0).getYDirAdj() == previous.getYDirAdj() && sublist.get(0)
.getXDirAdj() - (previous.getXDirAdj() + previous.getWidthDirAdj()) < 0.01) {
for (TextPosition t : sublist) {
@@ -21,16 +21,20 @@ import lombok.SneakyThrows;
public class RedTextPosition {
private String textMatrix;
private float[] position;
@JsonIgnore
private int rotation;
private float y;
@JsonIgnore
private float pageHeight;
@JsonIgnore
private float pageWidth;
private String unicode;
private float XDirAdj;
private float YDirAdj;
private float width;
private float heightDir;
private float widthDirAdj;
@JsonIgnore
private float dir;
// not used in reanalysis
@@ -51,6 +55,7 @@ public class RedTextPosition {
@SneakyThrows
public static RedTextPosition fromTextPosition(TextPosition textPosition) {
var pos = new RedTextPosition();
BeanUtils.copyProperties(textPosition, pos);
pos.setFontName(textPosition.getFont().getName());
@@ -59,8 +64,43 @@ public class RedTextPosition {
pos.setTextMatrix(textPosition.getTextMatrix().toString());
var position = new float[4];
position[0] = textPosition.getXDirAdj();
position[1] = textPosition.getYDirAdj();
position[2] = textPosition.getWidthDirAdj();
position[3] = textPosition.getHeightDir();
pos.setPosition(position);
return pos;
}
@JsonIgnore
public float getXDirAdj() {
return position[0];
}
@JsonIgnore
public float getYDirAdj() {
return position[1];
}
@JsonIgnore
public float getWidthDirAdj() {
return position[2];
}
@JsonIgnore
public float getHeightDir() {
return position[3];
}
}
@@ -0,0 +1,70 @@
package com.iqser.red.service.redaction.v1.server.parsing.model;
import java.util.Objects;
import com.fasterxml.jackson.annotation.JsonCreator;
import com.fasterxml.jackson.annotation.JsonValue;
import lombok.Getter;
@Getter
public enum TextDirection {
ZERO(0f),
QUARTER_CIRCLE(90f),
HALF_CIRCLE(180f),
THREE_QUARTER_CIRCLE(270f);
public static final String VALUE_STRING_SUFFIX = "°";
@JsonValue
private final float degrees;
private final float radians;
TextDirection(float degreeValue) {
degrees = degreeValue;
radians = (float) Math.toRadians(degreeValue);
}
@Override
public String toString() {
return degrees + VALUE_STRING_SUFFIX;
}
@com.dslplatform.json.JsonValue
public float jsonValue() {
return getDegrees();
}
@JsonCreator(mode = JsonCreator.Mode.DELEGATING)
public static TextDirection fromDegrees(float degrees) {
for (var dir : TextDirection.values()) {
if (degrees == dir.degrees) {
return dir;
}
}
throw new IllegalArgumentException(String.format("A value of %f is not supported by TextDirection", degrees));
}
public static TextDirection fromString(String degreesAsString) {
Objects.requireNonNull(degreesAsString, "Cannot construct a text direction from a null value");
String value = degreesAsString.strip();
if (degreesAsString.endsWith(VALUE_STRING_SUFFIX)) {
value = degreesAsString.replace(VALUE_STRING_SUFFIX + "$", "");
}
return fromDegrees(Float.parseFloat(value));
}
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.server.parsing.model;
import java.awt.geom.AffineTransform;
import java.awt.geom.Point2D;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Collectors;
@@ -17,6 +19,7 @@ import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@@ -28,11 +31,14 @@ import lombok.extern.slf4j.Slf4j;
@JsonIgnoreProperties({"empty"})
public class TextPositionSequence implements CharSequence {
public static final int HEIGHT_PADDING = 2;
private int page;
private List<RedTextPosition> textPositions = new ArrayList<>();
private float x1;
private float x2;
private TextDirection dir;
private int rotation;
private float pageHeight;
private float pageWidth;
public TextPositionSequence(int page) {
@@ -41,20 +47,14 @@ public class TextPositionSequence implements CharSequence {
}
public static TextPositionSequence fromData(List<RedTextPosition> textPositions, int page) {
var textPositionSequence = new TextPositionSequence();
textPositionSequence.textPositions = textPositions;
textPositionSequence.page = page;
return textPositionSequence;
}
public TextPositionSequence(List<TextPosition> textPositions, int page) {
this.textPositions = textPositions.stream().map(RedTextPosition::fromTextPosition).collect(Collectors.toList());
this.page = page;
this.dir = TextDirection.fromDegrees(textPositions.get(0).getDir());
this.rotation = textPositions.get(0).getRotation();
this.pageHeight = textPositions.get(0).getPageHeight();
this.pageWidth = textPositions.get(0).getPageWidth();
}
@@ -85,7 +85,15 @@ public class TextPositionSequence implements CharSequence {
@Override
public TextPositionSequence subSequence(int start, int end) {
return fromData(textPositions.subList(start, end), page);
var textPositionSequence = new TextPositionSequence();
textPositionSequence.textPositions = textPositions.subList(start, end);
textPositionSequence.page = page;
textPositionSequence.dir = dir;
textPositionSequence.rotation = rotation;
textPositionSequence.pageHeight = pageHeight;
textPositionSequence.pageWidth = pageWidth;
return textPositionSequence;
}
@@ -106,80 +114,86 @@ public class TextPositionSequence implements CharSequence {
}
public void add(RedTextPosition textPosition) {
public void add(TextPositionSequence textPositionSequence, RedTextPosition textPosition) {
this.textPositions.add(textPosition);
this.page = textPositionSequence.getPage();
this.dir = textPositionSequence.getDir();
this.rotation = textPositionSequence.getRotation();
this.pageHeight = textPositionSequence.getPageHeight();
this.pageWidth = textPositionSequence.getPageWidth();
}
public void add(TextPosition textPosition) {
this.textPositions.add(RedTextPosition.fromTextPosition(textPosition));
this.dir = TextDirection.fromDegrees(textPositions.get(0).getDir());
this.rotation = textPositions.get(0).getRotation();
this.pageHeight = textPositions.get(0).getPageHeight();
this.pageWidth = textPositions.get(0).getPageWidth();
}
/**
* This value is adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
* This method ignores the page rotation but takes the text rotation and adjusts the coordinates to awt.
*
* @return the text direction adjusted minX value
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getX1() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getYDirAdj() - getTextHeight();
} else {
return textPositions.get(0).getXDirAdj();
}
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getX2() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getYDirAdj();
} else {
return textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
}
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getRotationAdjustedY() {
return textPositions.get(0).getY();
}
@JsonIgnore
@JsonAttribute(ignore = true)
public float getRotationAdjustedX() {
public float getMinXDirAdj() {
return textPositions.get(0).getXDirAdj();
}
/**
* This value is adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
* This method ignores the page rotation but takes the text rotation and adjusts the coordinates to awt.
*
* @return the text direction adjusted maxX value
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getY1() {
public float getMaxXDirAdj() {
return textPositions.get(textPositions.size() - 1).getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + HEIGHT_PADDING;
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getXDirAdj();
} else {
return textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj();
}
}
/**
* This value is adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
* This method ignores the page rotation but takes the text rotation and adjusts the coordinates to awt.
*
* @return the text direction adjusted minY value. The upper border of the bounding box of the word.
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getY2() {
public float getMinYDirAdj() {
return textPositions.get(0).getYDirAdj() - getTextHeight();
}
/**
* This value is adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
* This method ignores the page rotation but takes the text rotation and adjusts the coordinates to awt.
*
* @return the text direction adjusted maxY value. The lower border of the bounding box of the word.
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public float getMaxYDirAdj() {
return textPositions.get(0).getYDirAdj();
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(textPositions.size() - 1).getXDirAdj() + getTextHeight() - 2;
} else {
return textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() + getTextHeight();
}
}
@@ -187,7 +201,7 @@ public class TextPositionSequence implements CharSequence {
@JsonAttribute(ignore = true)
public float getTextHeight() {
return textPositions.get(0).getHeightDir() + 2;
return textPositions.get(0).getHeightDir() + HEIGHT_PADDING;
}
@@ -195,7 +209,7 @@ public class TextPositionSequence implements CharSequence {
@JsonAttribute(ignore = true)
public float getHeight() {
return getY2() - getY1();
return getMaxYDirAdj() - getMinYDirAdj();
}
@@ -203,7 +217,7 @@ public class TextPositionSequence implements CharSequence {
@JsonAttribute(ignore = true)
public float getWidth() {
return getX2() - getX1();
return getMaxXDirAdj() - getMinXDirAdj();
}
@@ -250,159 +264,53 @@ public class TextPositionSequence implements CharSequence {
}
/**
* This returns the bounding box of the word in Pdf Coordinate System where {0,0} rotated with the page rotation.
* 0 -> LowerLeft
* 90 -> UpperLeft
* 180 -> UpperRight
* 270 -> LowerRight
*
* @return bounding box of the word in Pdf Coordinate System
*/
@JsonIgnore
@JsonAttribute(ignore = true)
public int getRotation() {
return textPositions.get(0).getRotation();
}
@JsonIgnore
@JsonAttribute(ignore = true)
@SneakyThrows
public Rectangle getRectangle() {
log.debug("Page: '{}', Word: '{}', Rotation: '{}', textRotation {}", page, toString(), textPositions.get(0)
.getRotation(), textPositions.get(0).getDir());
log.debug("Page: '{}', Word: '{}', Rotation: '{}', textRotation {}", page, this, rotation, dir);
float height = getTextHeight();
float textHeight = getTextHeight();
float posXInit = getX1();
float posXEnd;
float posYInit;
float posYEnd;
RedTextPosition firstTextPos = textPositions.get(0);
RedTextPosition lastTextPos = textPositions.get(textPositions.size() - 1);
if (textPositions.get(0).getRotation() == 0 && textPositions.get(0).getDir() == 90f) {
posYInit = getX1();
posYEnd = getX2() + textPositions.get(0).getWidthDirAdj() - textPositions.get(textPositions.size() - 1)
.getWidthDirAdj() - 3;
posXInit = textPositions.get(0).getYDirAdj() + 2;
posXEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height;
} else if (textPositions.get(0).getRotation() == 0 && textPositions.get(0).getDir() == 180f) {
posXInit = textPositions.get(0).getPageWidth() - getX1() + 1;
posXEnd = textPositions.get(0).getPageWidth() - getX2() + textPositions.get(0)
.getWidthDirAdj() - textPositions.get(textPositions.size() - 1).getWidthDirAdj() - 3;
posYInit = textPositions.get(0).getYDirAdj() - height + 2;
posYEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height + 2;
} else if (textPositions.get(0).getRotation() == 0 && textPositions.get(0).getDir() == 270f) {
posYInit = textPositions.get(0).getPageHeight() - getX1();
posYEnd = textPositions.get(0).getPageHeight() - getX2() - textPositions.get(0)
.getWidthDirAdj() - textPositions.get(textPositions.size() - 1).getWidthDirAdj() - 3;
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + height;
} else if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() == 0.0f) {
posXInit = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getHeightDir();
posXEnd = textPositions.get(0).getXDirAdj();
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
} else if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() == 90.0f) {
posXEnd = textPositions.get(0).getYDirAdj() + 2;
posYInit = getY1();
posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
} else if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() == 180.0f) {
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getXDirAdj() - 4;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(0).getXDirAdj();
posYInit = textPositions.get(0).getYDirAdj() - 2 - textPositions.get(textPositions.size() - 1)
.getHeightDir();
posYEnd = textPositions.get(textPositions.size() - 1)
.getYDirAdj() - textPositions.get(textPositions.size() - 1).getHeightDir();
} else if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() == 270.0f) {
posXInit = textPositions.get(0).getPageWidth() - getX1();
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posYInit = textPositions.get(0).getPageHeight() - getY1();
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getXDirAdj() - height - 4;
} else if (textPositions.get(0).getRotation() == 180 && textPositions.get(0).getDir() == 0f) {
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + 1;
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
} else if (textPositions.get(0).getRotation() == 180 && textPositions.get(0).getDir() == 90f) {
posYInit = getX1();
posYEnd = getX2() - 3;
posXInit = textPositions.get(0).getYDirAdj() + 2;
posXEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height;
} else if (textPositions.get(0).getRotation() == 180 && textPositions.get(0).getDir() == 180f) {
posXInit = textPositions.get(0).getPageWidth() - getX1() + 1;
posXEnd = textPositions.get(0).getPageWidth() - getX2() + textPositions.get(0)
.getWidthDirAdj() - textPositions.get(textPositions.size() - 1).getWidthDirAdj() - 3;
posYInit = textPositions.get(0).getYDirAdj() - height + 2;
posYEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height + 2;
} else if (textPositions.get(0).getRotation() == 180 && textPositions.get(0).getDir() == 270.0f) {
posYInit = textPositions.get(0).getPageHeight() - getX1();
posYEnd = textPositions.get(0).getPageHeight() - getX2() - textPositions.get(0)
.getWidthDirAdj() - textPositions.get(textPositions.size() - 1).getWidthDirAdj();
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + height;
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 0.0f) {
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = posYInit + 1;
posXInit = textPositions.get(0).getXDirAdj();
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + 0.1f;
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 90.0f) {
posYInit = getX1();
posYEnd = getX2() - height;
posXInit = textPositions.get(0).getYDirAdj() + 2;
posXEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height;
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 180.0f) {
posXInit = textPositions.get(0).getPageWidth() - getX1() + 1;
posXEnd = textPositions.get(0).getPageWidth() - getX2() - 4;
posYInit = textPositions.get(0).getYDirAdj() - height + 2;
posYEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height + 2;
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 270.0f) {
posYInit = textPositions.get(0).getPageHeight() - getX1();
posYEnd = textPositions.get(0).getPageHeight() - getX2() - height;
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + height;
Point2D bottomLeft = new Point2D.Double(firstTextPos.getXDirAdj(), firstTextPos.getYDirAdj() - HEIGHT_PADDING);
Point2D topRight = new Point2D.Double(lastTextPos.getXDirAdj() + lastTextPos.getWidthDirAdj(), lastTextPos.getYDirAdj() + textHeight + HEIGHT_PADDING);
AffineTransform transform = new AffineTransform();
if (dir == TextDirection.ZERO || dir == TextDirection.HALF_CIRCLE) {
transform.rotate(dir.getRadians(), pageWidth / 2f, pageHeight / 2f);
transform.translate(0f, pageHeight + textHeight);
transform.scale(1., -1.);
} else if (dir == TextDirection.QUARTER_CIRCLE) {
transform.rotate(dir.getRadians(), pageWidth / 2f, pageWidth / 2f);
transform.translate(0f, pageWidth + textHeight);
transform.scale(1., -1.);
} else {
// page rotation = 0 and text direction = 0
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + 1;
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
transform.rotate(dir.getRadians(), pageHeight / 2f, pageHeight / 2f);
transform.translate(0f, pageWidth + textHeight);
transform.scale(1., -1.);
}
var rectangle = new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
log.debug("Rectangle: {}", rectangle);
return rectangle;
bottomLeft = transform.transform(bottomLeft, null);
topRight = transform.transform(topRight, null);
return new Rectangle( //
new Point((float) bottomLeft.getX(), (float) bottomLeft.getY()),
(float) (topRight.getX() - bottomLeft.getX()),
(float) (topRight.getY() - bottomLeft.getY()),
page);
}
}
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.server.queue;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_QUEUE;
import org.springframework.amqp.core.Message;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
@@ -16,10 +17,12 @@ public class MessageReceiver {
private final RedactionMessageReceiver redactionMessageReceiver;
@RabbitHandler
@RabbitListener(queues = REDACTION_QUEUE)
public void receiveAnalyzeRequest(String in) throws JsonProcessingException {
redactionMessageReceiver.receiveAnalyzeRequest(in, false);
public void receiveAnalyzeRequest(Message message) {
redactionMessageReceiver.receiveAnalyzeRequest(message, false);
}
}
@@ -17,16 +17,19 @@ public class MessagingConfiguration {
public static final String REDACTION_PRIORITY_QUEUE = "redactionPriorityQueue";
@Bean
@ConditionalOnProperty(prefix = "redaction-service", name = "priorityMode", havingValue = "false")
public MessageReceiver messageReceiver(RedactionMessageReceiver redactionMessageReceiver){
public MessageReceiver messageReceiver(RedactionMessageReceiver redactionMessageReceiver) {
return new MessageReceiver(redactionMessageReceiver);
}
@Bean
@ConditionalOnProperty(prefix = "redaction-service", name = "priorityMode", havingValue = "true")
public PriorityMessageReceiver priorityMessageReceiver(RedactionMessageReceiver redactionMessageReceiver){
public PriorityMessageReceiver priorityMessageReceiver(RedactionMessageReceiver redactionMessageReceiver) {
return new PriorityMessageReceiver(redactionMessageReceiver);
}
@@ -34,11 +37,7 @@ public class MessagingConfiguration {
@Bean
public Queue redactionQueue() {
return QueueBuilder.durable(REDACTION_QUEUE)
.withArgument("x-dead-letter-exchange", "")
.withArgument("x-dead-letter-routing-key", REDACTION_DQL)
.maxPriority(2)
.build();
return QueueBuilder.durable(REDACTION_QUEUE).withArgument("x-dead-letter-exchange", "").withArgument("x-dead-letter-routing-key", REDACTION_DQL).maxPriority(2).build();
}
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.server.queue;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_PRIORITY_QUEUE;
import org.springframework.amqp.core.Message;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
@@ -19,9 +20,9 @@ public class PriorityMessageReceiver {
@RabbitHandler
@RabbitListener(queues = REDACTION_PRIORITY_QUEUE)
public void receiveAnalyzeRequest(String in) throws JsonProcessingException {
public void receiveAnalyzeRequest(Message message) {
redactionMessageReceiver.receiveAnalyzeRequest(in, true);
redactionMessageReceiver.receiveAnalyzeRequest(message, true);
}
}

Some files were not shown because too many files have changed in this diff Show More