Compare commits

...
Author SHA1 Message Date
Philipp Schramm 02d9ebc24c Pull request #360: RED-3568: Do not ignore textblocks if only rotated pages are in the document
Merge in RED/redaction-service from RED-3568 to master

* commit 'ae3e4570c15dbc2c46373ce68d55cec760e7da6b':
  RED-3568: Do not ignore textblocks if only rotated pages are in the document
2022-03-29 14:00:20 +02:00
deiflaender ae3e4570c1 RED-3568: Do not ignore textblocks if only rotated pages are in the document 2022-03-29 13:54:23 +02:00
Dominique Eiflaender a975596446 Pull request #359: RED-3699 master
Merge in RED/redaction-service from RED-3699-master to master

* commit 'a5b96f148950e28618028daef1e91aecdadc89e3':
  Fixed unknown type for manual type
  Fixed bug in getColors
  RED-3699:  Do not call updateDictionary on merge RedactionLog
2022-03-25 09:20:52 +01:00
deiflaender a5b96f1489 Fixed unknown type for manual type 2022-03-25 09:12:32 +01:00
deiflaender 4814ce3650 Fixed bug in getColors 2022-03-25 09:12:21 +01:00
deiflaender 7fccdb32c5 RED-3699: Do not call updateDictionary on merge RedactionLog 2022-03-25 09:12:00 +01:00
Timo Bejan 1d389b78fd Pull request #357: processed date fix
Merge in RED/redaction-service from processed-date-fix-3.3.x to master

* commit 'b4916b65b661c80f0924c163b8795205b9c8e6c3':
  processed date fix
2022-03-24 11:52:29 +01:00
Timo Bejan b4916b65b6 processed date fix 2022-03-24 10:16:49 +02:00
Dominique Eiflaender e1db383463 Pull request #356: RED-2836: Fixed calculating dictionary increments with false_positive and false_recommendation
Merge in RED/redaction-service from RED-2836-3 to master

* commit '4d2a7d454d916e7eee65f7cc155323924c2cb229':
  RED-2836: Fixed calculating dictionary increments with false_positive and false_recommendation
2022-03-22 10:31:05 +01:00
deiflaender 4d2a7d454d RED-2836: Fixed calculating dictionary increments with false_positive and false_recommendation 2022-03-22 10:24:26 +01:00
Dominique Eiflaender 3758720683 Pull request #355: Renamed result field in research services to data
Merge in RED/redaction-service from ResearchRename to master

* commit '22bb0b44db7a22f8088512570d4b2be6e742a887':
  Renamed result field in research services to data
2022-03-21 16:02:45 +01:00
deiflaender 22bb0b44db Renamed result field in research services to data 2022-03-21 15:54:37 +01:00
Dominique Eiflaender e93f5d147f Pull request #354: RED-2836: Enabled false positives per dictionary
Merge in RED/redaction-service from RED-2836 to master

* commit '88e6ac3d2291ced8bb2229c3e7daeb8d78d6da93':
  RED-2836: Enabled false positives per dictionary
2022-03-21 12:54:48 +01:00
deiflaender 88e6ac3d22 RED-2836: Enabled false positives per dictionary 2022-03-21 12:47:27 +01:00
Timo Bejan 17b885d901 Pull request #353: pending add to dict
Merge in RED/redaction-service from pending-master-fix to master

* commit '50d9f4fec22768310232988ee01f9e9aa3ba9dc8':
  pending add to dict
2022-03-21 11:49:16 +01:00
Timo Bejan 50d9f4fec2 pending add to dict 2022-03-21 11:19:26 +02:00
Philipp Schramm 1bcf0a3f07 Pull request #351: RED-3553 If image overlaps entity and the entity will be removed
Merge in RED/redaction-service from RED-3553 to master

* commit '1f6a9aa14deb7ec65fee03e28758f8e3157dbd5f':
  RED-3553 If image overlaps entity and the entity will be removed
2022-03-17 14:17:16 +01:00
Philipp Schramm 1f6a9aa14d RED-3553 If image overlaps entity and the entity will be removed 2022-03-17 14:04:59 +01:00
Dominique Eiflaender c8cf0078eb Pull request #352: PortsFrom3.2.x
Merge in RED/redaction-service from portsFrom3.2.x to master

* commit 'b13a695c64280a97ece198a18605c0f1b5cf753f':
  RED-3611: Added priority mode that only listens to priority queue
  RED-3620: Fixed remove from dictionary
2022-03-17 11:16:03 +01:00
deiflaender b13a695c64 RED-3611: Added priority mode that only listens to priority queue 2022-03-17 10:50:48 +01:00
deiflaender 32a907f697 RED-3620: Fixed remove from dictionary 2022-03-17 10:49:01 +01:00
Timo Bejan 1d1de6caba Pull request #350: Removed processed date filter for add to dictionary entries
Merge in RED/redaction-service from RED-3538 to master

* commit '7b2295393c1c2dcbcecbb11c64929517cb13a6de':
  Removed processed date filter for add to dictionary entries
2022-03-16 15:46:13 +01:00
Timo Bejan 7b2295393c Removed processed date filter for add to dictionary entries 2022-03-16 16:29:30 +02:00
Timo Bejan d0ba3f51c1 Pull request #347: From version master
Merge in RED/redaction-service from from-version-master to master

* commit 'c432c56a05ce58d711286bbbcc636ff1445c652d':
  Updated Persistence Service Version
  Release 3.80.x branch creation + incremental entry loading based on version for dictionaries
2022-03-14 16:36:14 +01:00
Timo Bejan c432c56a05 Updated Persistence Service Version 2022-03-14 17:29:04 +02:00
Timo Bejan e73da002d5 Release 3.80.x branch creation + incremental entry loading based on version for dictionaries 2022-03-14 17:27:52 +02:00
Philipp Schramm 0ceb8ae266 Pull request #346: RED-3403 Extended Section with negated Methods
Merge in RED/redaction-service from RED-3403 to master

* commit '97fe7383cd9737f25d22903a6124dfe8361f7a84':
  RED-3403 Extended Section with negated Methods
2022-02-28 14:51:21 +01:00
Philipp Schramm 97fe7383cd RED-3403 Extended Section with negated Methods 2022-02-28 14:40:40 +01:00
Dominique Eiflaender 7d132cbb88 Pull request #345: Fixed aws s3 config
Merge in RED/redaction-service from s3FixConf to master

* commit 'd18b5dfd138595f0cbf7cc2a99dc708994141425':
  Fixed aws s3 config
2022-02-25 13:04:36 +01:00
deiflaender d18b5dfd13 Fixed aws s3 config 2022-02-25 12:58:09 +01:00
Dominique Eiflaender e2451a08f4 Pull request #344: RED-3282: Do not add overlapping ai recommendactions
Merge in RED/redaction-service from RED-3282 to master

* commit '537ccb83d799cef0eb437a6bbe1b78da661b2b6b':
  RED-3282: Do not add overlapping ai recommendactions
2022-02-25 09:37:15 +01:00
deiflaender 537ccb83d7 RED-3282: Do not add overlapping ai recommendactions 2022-02-25 09:29:08 +01:00
Ali Oezyetimoglu fd027efa5c Pull request #343: RED-3407: Remove obsolete recommendation dictionaries
Merge in RED/redaction-service from RED-3407-rs1 to master

* commit 'c3984e054482acab3acde740b8dce437d97e9ffb':
  RED-3407: Remove obsolete recommendation dictionaries
2022-02-23 14:59:53 +01:00
aoezyetimoglu c3984e0544 RED-3407: Remove obsolete recommendation dictionaries 2022-02-23 14:29:01 +01:00
Timo Bejan 77feedcc8e Pull request #342: added imported flag
Merge in RED/redaction-service from imported-redaction to master

* commit 'de92117b35fcc2f3979450027acf1f7edf04befa':
  added imported flag
2022-02-11 11:41:32 +01:00
Timo Bejan de92117b35 added imported flag 2022-02-11 11:33:34 +02:00
Philipp Schramm 143f4cfb0d Pull request #341: RED-3350 Communication with entity recognition via queues
Merge in RED/redaction-service from feature/RED-3350 to master

* commit '68727178e79224b7ca7f9f293e48ca21b4ce2f91':
  RED-3350 Communication with entity recognition via queues
2022-02-09 15:24:47 +01:00
Philipp Schramm 68727178e7 RED-3350 Communication with entity recognition via queues 2022-02-09 15:19:55 +01:00
Dominique Eiflaender 90fe282313 Pull request #339: RED-3327: Ignore recommendations if touches other type
Merge in RED/redaction-service from RED-3327-master to master

* commit '8689125ed9b09f73781d3babfd65537c0b1f4afe':
  RED-3327: Ignore recommendations if touches other type
2022-02-01 11:19:26 +01:00
deiflaender 8689125ed9 RED-3327: Ignore recommendations if touches other type 2022-02-01 11:06:32 +01:00
Dominique Eiflaender e68c6798be Pull request #337: RED-3243: Improved imported redactions logic
Merge in RED/redaction-service from RED-3243-8 to master

* commit '58b1db66cc2cc67b3b4e26295478824e5052d6a3':
  RED-3243: Improved imported redactions logic
2022-01-31 14:47:54 +01:00
deiflaender 58b1db66cc RED-3243: Improved imported redactions logic 2022-01-31 14:42:49 +01:00
deiflaender 0072d38d68 Use base image to speedup build 2022-01-31 12:05:10 +01:00
deiflaender 25a409d3a5 RED-3243: Added ImportedRedaction model classes 2022-01-31 11:11:32 +01:00
Timo Bejan ca1da20ba3 Pull request #335: immutable field fix
Merge in RED/redaction-service from RED-3242 to master

* commit '3c691a1afc9ce67845aad5d15bbc99a0113f6a6b':
  immutable field fix
2022-01-28 10:57:04 +01:00
Timo Bejan 3c691a1afc immutable field fix 2022-01-28 11:29:43 +02:00
Timo Bejan 6813fbc3cb Pull request #333: since no more manual changes are fields of redactionLog, we can safely always merge, even legacy ones
Merge in RED/redaction-service from RED-3242 to master

* commit 'c9a75152fb9b4e02b3421dd16ae54056889b0aa9':
  since no more manual changes are fields of redactionLog, we can safely always merge, even legacy ones
2022-01-27 15:59:39 +01:00
Dominique Eiflaender a8a6fef07a Pull request #334: RED-3239: Protoyped logic for imported redactions
Merge in RED/redaction-service from RED-3239 to master

* commit '26661fa87beeb20b32f2be5a8b1b025e8d2d29fd':
  RED-3239: Protoyped logic for imported redactions
2022-01-27 15:31:16 +01:00
deiflaender 26661fa87b RED-3239: Protoyped logic for imported redactions 2022-01-27 15:21:33 +01:00
Timo Bejan c9a75152fb since no more manual changes are fields of redactionLog, we can safely always merge, even legacy ones 2022-01-27 15:30:43 +02:00
Timo Bejan 0da2a1ab26 Pull request #332: RED-3242 - excluded from automatic analysis - reworked redactionLog architecture
Merge in RED/redaction-service from RED-3242 to master

* commit 'd92a723fea21280d7bf482c3d74c7b177dbaa030':
  RED-3242 - excluded from automatic analysis - reworked redactionLog architecture
2022-01-27 14:06:35 +01:00
Timo Bejan d92a723fea RED-3242 - excluded from automatic analysis - reworked redactionLog architecture 2022-01-27 14:58:14 +02:00
Dominique Eiflaender 3109e3e4f0 Pull request #331: RED-3133: Ignore min length for values from ai
Merge in RED/redaction-service from RED-3133-Fix4 to master

* commit 'a1544588e8a3cfa0479d5d31d82463c5434674a1':
  RED-3133: Ignore min length for values from ai
2022-01-26 16:32:23 +01:00
deiflaender a1544588e8 RED-3133: Ignore min length for values from ai 2022-01-26 16:18:41 +01:00
Dominique Eiflaender 8ba2bd776d Pull request #330: RED-3133: Fixed wrong ai entries again
Merge in RED/redaction-service from RED-3133-Fix2 to master

* commit '8ac7c309d981c8899eca40872a49198284d15944':
  RED-3133: Fixed wrong ai entries again
2022-01-26 15:42:13 +01:00
deiflaender 8ac7c309d9 RED-3133: Fixed wrong ai entries again 2022-01-26 15:31:13 +01:00
Dominique Eiflaender b8f7be28e7 Pull request #329: RED-3133: Fixed wrong ai entries
Merge in RED/redaction-service from RED-3133-Fix to master

* commit '921ae1839cb424a4d82afd88401fddc9cf855f8c':
  RED-3133: Fixed wrong ai entries
2022-01-26 15:09:04 +01:00
deiflaender 921ae1839c RED-3133: Fixed wrong ai entries 2022-01-26 14:43:54 +01:00
Dominique Eiflaender 38a386bc87 Pull request #328: RED-3133: Fixed bug in address combination rule
Merge in RED/redaction-service from RED-3133-4 to master

* commit 'c21cf3dd5cdf70dfb838bb18f78a7c32715f8db8':
  RED-3133: Fixed bug in address combination rule
2022-01-26 13:09:13 +01:00
deiflaender c21cf3dd5c RED-3133: Fixed bug in address combination rule 2022-01-26 12:57:02 +01:00
Dominique Eiflaender 29ad27dea6 Pull request #326: RED-3243: Fixed bug in combining address parts
Merge in RED/redaction-service from RED-3243 to master

* commit '6a78d798912afd4d625455d62d2ec416a311873b':
  RED-3243: Fixed bug in combining address parts
2022-01-24 12:53:40 +01:00
deiflaender 6a78d79891 RED-3243: Fixed bug in combining address parts 2022-01-24 12:35:38 +01:00
Dominique Eiflaender 726c272e61 Pull request #325: Improved rule to combine AI matches
Merge in RED/redaction-service from RED-3133-2 to master

* commit 'f32d9a526787a6a74179a0e8f5589763a2080fca':
  Improved rule to combine AI matches
2022-01-21 13:15:39 +01:00
deiflaender f32d9a5267 Improved rule to combine AI matches 2022-01-21 12:53:33 +01:00
Dominique Eiflaender c4c96c1712 Pull request #324: RED-3133: Add and combine AI Entities in rules
Merge in RED/redaction-service from RED-3133 to master

* commit '6e720a15c2b9c9d46b52b94f63f0af157ac5d81a':
  RED-3133: Add and combine AI Entities in rules
2022-01-20 12:28:17 +01:00
deiflaender 6e720a15c2 RED-3133: Add and combine AI Entities in rules 2022-01-20 12:11:45 +01:00
Dominique Eiflaender 3687a7e5a9 Pull request #323: RED-3190: Ignore if surrounding text can not be calculate because of different parsing order from ui and backend
Merge in RED/redaction-service from RED-3190-master to master

* commit '6a31ae642d34c61865b55b6b138ef4406c27ab63':
  RED-3190: Ignore if surrounding text can not be calculate because of different parsing order from ui and backend
2022-01-14 12:10:42 +01:00
deiflaender 6a31ae642d RED-3190: Ignore if surrounding text can not be calculate because of different parsing order from ui and backend 2022-01-14 12:04:50 +01:00
Ali Oezyetimoglu e01d49754d Pull request #321: RED-3183: Upgrade log4j to 2.17.1
Merge in RED/redaction-service from RED-3183-rs1 to master

* commit '5de4702d5ac0c05ff23ef89396028be121474600':
  RED-3183: Upgrade log4j to 2.17.1
2022-01-13 15:38:28 +01:00
aoezyetimoglu 5de4702d5a RED-3183: Upgrade log4j to 2.17.1 2022-01-13 15:24:31 +01:00
Dominique Eiflaender c621eb9f04 Pull request #320: RED-3173: Fixed wrong skipped redaction afer remove and readd to dictionary
Merge in RED/redaction-service from RED-3173-master to master

* commit 'eb9fe0e3ddfc9b86cb3b0664e8bec5ef5b4d34c7':
  RED-3173: Fixed wrong skipped redaction afer remove and readd to dictionary
2022-01-13 11:20:56 +01:00
deiflaender eb9fe0e3dd RED-3173: Fixed wrong skipped redaction afer remove and readd to dictionary 2022-01-13 11:14:45 +01:00
Timo Bejan f05977b7ae Pull request #318: analysis version in redactionlog and changes
Merge in RED/redaction-service from analysis-number-in-redaction-log-m to master

* commit '1cad386db83a2d2057701989c0c179b770248f5e':
  analysis version in redactionlog and changes
2022-01-13 09:23:34 +01:00
Timo Bejan 1cad386db8 analysis version in redactionlog and changes 2022-01-13 10:08:34 +02:00
cschabert 21a8af5ca5 add build before sonar job 2022-01-12 13:32:47 +01:00
Dominique Eiflaender 9c9583067f Pull request #316: RED-2440: Integrated image-service-v2
Merge in RED/redaction-service from RED-2440 to master

* commit 'dee1aa1f01085e585bb3126b7883bd3d01090f17':
  RED-2440: Integrated image-service-v2
2021-12-22 10:40:04 +01:00
deiflaender dee1aa1f01 RED-2440: Integrated image-service-v2 2021-12-22 10:35:13 +01:00
Ali Oezyetimoglu db16f8c1da Pull request #314: RED-3127: Upgrade log4j to 2.17.0
Merge in RED/redaction-service from RED-3127-rs1 to master

* commit '6911c5cdd5a0e559f4abfbd083b29aca9fbae9d9':
  RED-3127: Upgrade log4j to 2.17.0
2021-12-21 12:17:40 +01:00
aoezyetimoglu 6911c5cdd5 RED-3127: Upgrade log4j to 2.17.0 2021-12-21 12:00:24 +01:00
Timo Bejan 37c3814ebd Pull request #313: condition fix
Merge in RED/redaction-service from redaction-log-merge-service-test to master

* commit 'e8ae8728f3abd82cacd641144406f9047ca9c262':
  condition fix
2021-12-21 11:07:37 +01:00
Timo Bejan e8ae8728f3 condition fix 2021-12-21 11:18:37 +02:00
Timo Bejan b0415adaf4 Pull request #312: Redaction log merge service test
Merge in RED/redaction-service from redaction-log-merge-service-test to master

* commit '9e64773d0db262481f8552678a6e6b86d81d93c4':
  pmd
  skip remove from dict entries from merge
2021-12-21 10:17:54 +01:00
Timo Bejan 9e64773d0d pmd 2021-12-21 11:08:15 +02:00
Timo Bejan 3145195d63 skip remove from dict entries from merge 2021-12-21 11:05:47 +02:00
Dominique Eiflaender 88b1274cd3 Pull request #311: RED-3117: Enbled to ignore invalid entries at addToDictionary for automated pushing entries in analysis
Merge in RED/redaction-service from RED-3117 to master

* commit '0296745efc0e4786386811d0037e2cae6ae511c4':
  RED-3117: Enbled to ignore invalid entries at addToDictionary for automated pushing entries in analysis
2021-12-20 11:58:34 +01:00
deiflaender 0296745efc RED-3117: Enbled to ignore invalid entries at addToDictionary for automated pushing entries in analysis 2021-12-20 11:53:36 +01:00
Dominique Eiflaender 60a0adb843 Pull request #310: RED-3114: Fixed unreliable redactions on rotated pages
Merge in RED/redaction-service from RED-3114 to master

* commit 'f1071de3a9ea9a3b34798f21ee9589262ef4f32b':
  RED-3114: Fixed unreliable redactions on rotated pages
2021-12-17 16:00:18 +01:00
deiflaender f1071de3a9 RED-3114: Fixed unreliable redactions on rotated pages 2021-12-17 15:55:09 +01:00
Ali Oezyetimoglu 681d17ad95 Pull request #309: RED-3116: Fix missing relativePath in mvn projects to leads to non working sync on lastest intellij
Merge in RED/redaction-service from RED-3116-rs1 to master

* commit '8deb24a5342c1c7a34fa6235e3d68f8161d884e4':
  RED-3116: Fix missing relativePath in mvn projects to leads to non working sync on lastest intellij
2021-12-17 13:53:14 +01:00
aoezyetimoglu 8deb24a534 RED-3116: Fix missing relativePath in mvn projects to leads to non working sync on lastest intellij 2021-12-17 13:10:43 +01:00
Ali Oezyetimoglu 98443f50ec Pull request #308: RED-3107 rs1
Merge in RED/redaction-service from RED-3107-rs1 to master

* commit '7d75961df45de93cfb2a708b640703dbd5693f40':
  RED-3107: Missing headline for tables appears as inconsistent text in report
  RED-3106: Upgrade log4j to 2.16.0
2021-12-17 12:46:53 +01:00
aoezyetimoglu 7d75961df4 RED-3107: Missing headline for tables appears as inconsistent text in report 2021-12-17 12:00:14 +01:00
aoezyetimoglu 3c1f4b4853 Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service 2021-12-17 11:52:52 +01:00
Dominique Eiflaender c9e74f4ff9 Pull request #307: RED-3105: Fixed change computation
Merge in RED/redaction-service from RED-3105 to master

* commit '19dcf0a67434442b2dd578523bbc5efa49c644e8':
  RED-3105: Fixed change computation
2021-12-16 14:22:55 +01:00
deiflaender 19dcf0a674 RED-3105: Fixed change computation 2021-12-16 14:17:04 +01:00
Philipp Schramm c46c41b790 Pull request #306: RED-3067 Bugfix to show redactions on rotated pages correctly
Merge in RED/redaction-service from RED-3067 to master

* commit 'f49782438461ec34e6e8c0ccf88b3017755e593d':
  RED-3067 Bugfix to show redactions on rotated pages correctly and fixed getting tmp directory
  RED-3067 Bugfix to show redactions on rotated pages correctly
2021-12-16 14:14:40 +01:00
Philipp Schramm f497824384 RED-3067 Bugfix to show redactions on rotated pages correctly and fixed getting tmp directory 2021-12-16 14:08:05 +01:00
Philipp Schramm f7fbe8f0f6 RED-3067 Bugfix to show redactions on rotated pages correctly 2021-12-16 12:43:17 +01:00
Ali Oezyetimoglu e8dce509f3 Pull request #303: RED-3106: Upgrade log4j to 2.16.0
Merge in RED/redaction-service from RED-3106-rs1 to master

* commit '9346cc701e1601b6c9d44e41de9fc193fbc81b1e':
  RED-3106: Upgrade log4j to 2.16.0
2021-12-16 09:57:54 +01:00
aoezyetimoglu b125de2b55 RED-3106: Upgrade log4j to 2.16.0 2021-12-16 09:52:21 +01:00
aoezyetimoglu 9346cc701e RED-3106: Upgrade log4j to 2.16.0 2021-12-16 09:44:04 +01:00
Dominique Eiflaender 1987e7b430 Pull request #302: RED-3025: Ported code to ignore hints
Merge in RED/redaction-service from RED-3025 to master

* commit '010a3ee6ab7f06297c2c37691bcac2f7019fb3a0':
  RED-3025: Ported code to ignore hints
2021-12-15 13:26:48 +01:00
deiflaender 010a3ee6ab RED-3025: Ported code to ignore hints 2021-12-15 13:22:28 +01:00
Philipp Schramm 358f8814cd Pull request #300: RED-3032 Bugfix: Throw NotFoundException in case section grid was not found
Merge in RED/redaction-service from bugfix/RED-3032 to master

* commit '9fb4024fb7dfe7ae3f92f5c17decf15fbac58451':
  RED-3032 Bugfix: Throw NotFoundException in case section grid was not found
2021-12-15 09:05:10 +01:00
Philipp Schramm 9fb4024fb7 RED-3032 Bugfix: Throw NotFoundException in case section grid was not found 2021-12-14 17:02:24 +01:00
Ali Oezyetimoglu a155f18adc Pull request #299: RED-3061 rs3
Merge in RED/redaction-service from RED-3061-rs3 to master

* commit '1ccf0ab493fcb367a071641e2fea050cf6ad26cf':
  RED-3061: Logic for value matcher for initials expansion rule is inverted
  RED-3061: Logic for value matcher for initials expansion rule is inverted
2021-12-14 12:35:45 +01:00
aoezyetimoglu 1ccf0ab493 RED-3061: Logic for value matcher for initials expansion rule is inverted 2021-12-14 12:25:21 +01:00
aoezyetimoglu 1b9251b0a2 RED-3061: Logic for value matcher for initials expansion rule is inverted 2021-12-14 11:10:59 +01:00
Dominique Eiflaender 6bfa54583f Pull request #297: RED-2439: Fixed missing MessageType in AnalyseResult
Merge in RED/redaction-service from RED-2439-4 to master

* commit '83a757976adf26cfeae8fb682593ce1d5f606dfc':
  RED-2439: Fixed missing MessageType in AnalyseResult
2021-12-13 13:55:52 +01:00
deiflaender 83a757976a RED-2439: Fixed missing MessageType in AnalyseResult 2021-12-13 13:51:33 +01:00
Dominique Eiflaender bd5c997572 Pull request #296: RED-2439: Enabled to find surrounding text for manual redactions via queue
Merge in RED/redaction-service from RED-2439-3 to master

* commit 'e4cf3bfa602014cd8a2f2162199557880600e93f':
  RED-2439: Enabled to find surrounding text for manual redactions via queue
2021-12-13 12:32:20 +01:00
deiflaender e4cf3bfa60 RED-2439: Enabled to find surrounding text for manual redactions via queue 2021-12-13 12:24:26 +01:00
Ali Oezyetimoglu 3f153c1df5 Pull request #294: RED-3075: Upgraded log4j to 2.15.0
Merge in RED/redaction-service from RED-3075-rs2 to master

* commit '142c81d27d9e5e67532b3d8795c2d4c559342d37':
  RED-3075: Upgraded log4j to 2.15.0
2021-12-13 11:03:56 +01:00
aoezyetimoglu 142c81d27d RED-3075: Upgraded log4j to 2.15.0 2021-12-13 10:59:16 +01:00
Dominique Eiflaender a2c57b94a2 Pull request #292: RED-3059: Fixed inital expansion overlaps
Merge in RED/redaction-service from RED-3059-port to master

* commit '0e21e5b4d988d43040573a4fc71547829c815ec0':
  RED-3059: Fixed inital expansion overlaps
2021-12-10 11:37:24 +01:00
deiflaender 0e21e5b4d9 RED-3059: Fixed inital expansion overlaps 2021-12-10 11:32:53 +01:00
Dominique Eiflaender 29a66ee5cf Pull request #290: RED-2439: Added endpoint to calculate textBefore and textAfter for manual redactions
Merge in RED/redaction-service from RED-2439 to master

* commit '6950c00711dc113bb012a9cd3cc26c5c276409a0':
  RED-2439: Added endpoint to calculate textBefore and textAfter for manual redactions
2021-12-09 16:19:24 +01:00
deiflaender 6950c00711 RED-2439: Added endpoint to calculate textBefore and textAfter for manual redactions 2021-12-09 16:12:02 +01:00
Philipp Schramm fa4b2263bc Pull request #288: RED-2795 Bugfix with inconsistent types for recategorizing Images in specific cases
Merge in RED/redaction-service from bugfix/RED-2795 to master

* commit '415df2b3658b4525120a15fb0f341164a73bb3d4':
  RED-2795 Bugfix with inconsistent types for recategorizing Images in specific cases
2021-12-07 11:15:30 +01:00
Philipp Schramm 415df2b365 RED-2795 Bugfix with inconsistent types for recategorizing Images in specific cases 2021-12-07 11:10:53 +01:00
Philipp Schramm cda8e51d9d Pull request #286: Bugfix/RED-2845 3.0
Merge in RED/redaction-service from bugfix/RED-2845_3.0 to master

* commit '1c982a40422a16e9caa8d264e4221524c7e1c92b':
  RED-2845 Bugfix: Avoid the expansion if it would result in a redaction overlap and bugfix in RegExp in drools
2021-12-02 14:07:45 +01:00
Philipp Schramm 1c982a4042 RED-2845 Bugfix: Avoid the expansion if it would result in a redaction overlap and bugfix in RegExp in drools 2021-12-02 14:02:31 +01:00
Dominique Eiflaender 456bd7e093 Pull request #285: RED-2862: Added possibility to ignore dictionaries by rule
Merge in RED/redaction-service from RED-2862-3.0.0 to master

* commit 'f37f3b2dcf10102d03bbf3fb3233a801537d45ad':
  RED-2862: Added possibility to ignore dictionaries by rule
2021-12-01 12:46:10 +01:00
deiflaender f37f3b2dcf RED-2862: Added possibility to ignore dictionaries by rule 2021-12-01 12:40:59 +01:00
Dominique Eiflaender 809c563934 Pull request #283: Hotfix for duplicate entries in redactionLog
Merge in RED/redaction-service from DuplicateIdsMaster to master

* commit '46d14cfc6441588ee57e19d4780ffebcf70e3367':
  Hotfix for duplicate entries in redactionLog
2021-11-30 16:36:03 +01:00
deiflaender 46d14cfc64 Hotfix for duplicate entries in redactionLog 2021-11-30 16:27:59 +01:00
Dominique Eiflaender 459c977f7e Pull request #281: RED-2861: Ported changes for rectangle redaction improvements
Merge in RED/redaction-service from RED-2861-port to master

* commit '7b1c9c97e672d78e0fbf7f05138db853a11e7d04':
  RED-2861: Ported changes for rectangle redaction improvements
2021-11-30 12:52:26 +01:00
deiflaender 7b1c9c97e6 RED-2861: Ported changes for rectangle redaction improvements 2021-11-30 12:47:22 +01:00
Philipp Schramm 8d881d4ca6 Pull request #280: RED-2756 Bugfix with redactions are not continuous and added some entries to dictionary
Merge in RED/redaction-service from bugfix/RED-2756 to master

* commit '35a4d004fda0404e9b19e7112d25627cbde2a932':
  RED-2756 Bugfix with redactions are not continuous and added some entries to dictionary
2021-11-30 11:15:32 +01:00
Philipp Schramm 35a4d004fd RED-2756 Bugfix with redactions are not continuous and added some entries to dictionary 2021-11-30 11:05:06 +01:00
Ali Oezyetimoglu ad80d4247c Pull request #278: RED-2841: INC6207970 Rule for initials expansion should be applied only to dictionary entries without whitespaces
Merge in RED/redaction-service from RED-2841-rs1 to master

* commit 'd1317c5bd4f9522eecd78775656a0f7b5a74c3d7':
  RED-2841: INC6207970 Rule for initials expansion should be applied only to dictionary entries without whitespaces
2021-11-29 15:40:51 +01:00
aoezyetimoglu d1317c5bd4 RED-2841: INC6207970 Rule for initials expansion should be applied only to dictionary entries without whitespaces 2021-11-29 12:38:53 +01:00
Philipp Schramm f5817204bf Pull request #277: Bugfix/RED-2756
Merge in RED/redaction-service from bugfix/RED-2756 to master

* commit '9d0fafd63ddb23fd8ee2f154e02df88687e08f96':
  RED-2756 Bugfix with redactions are not continuous
  RED-2756 Bugfix for 'Redaction is not continuous', compare line height and y position instead of rounding y values
2021-11-29 10:36:05 +01:00
Philipp Schramm 9d0fafd63d RED-2756 Bugfix with redactions are not continuous 2021-11-29 10:31:32 +01:00
Philipp Schramm 6230f42fcf RED-2756 Bugfix for 'Redaction is not continuous', compare line height and y position instead of rounding y values 2021-11-25 16:40:26 +01:00
Timo Bejan 571035dda1 Pull request #275: redaction-log info for RED-2659
Merge in RED/redaction-service from section-info to master

* commit '892fa6ade0c9f260183686535612e7423ed2ec09':
  redaction-log info for RED-2659
2021-11-22 13:49:44 +01:00
Timo Bejan 892fa6ade0 redaction-log info for RED-2659 2021-11-22 14:45:02 +02:00
Timo Bejan ab9dcbbafb Pull request #274: section info port
Merge in RED/redaction-service from section-info to master

* commit '980d14b6049c0e7f52c2cc4fffbffa7d5bf54b08':
  section info port
2021-11-22 08:51:10 +01:00
Timo Bejan 980d14b604 section info port 2021-11-22 09:46:12 +02:00
Timo Bejan 1cefa3dc05 Pull request #273: 2 new falgs for UI indicators
Merge in RED/redaction-service from redaction-log-updates to master

* commit 'f80eb6d791f35aa96249ef783e56164db85c8278':
  2 new falgs for UI indicators
2021-11-18 10:24:39 +01:00
Timo Bejan f80eb6d791 2 new falgs for UI indicators 2021-11-18 11:18:34 +02:00
Philipp Schramm e42d5180ec Pull request #272: RED-2430 do not exclude manual entries from excluded pages
Merge in RED/redaction-service from feature/RED-2430 to master

* commit '3ae8b2cd666eee2fff10e2254cd829084e503893':
  RED-2430 do not exclude manual entries from excluded pages
2021-11-16 08:53:19 +01:00
Philipp Schramm 3ae8b2cd66 RED-2430 do not exclude manual entries from excluded pages 2021-11-15 16:39:53 +01:00
Philipp Schramm a6a1d51c17 Pull request #271: RED-2633 Added section to manual redactions, updated version from persistence-service and extracted versions in pom.xm
Merge in RED/redaction-service from RED-2633 to master

* commit '6e713c4fcc010e6461e403256c4926974cec2cf8':
  RED-2633 Added section to manual redactions, updated version from persistence-service and extracted versions in pom.xm
2021-11-15 14:56:52 +01:00
Philipp Schramm 6e713c4fcc RED-2633 Added section to manual redactions, updated version from persistence-service and extracted versions in pom.xm 2021-11-15 14:36:35 +01:00
Dominique Eiflaender 2c6957360b Pull request #269: RED-2742: Fixed annotation positions for rotation = 0 and direction = 90
Merge in RED/redaction-service from RED-2742 to master

* commit 'a91877769370c804b7d9d9fa4d4e9e16752598bd':
  RED-2742: Fixed annotation positions for rotation = 0 and direction = 90
2021-11-12 10:16:52 +01:00
deiflaender a918777693 RED-2742: Fixed annotation positions for rotation = 0 and direction = 90 2021-11-12 10:06:27 +01:00
Dominique Eiflaender d8662e8e20 Pull request #266: RED-2624: Fixed footnote is recognized as headline
Merge in RED/redaction-service from RED-2624 to master

* commit '3c861074c9923a59fc71e2f57ee47c838ddbe3fe':
  RED-2624: Fixed footnote is recognized as headline
2021-11-03 12:25:43 +01:00
Dominique Eifländer 3c861074c9 RED-2624: Fixed footnote is recognized as headline 2021-11-03 12:11:38 +01:00
Dominique Eiflaender 3250d1791c Pull request #265: RED-2623: Fixed headline handline on strange ocr document
Merge in RED/redaction-service from RED-2623 to master

* commit '1b328b2b68da0b5be71d2f7464c767357003e4f2':
  RED-2623: Fixed headline handline on strange ocr document
2021-11-02 16:35:06 +01:00
Dominique Eifländer 1b328b2b68 RED-2623: Fixed headline handline on strange ocr document 2021-11-02 14:47:41 +01:00
Christoph  Schabert affed2258c Pull request #261: Split nightly and normal build
Merge in RED/redaction-service from RED-2312 to master

* commit '499c36b53d9b307c4899774e1d669482aa850a9f':
  Split nightly and normal build
2021-11-02 11:13:47 +01:00
Dominique Eiflaender 9d0973c037 Pull request #262: RED-2482: Always redact manual justification changes
Merge in RED/redaction-service from RED-2482-A to master

* commit '394e2fd4ed24e765e1bfc919688107a96af85fec':
  RED-2482: Always redact manual justification changes
2021-10-28 15:18:59 +02:00
Dominique Eifländer 394e2fd4ed RED-2482: Always redact manual justification changes 2021-10-28 15:09:34 +02:00
cschabert 499c36b53d Split nightly and normal build 2021-10-28 15:00:46 +02:00
Dominique Eiflaender 65b186be28 Pull request #259: RED-2536
Merge in RED/redaction-service from RED-2536 to master

* commit 'd892d6e81ee12927e57dc46c6439a65276824896':
  RED-2536: Treat \t same as whitespace
  RED-2223: Fixed stange end of textposition sequences that leads to wrong whitespaces
2021-10-28 12:52:47 +02:00
Dominique Eifländer d892d6e81e RED-2536: Treat \t same as whitespace 2021-10-28 12:42:18 +02:00
Dominique Eifländer 50017e7230 RED-2223: Fixed stange end of textposition sequences that leads to wrong whitespaces 2021-10-28 12:23:58 +02:00
Dominique Eiflaender 997317a683 Pull request #258: RED-2575: Fixed annotating entries on pages with cropbox != mediabox
Merge in RED/redaction-service from RED-2575 to master

* commit '012c5a794639eee34e392e940d5e380a5ea06df7':
  RED-2575: Fixed annotating entries on pages with cropbox != mediabox
2021-10-26 14:23:32 +02:00
Dominique Eifländer 012c5a7946 RED-2575: Fixed annotating entries on pages with cropbox != mediabox 2021-10-26 13:51:47 +02:00
Ali Oezyetimoglu a351761257 Pull request #256: RED-2429: As a user I want to resize a redaction
Merge in RED/redaction-service from RED-2429-rs1 to master

* commit 'd8f8702b2a78e7a3bc12e55791c251f3445d5b5e':
  RED-2429: As a user I want to resize a redaction
2021-10-22 12:23:54 +02:00
aoezyetimoglu d8f8702b2a RED-2429: As a user I want to resize a redaction 2021-10-22 12:05:31 +02:00
Dominique Eiflaender 0abcca97b3 Pull request #255: RED-2418: Fixed not working exception handling if redactionlog is not present
Merge in RED/redaction-service from RED-2418 to master

* commit 'f66a054cdea311376ffa88a6641b7eae9be6689f':
  RED-2418: Fixed not working exception handling if redactionlog is not present
2021-10-22 11:05:38 +02:00
Dominique Eifländer f66a054cde RED-2418: Fixed not working exception handling if redactionlog is not present 2021-10-22 10:33:42 +02:00
Dominique Eiflaender 0cc71897b8 Pull request #254: RED-2224: Fixed overriding entries with higher rank
Merge in RED/redaction-service from RED-2224-2 to master

* commit '634eee17167e01e2d4ee4d3d7b9613225e8c5643':
  RED-2224: Fixed overriding entries with higher rank
2021-10-19 15:38:44 +02:00
Dominique Eifländer 634eee1716 RED-2224: Fixed overriding entries with higher rank 2021-10-19 15:32:20 +02:00
Dominique Eiflaender fbad9a759b Pull request #253: RED-2224: Fixed missing values in engines
Merge in RED/redaction-service from RED-2224 to master

* commit 'fa8f598763efb334e21624df468d7e1802477076':
  RED-2224: Fixed missing values in engines
2021-10-19 11:45:38 +02:00
Dominique Eifländer fa8f598763 RED-2224: Fixed missing values in engines 2021-10-19 10:55:51 +02:00
Dominique Eiflaender 46c6385d11 Pull request #250: RED-2223: Fixed adding annotations to rotated words on non rotated pages
Merge in RED/redaction-service from RED-2223 to master

* commit '6005c476962dcdfd1d4e4fabf10fa840377b78db':
  RED-2223: Fixed adding annotations to rotated words on non rotated pages
2021-10-15 15:35:28 +02:00
Dominique Eifländer 6005c47696 RED-2223: Fixed adding annotations to rotated words on non rotated pages 2021-10-15 15:27:15 +02:00
Dominique Eiflaender ac556dfc9f Pull request #249: RED-2468: Fixed push recommendations to dictionary
Merge in RED/redaction-service from RED-2468 to master

* commit '15d836644b6a74273e81de361091d6ea1b71bea9':
  RED-2468: Fixed push recommendations to dictionary
2021-10-15 10:45:10 +02:00
Dominique Eifländer 15d836644b RED-2468: Fixed push recommendations to dictionary 2021-10-15 10:36:01 +02:00
Ali Oezyetimoglu d0cd3853cc Pull request #248: RED-2263: fixed Regex-Issue that can lead to stack overflow in redaction-service
Merge in RED/redaction-service from RED-2263-rs1 to master

* commit 'b96c39d5e58317e989faf8dc881ee6e596dbf363':
  RED-2263: fixed Regex-Issue that can lead to stack overflow in redaction-service
2021-10-14 13:21:58 +02:00
aoezyetimoglu b96c39d5e5 RED-2263: fixed Regex-Issue that can lead to stack overflow in redaction-service 2021-10-14 13:04:49 +02:00
Timo Bejan b83baf4bc4 Pull request #247: RED-2334
Merge in RED/redaction-service from 3.0-efsa-readiness to master

* commit '8bfe8acb85e33502f4978b12b5cc09f68a602f47':
  RED-2334
2021-10-13 09:24:16 +02:00
Timo Bejan 8bfe8acb85 RED-2334 2021-10-12 17:23:28 +03:00
Ali Oezyetimoglu 57f6bd77ed Pull request #246: RED-2246: updated platform-commons-dependency
Merge in RED/redaction-service from RED-2246-rs1 to master

* commit '15ed333478313f3f23d439ad6163553f9b68760e':
  RED-2246: updated platform-commons-dependency
2021-10-08 15:30:38 +02:00
aoezyetimoglu 15ed333478 RED-2246: updated platform-commons-dependency 2021-10-08 15:18:48 +02:00
Timo Bejan b3393a5f16 Pull request #245: RED-2378
Merge in RED/redaction-service from 3.0-efsa-readiness to master

* commit 'a299a326eb5587cd76b19ba848f5a771e2741edc':
  RED-2378
2021-10-08 13:59:30 +02:00
Timo Bejan a299a326eb RED-2378 2021-10-08 14:13:50 +03:00
Philipp Schramm 1034066e38 Pull request #244: Bugfix/workaround for windows
Merge in RED/redaction-service from bugfix/workaround-for-windows to master

* commit '535316c3a7740f61c682ae3d938f33671660c667':
  Bugfix with 'posix:permissions' for Windows systems
  Just a test without PosixFilePermission
2021-10-07 16:51:34 +02:00
Philipp Schramm 535316c3a7 Bugfix with 'posix:permissions' for Windows systems 2021-10-07 15:35:42 +02:00
Philipp Schramm 1e86073dff Just a test without PosixFilePermission 2021-10-07 14:35:44 +02:00
Philipp Schramm 5dc0e1932e Get temporary directory for tests dynamically to prevent issues with Windows systems 2021-10-07 14:12:39 +02:00
Philipp Schramm 8f75ff89a1 Use platform-docker-dependency version 1.1.0 which contains bugfix for Windows systems 2021-10-07 14:02:51 +02:00
Timo Bejan 1e61ea8736 Pull request #243: Updated Persistence Service Version
Merge in RED/redaction-service from 3.0-efsa-readiness to master

* commit '0fa4c54490422cf06f0a9034e3e9a83c78a66f6e':
  Updated Persistence Service Version
2021-10-07 09:12:46 +02:00
Timo Bejan 0fa4c54490 Updated Persistence Service Version 2021-10-07 10:06:36 +03:00
Dominique Eiflaender 243b4e4808 Pull request #242: RED-2172: Enbled to configure azure blob storage as storage backend
Merge in RED/redaction-service from RED-2172 to master

* commit 'e22e454fb5c16b326183179193bf24a8e7d724be':
  RED-2172: Enbled to configure azure blob storage as storage backend
2021-10-06 15:13:09 +02:00
Dominique Eifländer e22e454fb5 RED-2172: Enbled to configure azure blob storage as storage backend 2021-10-06 15:02:44 +02:00
Ali Oezyetimoglu 4316f36484 Pull request #241: RED-2261 rs7
Merge in RED/redaction-service from RED-2261-rs7 to master

* commit '22ae1ad6a98a0db4383e545c931d41111d0463ed':
  RED-2261: update dependencies
  RED-2261: update dependencies
2021-10-04 16:47:17 +02:00
aoezyetimoglu 22ae1ad6a9 RED-2261: update dependencies 2021-10-01 15:31:39 +02:00
aoezyetimoglu e7fc3f8079 RED-2261: update dependencies 2021-10-01 14:56:19 +02:00
Ali Oezyetimoglu 8f54adc8e6 Pull request #239: RED-2261: update dependencies
Merge in RED/redaction-service from RED-2261-rs5 to master

* commit '94622a90f373248ed0c10d7c6d7fddbee65c3028':
  RED-2261: update dependencies
2021-10-01 13:18:53 +02:00
aoezyetimoglu 94622a90f3 RED-2261: update dependencies 2021-10-01 10:59:37 +02:00
Ali Oezyetimoglu 58383467bb Pull request #238: RED-2261: update dependencies
Merge in RED/redaction-service from RED-2261-rs4 to master

* commit '8f31abf8dc9817be3a9a110eefd33ee919ae5313':
  RED-2261: update dependencies
2021-09-30 17:00:57 +02:00
aoezyetimoglu 8f31abf8dc RED-2261: update dependencies 2021-09-30 16:54:28 +02:00
Ali Oezyetimoglu 71e698da9d Pull request #237: RED-2261 rs3
Merge in RED/redaction-service from RED-2261-rs3 to master

* commit 'f3b6a2e57e00fc3892370408f6ba0a70b20791a1':
  RED-2261: update dependencies
  RED-2261: update dependencies
2021-09-30 16:51:41 +02:00
aoezyetimoglu f3b6a2e57e RED-2261: update dependencies 2021-09-30 16:41:35 +02:00
aoezyetimoglu a058bb996b RED-2261: update dependencies 2021-09-30 16:00:35 +02:00
Ali Oezyetimoglu 8633b02f4e Pull request #233: RED-2272: Make sure publicly writable directories are used safely && regex upper bound
Merge in RED/redaction-service from RED-2272-rs1 to master

* commit 'c4e47a48f8973ca1bbd602727a205647deb809fe':
  RED-2272: Make sure publicly writable directories are used safely && regex upper bound
  RED-2272: Make sure publicly writable directories are used safely && regex upper bound
2021-09-30 12:45:46 +02:00
aoezyetimoglu c4e47a48f8 RED-2272: Make sure publicly writable directories are used safely && regex upper bound 2021-09-30 12:37:01 +02:00
aoezyetimoglu adba0f99a0 RED-2272: Make sure publicly writable directories are used safely && regex upper bound 2021-09-30 11:51:48 +02:00
Timo Bejan 10ae7fc20d Pull request #232: typeId to type internal refactor
Merge in RED/redaction-service from persistence-service-integration to master

* commit 'dcd6d9d2c9cb66ad55e08d82a6044ea1b01c54eb':
  fixed pmd
  typeId to type internal refactor
2021-09-29 15:57:45 +02:00
Timo Bejan dcd6d9d2c9 fixed pmd 2021-09-29 16:52:06 +03:00
Timo Bejan 0856c542bc typeId to type internal refactor 2021-09-29 16:34:04 +03:00
Dominique Eiflaender dd8a07d2ad Pull request #231: Fixed update dictionary request
Merge in RED/redaction-service from UpdateDictionaryFix to master

* commit 'ded1b6eb66bd179589cb6191f748e39ea834a7a6':
  Fixed update dictionary request
2021-09-29 09:49:45 +02:00
Dominique Eifländer ded1b6eb66 Fixed update dictionary request 2021-09-29 09:40:40 +02:00
Timo Bejan 1a82cd0b84 Pull request #230: dep cleanup
Merge in RED/redaction-service from persistence-service-integration to master

* commit 'cd192ebf4e0f1937f5a3ba1b1d1f2701a6e74443':
  dep cleanup
2021-09-28 15:46:53 +02:00
Timo Bejan cd192ebf4e dep cleanup 2021-09-28 16:33:21 +03:00
Timo Bejan 6d50fe1d06 Pull request #229: upgraded ps integration
Merge in RED/redaction-service from persistence-service-integration to master

* commit '530747fa20bf5beb95a6b3d167cbfe09f0d145d5':
  upgraded ps integration
  upgraded ps integration
2021-09-28 14:53:46 +02:00
Timo Bejan 530747fa20 upgraded ps integration 2021-09-28 15:47:39 +03:00
Timo Bejan 3a9911dcac upgraded ps integration 2021-09-28 15:30:30 +03:00
Dominique Eiflaender bee124f394 Pull request #228: RED-2276: Fixed sonar bug findings
Merge in RED/redaction-service from RED-2276 to master

* commit '34daa4c22c3b5c8692f2f5f3056a633f9ec09649':
  RED-2276: Fixed sonar bug findings
2021-09-24 13:13:08 +02:00
Dominique Eifländer 34daa4c22c RED-2276: Fixed sonar bug findings 2021-09-24 13:06:54 +02:00
Dominique Eiflaender 100c1d68c3 Pull request #227: RED-2228: Use manual redaction classes from persistence-service, changed type in redactionlog to typeId
Merge in RED/redaction-service from RED-2228-2 to master

* commit '6ca134bdbd1084d530ba81a4a9895549d927d7a4':
  RED-2228: Use manual redaction classes from persistence-service, changed type in redactionlog to typeId
2021-09-23 12:39:53 +02:00
Dominique Eifländer 6ca134bdbd RED-2228: Use manual redaction classes from persistence-service, changed type in redactionlog to typeId 2021-09-23 12:15:00 +02:00
cschabert d9b78643fb Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service 2021-09-22 16:11:54 +02:00
cschabert 397870db8b add sonar 2021-09-22 16:11:51 +02:00
Dominique Eiflaender e3073f665c Pull request #226: RED-2228: Use persistence-service instead of configuration/file-management-service
Merge in RED/redaction-service from RED-2228 to master

* commit '146b0f5c8eb1fe8d8c530b940049e40a6eebc80a':
  RED-2228: Use persistence-service instead of configuration/file-management-service
2021-09-22 15:16:04 +02:00
Dominique Eifländer 146b0f5c8e RED-2228: Use persistence-service instead of configuration/file-management-service 2021-09-22 15:11:14 +02:00
Ali Oezyetimoglu 86b28b0612 Pull request #225: RED-2167 rs1
Merge in RED/redaction-service from RED-2167-rs1 to master

* commit 'ff97cba0c8935a44edf912310a459df879522c71':
  RED-2167: added computation of cellstarts for tables with two colums
  Adding persistence tests for  file-management-service-processor
  Testcommit
2021-09-15 13:40:50 +02:00
aoezyetimoglu ff97cba0c8 RED-2167: added computation of cellstarts for tables with two colums 2021-09-15 13:36:44 +02:00
aoezyetimoglu dd81ac79e4 Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service into Test 2021-09-14 16:36:37 +02:00
Dominique Eiflaender 16b06c2fc5 Pull request #224: RED-2125: Enabled possibility to reference annotations in rules
Merge in RED/redaction-service from RED-2125 to master

* commit '6f930b18133286289e330adbbbf390671b63d1b5':
  RED-2125: Enabled possibility to reference annotations in rules
2021-09-14 09:13:11 +02:00
Dominique Eifländer 6f930b1813 RED-2125: Enabled possibility to reference annotations in rules 2021-09-13 12:43:26 +02:00
Dominique Eiflaender c4b99378a7 Pull request #223: RED-2082: Added engines to redactionLog, to identify where a entry comes from
Merge in RED/redaction-service from RED-2082 to master

* commit 'd89a41caca623eedbeee7bc9b058b605db8fc359':
  RED-2082: Added engines to redactionLog, to identify where a entry comes from
2021-09-08 14:10:22 +02:00
Dominique Eifländer d89a41caca RED-2082: Added engines to redactionLog, to identify where a entry comes from 2021-09-08 13:20:38 +02:00
Dominique Eiflaender bcc069a826 Pull request #222: RED-1970: Ignore NER entities that are found over multiple table columns
Merge in RED/redaction-service from RED-1970-4 to master

* commit '6a75fc74d6eabbb0d99bdbca4d6cefcbb17d224f':
  RED-1970: Ignore NER entities that are found over multiple table columns
2021-09-07 15:30:55 +02:00
Timo Bejan 4b51a3eeeb Pull request #221: rule builder model
Merge in RED/redaction-service from rule-builder to master

* commit '7a3924d96ed40c932cfe02fb410060d4e820f88f':
  rule builder model
2021-09-07 14:56:50 +02:00
Dominique Eifländer 6a75fc74d6 RED-1970: Ignore NER entities that are found over multiple table columns 2021-09-07 14:56:19 +02:00
Timo Bejan 7a3924d96e rule builder model 2021-09-07 15:25:47 +03:00
Dominique Eiflaender abd0c2a2a8 Pull request #220: RED-1970: Base64 encode text in communication with entity-recogintion-service
Merge in RED/redaction-service from RED-1970-3 to master

* commit '8de655d8848c05de3ce5d7a19ad0887d516bcb72':
  RED-1970: Base64 encode text in communication with entity-recogintion-service
2021-09-07 10:41:48 +02:00
Dominique Eifländer 8de655d884 RED-1970: Base64 encode text in communication with entity-recogintion-service 2021-09-07 10:36:23 +02:00
Dominique Eiflaender a388a28cec Pull request #219: RED-1970: Call entity-redaction-service once per document
Merge in RED/redaction-service from RED-1970-2 to master

* commit 'f1b3d129ee32d40a78fad038a69027a24d8ccdd8':
  RED-1970: Call entity-redaction-service once per document
2021-09-07 09:46:41 +02:00
Dominique Eifländer f1b3d129ee RED-1970: Call entity-redaction-service once per document 2021-09-07 09:42:14 +02:00
Dominique Eiflaender 89acbb1001 Pull request #218: RED-1970: Seperated structur analysis from entity analysis
Merge in RED/redaction-service from RED-1970 to master

* commit '5939c3d460dde4585280c1372306acfaae89dd61':
  RED-1970: Seperated structur analysis from entity analysis
2021-09-03 12:43:42 +02:00
Dominique Eifländer 5939c3d460 RED-1970: Seperated structur analysis from entity analysis 2021-09-03 12:36:19 +02:00
aoezyetimoglu 6e9724f5bc Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service into Test 2021-08-27 14:32:32 +02:00
Dominique Eifländer 0ab0ff5f50 Fixed missing endpoint variable 2021-08-26 14:40:34 +02:00
Dominique Eiflaender 64cc23a883 Pull request #217: RED-1920: Integrated entity-recognition-service
Merge in RED/redaction-service from RED-1920 to master

* commit '76bf6773db1f32f0b178537ebbff7bdd2f1056c1':
  RED-1920: Integrated entity-recognition-service
2021-08-26 12:01:53 +02:00
Dominique Eifländer 76bf6773db RED-1920: Integrated entity-recognition-service 2021-08-26 11:53:02 +02:00
Dominique Eiflaender e3992f169a Pull request #216: Added equalsIgnoreCase fileattributes methods for rules
Merge in RED/redaction-service from FileAttributesIgnoreCase-master to master

* commit 'bcd469bb657f003813b09fd98bdbb2f624aa5e99':
  Added equalsIgnoreCase fileattributes methods for rules
2021-08-20 15:12:56 +02:00
Dominique Eifländer bcd469bb65 Added equalsIgnoreCase fileattributes methods for rules 2021-08-20 14:58:56 +02:00
Dominique Eiflaender 0b3298e6e2 Pull request #214: RED-1755: Calculate falgs in file-management-service async
Merge in RED/redaction-service from RED-1755-3master to master

* commit '078dc2c3c3508a2407a527ff3512239998a30869':
  RED-1755: Calculate falgs in file-management-service async
2021-08-18 16:06:59 +02:00
Dominique Eifländer 078dc2c3c3 RED-1755: Calculate falgs in file-management-service async 2021-08-18 16:02:38 +02:00
Dominique Eiflaender 4dda7ae543 Pull request #211: RED-2015: Fixed (Manual) Redactions in download reports/files for excluded pages
Merge in RED/redaction-service from RED-2015-master to master

* commit '4d514d0e5ed1c22436bd8ba58578022d5fb8c4db':
  RED-2015: Fixed (Manual) Redactions in download reports/files for excluded pages
2021-08-17 12:39:11 +02:00
Dominique Eifländer 4d514d0e5e RED-2015: Fixed (Manual) Redactions in download reports/files for excluded pages 2021-08-17 11:34:03 +02:00
Timo Bejan 1f12f9571f RedactionLogMergeService.java edited online with Bitbucket 2021-08-13 13:56:13 +02:00
Timo Bejan 5312d5ffdf RedactionLogMergeService.java edited online with Bitbucket 2021-08-13 13:19:42 +02:00
Timo Bejan e43e8bccbf RedactionLogMergeService.java edited online with Bitbucket 2021-08-13 13:00:32 +02:00
Timo Bejan 54250583b1 Pull request #207: Proper Order of manual actions based on time
Merge in RED/redaction-service from RED-1788 to master

* commit 'bb80b7ae7550d70a98b4d500d3870a13585b57e8':
  Proper Order of manual actions based on time
2021-08-13 10:27:23 +02:00
Timo Bejan bb80b7ae75 Proper Order of manual actions based on time 2021-08-13 11:19:54 +03:00
Dominique Eiflaender 25264f1450 Pull request #206: RED-1908: Removed RedactionChangeLog added changes to RedactionLog
Merge in RED/redaction-service from RED-1908 to master

* commit 'e8a4ef172c994ab8dd99f84f7a725e3dddfaca5f':
  RED-1908: Removed RedactionChangeLog added changes to RedactionLog
2021-08-11 10:00:48 +02:00
Dominique Eifländer e8a4ef172c RED-1908: Removed RedactionChangeLog added changes to RedactionLog 2021-08-11 09:45:00 +02:00
aoezyetimoglu 83b22f3e14 Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service into Test 2021-07-09 12:03:28 +02:00
aoezyetimoglu 167ae3be94 Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service into Test 2021-06-07 09:55:09 +02:00
aoezyetimoglu a6f8ea0f92 Merge branch 'master' of ssh://git.iqser.com:2222/red/redaction-service into Test
 Conflicts:
	redaction-service-v1/redaction-service-server-v1/pom.xml
2021-05-06 15:04:02 +02:00
aoezyetimoglu 4f1bfb972f Adding persistence tests for file-management-service-processor 2021-04-21 11:14:43 +02:00
aoezyetimoglu 95b351d3fe Testcommit 2021-04-15 16:26:47 +02:00
157 changed files with 9724 additions and 3753 deletions
+1 -1
View File
@@ -5,7 +5,7 @@
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>7.2.2</version>
<version>8.0.3</version>
<relativePath/>
</parent>
@@ -1,5 +1,8 @@
package buildjob;
import java.time.DayOfWeek;
import java.time.LocalTime;
import com.atlassian.bamboo.specs.api.BambooSpec;
import com.atlassian.bamboo.specs.api.builders.BambooKey;
import com.atlassian.bamboo.specs.api.builders.docker.DockerConfiguration;
@@ -19,6 +22,8 @@ import com.atlassian.bamboo.specs.builders.task.ScriptTask;
import com.atlassian.bamboo.specs.builders.task.VcsCheckoutTask;
import com.atlassian.bamboo.specs.builders.task.VcsTagTask;
import com.atlassian.bamboo.specs.builders.trigger.BitbucketServerTrigger;
import com.atlassian.bamboo.specs.builders.trigger.RepositoryPollingTrigger;
import com.atlassian.bamboo.specs.builders.trigger.ScheduledTrigger;
import com.atlassian.bamboo.specs.model.task.InjectVariablesScope;
import com.atlassian.bamboo.specs.util.BambooServer;
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
@@ -50,12 +55,18 @@ public class PlanSpec {
bambooServer.publish(plan);
PlanPermissions planPermission = new PlanSpec().createPlanPermission(plan.getIdentifier());
bambooServer.publish(planPermission);
Plan secPlan = new PlanSpec().createSecBuild();
bambooServer.publish(secPlan);
PlanPermissions secPlanPermission = new PlanSpec().createPlanPermission(secPlan.getIdentifier());
bambooServer.publish(secPlanPermission);
}
private PlanPermissions createPlanPermission(PlanIdentifier planIdentifier) {
Permissions permission = new Permissions()
.userPermissions("atlbamboo", PermissionType.EDIT, PermissionType.VIEW, PermissionType.ADMIN, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("red-backend", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("development", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("devplant", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.loggedInUserPermissions(PermissionType.VIEW)
.anonymousUserPermissionView();
return new PlanPermissions(planIdentifier.getProjectKey(), planIdentifier.getPlanKey()).permissions(permission);
@@ -119,4 +130,45 @@ public class PlanSpec {
.whenInactiveInRepositoryAfterDays(14))
.notificationForCommitters());
}
public Plan createSecBuild() {
return new Plan(
project(),
SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC"))
.description("Security Analysis Plan")
.stages(new Stage("Default Stage")
.jobs(new Job("Default Job",
new BambooKey("JOB1"))
.tasks(
new ScriptTask()
.description("Clean")
.inlineBody("#!/bin/bash\n" +
"set -e\n" +
"rm -rf ./*"),
new VcsCheckoutTask()
.description("Checkout Default Repository")
.checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask()
.description("Sonar")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh")
.argument(SERVICE_NAME))
.dockerConfiguration(
new DockerConfiguration()
.image("nexus.iqser.com:5001/infra/maven:3.6.2-jdk-13-3.0.0")
.dockerRunArguments("--net=host")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock")
)
)
)
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(
new ScheduledTrigger()
.scheduleOnceDaily(LocalTime.of(12, 00)),
new BitbucketServerTrigger())
.planBranchManagement(new PlanBranchManagement()
.createForVcsBranchMatching("release.*")
.notificationForCommitters());
}
}
+48
View File
@@ -0,0 +1,48 @@
#!/bin/bash
set -e
SERVICE_NAME=$1
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
--no-transfer-progress \
clean install \
-Djava.security.egd=file:/dev/./urandomelse
echo "dependency-check:aggregate"
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
--no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
org.owasp:dependency-check-maven:aggregate
if [[ -z "${bamboo_repository_pr_key}" ]]
then
echo "Sonar Scan for branch: ${bamboo_planRepository_1_branch}"
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
--no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
sonar:sonar \
-Dsonar.projectKey=RED_$SERVICE_NAME \
-Dsonar.host.url=https://sonarqube.iqser.com \
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
-Dsonar.branch.name=${bamboo_planRepository_1_branch} \
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
else
echo "Sonar Scan for PR with key1: ${bamboo_repository_pr_key}"
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
--no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
sonar:sonar \
-Dsonar.projectKey=RED_$SERVICE_NAME \
-Dsonar.host.url=https://sonarqube.iqser.com \
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
-Dsonar.pullrequest.key=${bamboo_repository_pr_key} \
-Dsonar.pullrequest.branch=${bamboo_repository_pr_sourceBranch} \
-Dsonar.pullrequest.base=${bamboo_repository_pr_targetBranch} \
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
fi
@@ -11,7 +11,9 @@ public class PlanSpecTest {
@Test
public void checkYourPlanOffline() throws PropertiesValidationException {
Plan plan = new PlanSpec().createPlan();
EntityPropertiesBuilders.build(plan);
Plan secPlan = new PlanSpec().createSecBuild();
EntityPropertiesBuilders.build(secPlan);
}
}
+2 -1
View File
@@ -5,7 +5,8 @@
<parent>
<groupId>com.iqser.red</groupId>
<artifactId>platform-docker-dependency</artifactId>
<version>1.0.0</version>
<version>1.1.0</version>
<relativePath />
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -1,4 +1,4 @@
FROM red/base-image:1.0.0
FROM red/redaction-service-base-v1:1.0.0
ARG PLATFORM_JAR
@@ -7,12 +7,3 @@ ENV PLATFORM_JAR ${PLATFORM_JAR}
ENV USES_ELASTICSEARCH false
COPY ["${PLATFORM_JAR}", "/"]
RUN apt-get update \
&& DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
wget cabextract xfonts-utils fonts-liberation \
&& rm -rf /var/lib/apt/lists/*
RUN curl http://ftp.br.debian.org/debian/pool/contrib/m/msttcorefonts/ttf-mscorefonts-installer_3.8_all.deb -o /tmp/ttf-mscorefonts-installer_3.8_all.deb \
&& dpkg -i /tmp/ttf-mscorefonts-installer_3.8_all.deb \
&& rm /tmp/ttf-mscorefonts-installer_3.8_all.deb \
+63 -4
View File
@@ -5,7 +5,8 @@
<parent>
<artifactId>platform-dependency</artifactId>
<groupId>com.iqser.red</groupId>
<version>1.1.3</version>
<version>1.6.0</version>
<relativePath />
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -13,7 +14,6 @@
<groupId>com.iqser.red.service</groupId>
<version>1.0-SNAPSHOT</version>
<packaging>pom</packaging>
<modules>
@@ -22,7 +22,7 @@
</modules>
<properties>
<pdfbox.version>2.0.21</pdfbox.version>
<pdfbox.version>2.0.24</pdfbox.version>
</properties>
@@ -32,7 +32,7 @@
<dependency>
<groupId>com.iqser.red</groupId>
<artifactId>platform-commons-dependency</artifactId>
<version>1.3.6</version>
<version>1.11.0</version>
<scope>import</scope>
<type>pom</type>
</dependency>
@@ -52,4 +52,63 @@
</dependencyManagement>
<build>
<pluginManagement>
<plugins>
<plugin>
<groupId>org.sonarsource.scanner.maven</groupId>
<artifactId>sonar-maven-plugin</artifactId>
<version>3.9.0.2155</version>
</plugin>
<plugin>
<groupId>org.owasp</groupId>
<artifactId>dependency-check-maven</artifactId>
<version>6.3.1</version>
<configuration>
<format>ALL</format>
</configuration>
</plugin>
<plugin>
<groupId>org.jacoco</groupId>
<artifactId>jacoco-maven-plugin</artifactId>
<executions>
<execution>
<id>prepare-agent</id>
<goals>
<goal>prepare-agent</goal>
</goals>
</execution>
<execution>
<id>report</id>
<goals>
<goal>report</goal>
</goals>
</execution>
</executions>
</plugin>
</plugins>
</pluginManagement>
<plugins>
<plugin>
<groupId>org.jacoco</groupId>
<artifactId>jacoco-maven-plugin</artifactId>
<version>0.8.7</version>
<executions>
<execution>
<id>prepare-agent</id>
<goals>
<goal>prepare-agent</goal>
</goals>
</execution>
<execution>
<id>report</id>
<goals>
<goal>report-aggregate</goal>
</goals>
<phase>verify</phase>
</execution>
</executions>
</plugin>
</plugins>
</build>
</project>
@@ -11,6 +11,10 @@
<artifactId>redaction-service-api-v1</artifactId>
<properties>
<persistence-service.version>1.85.0</persistence-service.version>
</properties>
<dependencies>
<dependency>
<groupId>org.springframework</groupId>
@@ -19,14 +23,8 @@
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>configuration-service-api-v1</artifactId>
<version>2.11.0</version>
<exclusions>
<exclusion>
<groupId>com.iqser.red.service</groupId>
<artifactId>file-management-service-api-v1</artifactId>
</exclusion>
</exclusions>
<artifactId>persistence-service-api-v1</artifactId>
<version>${persistence-service.version}</version>
</dependency>
</dependencies>
</project>
@@ -1,28 +1,32 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnalyzeRequest {
private MessageType messageType;
private String dossierId;
private String fileId;
private String dossierTemplateId;
private boolean reanalyseOnlyIfPossible;
private ManualRedactions manualRedactions;
private OffsetDateTime lastProcessed;
private int analysisNumber;
@Builder.Default
private Set<Integer> excludedPages = new HashSet<>();
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -11,14 +13,12 @@ import lombok.NoArgsConstructor;
@AllArgsConstructor
public class AnalyzeResult {
private MessageType messageType;
private String dossierId;
private String fileId;
private long duration;
private int numberOfPages;
private boolean hasHints;
private boolean hasRequests;
private boolean hasRedactions;
private boolean hasImages;
private boolean hasUpdates;
private long dictionaryVersion;
private long dossierDictionaryVersion;
@@ -28,6 +28,9 @@ public class AnalyzeResult {
private boolean wasReanalyzed;
private int analysisVersion;
private int analysisNumber;
private ManualRedactions manualRedactions;
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -14,4 +16,5 @@ public class AnnotateRequest {
private String dossierId;
private String dossierTemplateId;
private String fileId;
private ManualRedactions manualRedactions;
}
@@ -1,4 +1,4 @@
package com.iqser.red.service.redaction.v1.server.client;
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
@@ -7,7 +7,9 @@ import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class ImageClassificationResponse {
public class Argument {
private String category;
}
private String name;
private ArgumentType type;
}
@@ -0,0 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
public enum ArgumentType {
INTEGER, BOOLEAN, STRING, FILE_ATTRIBUTE, REGEX, TYPE, RULE_NUMBER, LEGAL_BASIS, REFERENCE_TYPE
}
@@ -1,21 +1,19 @@
package com.iqser.red.service.redaction.v1.model;
import java.time.OffsetDateTime;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class Comment {
private String id;
private OffsetDateTime date;
private String text;
private String user;
public class Change {
private int analysisNumber;
private ChangeType type;
private OffsetDateTime dateTime;
}
@@ -1,5 +1,5 @@
package com.iqser.red.service.redaction.v1.model;
public enum ChangeType {
ADDED, REMOVED
ADDED, REMOVED, CHANGED
}
@@ -0,0 +1,5 @@
package com.iqser.red.service.redaction.v1.model;
public enum Engine {
DICTIONARY, NER, RULE
}
@@ -1,25 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class IdRemoval {
private String id;
private String user;
private Status status;
private boolean removeFromDictionary;
private OffsetDateTime requestDate;
private OffsetDateTime processedDate;
private OffsetDateTime softDeletedTime;
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImportedRedaction {
private String id;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
}
@@ -0,0 +1,20 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImportedRedactions {
@Builder.Default
private Map<Integer, List<ImportedRedaction>> importedRedactions = new HashMap<>();
}
@@ -0,0 +1,50 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.BaseAnnotation;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
import java.util.HashMap;
import java.util.Map;
@Data
@AllArgsConstructor
@NoArgsConstructor
@Builder
public class ManualChange {
private AnnotationStatus annotationStatus;
private ManualRedactionType manualRedactionType;
private OffsetDateTime processedDate;
private OffsetDateTime requestedDate;
private String userId;
private Map<String, Object> propertyChanges = new HashMap<>();
public boolean isProcessed() {
return processedDate != null;
}
public static ManualChange from(BaseAnnotation baseAnnotation) {
ManualChange manualChange = new ManualChange();
manualChange.annotationStatus = baseAnnotation.getStatus();
manualChange.processedDate = baseAnnotation.getProcessedDate();
manualChange.requestedDate = baseAnnotation.getRequestDate();
manualChange.userId = baseAnnotation.getUser();
return manualChange;
}
public ManualChange withManualRedactionType(ManualRedactionType manualRedactionType) {
this.manualRedactionType = manualRedactionType;
return this;
}
public ManualChange withChange(String property, Object value) {
this.propertyChanges.put(property, value);
return this;
}
}
@@ -1,25 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualForceRedact {
private String id;
private String user;
private Status status;
private String legalBasis;
private OffsetDateTime requestDate;
private OffsetDateTime processedDate;
private OffsetDateTime softDeletedTime;
}
@@ -1,25 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualImageRecategorization {
private String id;
private String user;
private Status status;
private String type;
private OffsetDateTime requestDate;
private OffsetDateTime processedDate;
private OffsetDateTime softDeletedTime;
}
@@ -1,25 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualLegalBasisChange {
private String id;
private String user;
private Status status;
private String legalBasis;
private OffsetDateTime requestDate;
private OffsetDateTime processedDate;
private OffsetDateTime softDeletedTime;
}
@@ -1,34 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.List;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactionEntry {
private String id;
private String user;
private String type;
private String value;
private String reason;
private String legalBasis;
private List<Rectangle> positions = new ArrayList<>();
private Status status;
private boolean addToDictionary;
private boolean addToDossierDictionary;
private OffsetDateTime requestDate;
private OffsetDateTime processedDate;
private OffsetDateTime softDeletedTime;
}
@@ -1,5 +1,13 @@
package com.iqser.red.service.redaction.v1.model;
public enum ManualRedactionType {
ADD, REMOVE, FORCE_REDACT, RECATEGORIZE, LEGAL_BASIS_CHANGE
ADD_LOCALLY,
ADD_TO_DICTIONARY,
REMOVE_LOCALLY,
REMOVE_FROM_DICTIONARY,
FORCE_REDACT,
FORCE_HINT,
RECATEGORIZE,
LEGAL_BASIS_CHANGE,
RESIZE
}
@@ -1,38 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactions {
@Builder.Default
private Set<IdRemoval> idsToRemove = new HashSet<>();
@Builder.Default
private Set<ManualForceRedact> forceRedacts = new HashSet<>();
@Builder.Default
private Set<ManualRedactionEntry> entriesToAdd = new HashSet<>();
@Builder.Default
private Set<ManualImageRecategorization> imageRecategorizations = new HashSet<>();
@Builder.Default
private Set<ManualLegalBasisChange> manualLegalBasisChanges = new HashSet<>();
@Builder.Default
private Map<String, List<Comment>> comments = new HashMap<>();
}
@@ -0,0 +1,7 @@
package com.iqser.red.service.redaction.v1.model;
public enum MessageType {
ANALYSE, REANALYSE, STRUCTURE_ANALYSE, SURROUNDING_TEXT
}
@@ -1,22 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class RedactionChangeLog {
private List<RedactionChangeLogEntry> redactionLogEntry = new ArrayList<>();
private long dictionaryVersion = -1;
private long dossierDictionaryVersion = -1;
private long rulesVersion = -1;
private long legalBasisVersion = -1;
}
@@ -1,49 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class RedactionChangeLogEntry {
private String id;
private String type;
private String value;
private String reason;
private int matchedRule;
private String legalBasis;
private boolean redacted;
private boolean isHint;
private boolean isRecommendation;
private String section;
private float[] color;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private boolean manual;
private Status status;
private ManualRedactionType manualRedactionType;
private boolean isDictionaryEntry;
private String textBefore;
private String textAfter;
@Builder.Default
private List<Comment> comments = new ArrayList<>();
private ChangeType changeType;
private boolean isDossierDictionaryEntry;
private boolean excluded;
}
@@ -1,11 +1,12 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.configuration.v1.api.model.LegalBasisMapping;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.legalbasis.LegalBasis;
import lombok.AllArgsConstructor;
import lombok.Data;
import java.util.List;
@Data
@AllArgsConstructor
public class RedactionLog {
@@ -17,12 +18,18 @@ public class RedactionLog {
*/
private long analysisVersion;
/**
* Which analysis created this redactionLog.
*/
private int analysisNumber;
private List<RedactionLogEntry> redactionLogEntry;
private List<LegalBasisMapping> legalBasis;
private List<LegalBasis> legalBasis;
private long dictionaryVersion = -1;
private long dossierDictionaryVersion = -1;
private long rulesVersion = -1;
private long legalBasisVersion = -1;
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class RedactionLogChanges {
private RedactionLog redactionLog;
private boolean hasChanges;
}
@@ -1,14 +1,10 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.Comment;
import lombok.*;
import java.util.*;
@Data
@@ -23,21 +19,23 @@ public class RedactionLogEntry {
private String value;
private String reason;
private int matchedRule;
private boolean rectangle;
private String legalBasis;
private boolean imported;
private boolean redacted;
private boolean isHint;
private boolean isRecommendation;
private boolean isFalsePositive;
private String section;
private float[] color;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private boolean manual;
private Status status;
private ManualRedactionType manualRedactionType;
private String manualRedactionUserId;
private boolean isDictionaryEntry;
private String textBefore;
private String textAfter;
@@ -51,11 +49,39 @@ public class RedactionLogEntry {
private boolean isImage;
private boolean imageHasTransparency;
private boolean isDictionaryEntry;
private boolean isDossierDictionaryEntry;
private boolean excluded;
private String recategorizationType;
private String legalBasisChangeValue;
@EqualsAndHashCode.Exclude
@Builder.Default
private List<Change> changes = new ArrayList<>();
@EqualsAndHashCode.Exclude
@Builder.Default
private List<ManualChange> manualChanges = new ArrayList<>();
private Set<Engine> engines = new HashSet<>();
private Set<String> reference = new HashSet<>();
public boolean lastChangeIsRemoved() {
return last(changes).map(c -> c.getType() == ChangeType.REMOVED).orElse(false);
}
public boolean isLocalManualRedaction() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
public boolean isManuallyRemoved() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
private <T> Optional<T> last(List<T> list) {
return list.isEmpty() ? Optional.empty() : Optional.of(list.get(list.size() - 1));
}
}
@@ -1,5 +1,15 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.Type;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -15,4 +25,11 @@ public class RedactionRequest {
private String fileId;
private String dossierTemplateId;
private ManualRedactions manualRedactions;
@Builder.Default
private Set<Integer> excludedPages = new HashSet<>();
private Colors colors;
private List<Type> types;
private boolean includeFalsePositives;
}
@@ -0,0 +1,14 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.Data;
import java.util.ArrayList;
import java.util.List;
@Data
public class RuleBuilderModel {
private List<RuleElement> whenClauses = new ArrayList<>();
private List<RuleElement> thenConditions = new ArrayList<>();
}
@@ -0,0 +1,18 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class RuleElement {
private String conditionName;
private List<Argument> arguments = new ArrayList<>();
}
@@ -30,4 +30,9 @@ public class SectionArea {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft().getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft().getY() + other.getHeight();
}
// TODO we should only use one rectangle class.
public boolean contains(com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
}
}
@@ -3,10 +3,9 @@ package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.RequiredArgsConstructor;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.*;
@Data
@AllArgsConstructor
@@ -15,4 +14,16 @@ public class SectionGrid {
private Map<Integer, List<SectionRectangle>> rectanglesPerPage = new HashMap<>();
private List<SectionGridSection> sections = new ArrayList<>();
@Data
@RequiredArgsConstructor
public static class SectionGridSection {
private final int sectionNumber;
private final String headline;
private final Set<Integer> pages;
private final List<SectionArea> sectionAreas;
}
}
@@ -1,5 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum Status {
REQUESTED, APPROVED, DECLINED
}
@@ -0,0 +1,18 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class StructureAnalyzeRequest {
private String dossierId;
private String fileId;
}
@@ -1,29 +1,43 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.*;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RequestParam;
public interface RedactionResource {
@PostMapping(value = "/annotate", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest);
@PostMapping(value = "/debug/classifications", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult classify(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/debug/sections", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult sections(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/debug/htmlTables", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
void testRules(@RequestBody String rules);
@PostMapping(value = "/redaction-log/preview", consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionLog getRedactionLog(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/manual/surrounding-text/{dossierId}/{fileId}", consumes = MediaType.APPLICATION_JSON_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions);
}
@@ -0,0 +1,12 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
public interface RuleBuilderResource {
@PostMapping(value = "/rule-builder-model", produces = MediaType.APPLICATION_JSON_VALUE)
RuleBuilderModel getRuleBuilderModel();
}
@@ -11,6 +11,14 @@
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<drools.version>7.59.0.Final</drools.version>
<kie.version>7.59.0.Final</kie.version>
<locationtech.version>1.16.1</locationtech.version>
<pdfbox.jbig2-imageio.version>3.0.3</pdfbox.jbig2-imageio.version>
<jai-imageio.version>1.4.0</jai-imageio.version>
</properties>
<dependencies>
<dependency>
<groupId>com.iqser.red.commons</groupId>
@@ -21,56 +29,40 @@
<artifactId>redaction-service-api-v1</artifactId>
<version>${project.version}</version>
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>file-management-service-api-v1</artifactId>
<version>2.25.0</version>
<exclusions>
<exclusion>
<groupId>com.iqser.red.service</groupId>
<artifactId>redaction-service-api-v1</artifactId>
</exclusion>
<exclusion>
<groupId>com.iqser.red.service</groupId>
<artifactId>configuration-service-api-v1</artifactId>
</exclusion>
</exclusions>
</dependency>
<dependency>
<groupId>org.drools</groupId>
<artifactId>drools-core</artifactId>
<version>7.37.0.Final</version>
<version>${drools.version}</version>
</dependency>
<dependency>
<groupId>org.kie</groupId>
<artifactId>kie-spring</artifactId>
<version>7.37.0.Final</version>
<version>${kie.version}</version>
</dependency>
<dependency>
<groupId>org.locationtech.jts</groupId>
<artifactId>jts-core</artifactId>
<version>1.16.1</version>
<version>${locationtech.version}</version>
</dependency>
<dependency>
<groupId>com.google.guava</groupId>
<artifactId>guava</artifactId>
<version>29.0-jre</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>jbig2-imageio</artifactId>
<version>3.0.3</version>
<version>${pdfbox.jbig2-imageio.version}</version>
</dependency>
<dependency>
<groupId>com.github.jai-imageio</groupId>
<artifactId>jai-imageio-core</artifactId>
<version>1.4.0</version>
<version>${jai-imageio.version}</version>
</dependency>
<dependency>
<groupId>com.github.jai-imageio</groupId>
<artifactId>jai-imageio-jpeg2000</artifactId>
<version>1.4.0</version>
<version>${jai-imageio.version}</version>
</dependency>
<!-- commons -->
@@ -87,10 +79,6 @@
<artifactId>metric-commons</artifactId>
</dependency>
<!-- other external -->
<dependency>
<groupId>org.apache.commons</groupId>
<artifactId>commons-lang3</artifactId>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
@@ -108,7 +96,6 @@
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-amqp</artifactId>
<version>2.3.1.RELEASE</version>
</dependency>
<!-- test dependencies -->
@@ -122,12 +109,6 @@
<artifactId>test-commons</artifactId>
<scope>test</scope>
</dependency>
<dependency>
<groupId>org.springframework.amqp</groupId>
<artifactId>spring-rabbit-test</artifactId>
<version>2.3.1</version>
<scope>test</scope>
</dependency>
</dependencies>
<build>
@@ -1,19 +1,14 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryVersion;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
@Data
@NoArgsConstructor
public class Document {
@@ -23,20 +18,14 @@ public class Document {
private List<Header> headers = new ArrayList<>();
private List<Footer> footers = new ArrayList<>();
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
private Map<Integer, List<Entity>> entities = new HashMap<>();
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
private FloatFrequencyCounter fontSizeCounter = new FloatFrequencyCounter();
private StringFrequencyCounter fontCounter = new StringFrequencyCounter();
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private boolean headlines;
private List<RedactionLogEntry> redactionLogEntities = new ArrayList<>();
private SectionGrid sectionGrid = new SectionGrid();
private DictionaryVersion dictionaryVersion;
private long rulesVersion;
private List<SectionText> sectionText = new ArrayList<>();
private Map<Integer, Set<Image>> images = new HashMap<>();
}
@@ -7,6 +7,7 @@ import lombok.Data;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@@ -16,7 +17,7 @@ public class Page {
@NonNull
private List<AbstractTextContainer> textBlocks;
private List<PdfImage> images;
private List<PdfImage> images = new ArrayList<>();
private Rectangle bodyTextFrame;
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.server.classification.service;
import static java.util.stream.Collectors.toSet;
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.Orientation;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
@@ -16,6 +18,7 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import org.springframework.stereotype.Service;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.Iterator;
import java.util.List;
@@ -25,9 +28,12 @@ public class BlockificationService {
static final float THRESHOLD = 1f;
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
sortRotatedSequences(textPositions);
List<TextPositionSequence> chunkWords = new ArrayList<>();
List<AbstractTextContainer> chunkBlockList1 = new ArrayList<>();
@@ -50,7 +56,7 @@ public class BlockificationService {
if (prev != null && (lineSeparation || startFromTop || splitByX || newLineAfterSplit || splittedByRuling)) {
Orientation prevOrientation = null;
if(!chunkBlockList1.isEmpty()) {
if (!chunkBlockList1.isEmpty()) {
prevOrientation = chunkBlockList1.get(chunkBlockList1.size() - 1).getOrientation();
}
@@ -62,15 +68,11 @@ public class BlockificationService {
wasSplitted = true;
cb1.setOrientation(Orientation.LEFT);
splitX1 = word.getX1();
} else
if (newLineAfterSplit && !splittedByRuling) {
} else if (newLineAfterSplit && !splittedByRuling) {
wasSplitted = false;
cb1.setOrientation(Orientation.RIGHT);
splitX1 = null;
} else
if(prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !splittedByRuling)){
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !splittedByRuling)) {
cb1.setOrientation(Orientation.LEFT);
}
@@ -110,16 +112,18 @@ public class BlockificationService {
while (itty.hasNext()) {
TextBlock block = (TextBlock) itty.next();
if(previousLeft != null && block.getOrientation().equals(Orientation.LEFT)){
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft.getMinY()){
if (previousLeft != null && block.getOrientation().equals(Orientation.LEFT)) {
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft
.getMinY()) {
previousLeft.add(block);
itty.remove();
continue;
}
}
if(previousRight != null && block.getOrientation().equals(Orientation.RIGHT)){
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight.getMinY()){
if (previousRight != null && block.getOrientation().equals(Orientation.RIGHT)) {
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight
.getMinY()) {
previousRight.add(block);
itty.remove();
continue;
@@ -133,16 +137,16 @@ public class BlockificationService {
}
}
itty = chunkBlockList1.iterator();
TextBlock previous = null;
while (itty.hasNext()) {
TextBlock block = (TextBlock) itty.next();
if(previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation().equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(), previous
.getMaxY())||
previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation().equals(Orientation.RIGHT) && equalsWithThreshold(block.getMaxY(), previous
.getMaxY())){
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY()) || previous != null && previous
.getOrientation()
.equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.RIGHT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY())) {
previous.add(block);
itty.remove();
continue;
@@ -151,11 +155,12 @@ public class BlockificationService {
previous = block;
}
return new Page(chunkBlockList1);
}
private boolean equalsWithThreshold(float f1, float f2){
private boolean equalsWithThreshold(float f1, float f2) {
return Math.abs(f1 - f2) < THRESHOLD;
}
@@ -197,6 +202,13 @@ public class BlockificationService {
textBlock.setHighestFontSize(fontSizeFrequencyCounter.getHighest());
}
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences()
.stream()
.map(t -> round(t.getY1(), 3))
.collect(toSet())
.size() == 1) {
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getX1));
}
return textBlock;
}
@@ -291,4 +303,30 @@ public class BlockificationService {
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
}
private void sortRotatedSequences(List<TextPositionSequence> sequences) {
List<TextPositionSequence> rotatedWords = new ArrayList<>();
Iterator<TextPositionSequence> itty = sequences.iterator();
while (itty.hasNext()) {
var pos = itty.next();
if (pos.getTextPositions().get(0).getDir() == 270) {
rotatedWords.add(pos);
itty.remove();
}
}
if (!rotatedWords.isEmpty() && !sequences.isEmpty()) {
rotatedWords.sort(Comparator.comparing(TextPositionSequence::getX1));
}
sequences.addAll(rotatedWords);
}
private double round(float value, int decimalPoints) {
var d = Math.pow(10, decimalPoints);
return Math.round(value * d) / d;
}
}
@@ -52,7 +52,7 @@ public class ClassificationService {
List<Float> headlineFontSizes) {
if (document.getFontSizeCounter().getMostPopular() == null) {
// TODO Figure out why this happens.
textBlock.setClassification("Other");
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter()
@@ -73,8 +73,10 @@ public class ClassificationService {
}
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && textBlock.getMostPopularWordStyle()
.equals("bold")) {
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle()
.equals("bold") || !document.getFontStyleCounter().getCountPerValue().containsKey("bold") && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() + 1) && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
for (int i = 1; i <= headlineFontSizes.size(); i++) {
if (textBlock.getMostPopularWordFontSize() == headlineFontSizes.get(i - 1)) {
@@ -86,7 +88,7 @@ public class ClassificationService {
.startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter()
.getMostPopular()
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
textBlock.setClassification("H " + (headlineFontSizes.size() + 1));
document.setHeadlines(true);
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
@@ -1,8 +1,9 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.configuration.v1.api.resource.DictionaryResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = "DictionaryResource", url = "${configuration-service.url}")
import com.iqser.red.service.persistence.service.v1.api.resources.DictionaryResource;
@FeignClient(name = "DictionaryResource", url = "${persistence-service.url}")
public interface DictionaryClient extends DictionaryResource {
}
@@ -0,0 +1,16 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
import com.iqser.red.service.redaction.v1.server.client.model.EntityRecognitionRequest;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
@FeignClient(name = "EntityRecognitionClient", url = "${entity-recognition-service.url}")
public interface EntityRecognitionClient {
@PostMapping(value = "/find_authors", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
NerEntities findAuthors(EntityRecognitionRequest entityRecognitionRequest);
}
@@ -1,9 +1,10 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.file.management.v1.api.resources.FileStatusProcessingUpdateResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${file-management-service.url}")
import com.iqser.red.service.persistence.service.v1.api.resources.FileStatusProcessingUpdateResource;
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${persistence-service.url}")
public interface FileStatusProcessingUpdateClient extends FileStatusProcessingUpdateResource {
}
@@ -1,15 +0,0 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.multipart.MultipartFile;
@FeignClient(name = "ImageClassificationResource", url = "${image-service.url}")
public interface ImageClassificationClient {
@PostMapping(value = "/process_full_img", consumes = MediaType.MULTIPART_FORM_DATA_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
ImageClassificationResponse classify(@RequestBody MultipartFile file);
}
@@ -1,8 +1,9 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.configuration.v1.api.resource.LegalBasisMappingResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = "LegalBasisMappingResource", url = "${configuration-service.url}")
import com.iqser.red.service.persistence.service.v1.api.resources.LegalBasisMappingResource;
@FeignClient(name = "LegalBasisMappingResource", url = "${persistence-service.url}")
public interface LegalBasisClient extends LegalBasisMappingResource {
}
@@ -1,8 +1,9 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = "RulesResource", url = "${configuration-service.url}")
import com.iqser.red.service.persistence.service.v1.api.resources.RulesResource;
@FeignClient(name = "RulesResource", url = "${persistence-service.url}")
public interface RulesClient extends RulesResource {
}
@@ -0,0 +1,19 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecogintionEntity {
private String value;
private int startOffset;
private int endOffset;
private String type;
}
@@ -0,0 +1,18 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecognitionRequest {
private List<EntityRecognitionSection> data;
}
@@ -0,0 +1,20 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecognitionResult {
@Builder.Default
private Map<Integer, List<EntityRecogintionEntity>> entities = new HashMap<>();
}
@@ -0,0 +1,16 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecognitionSection {
private int sectionNumber;
private String text;
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class NerEntities {
@Builder.Default
private Map<Integer, List<EntityRecogintionEntity>> data = new HashMap<>();
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.commons.spring.ErrorMessage;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import lombok.extern.slf4j.Slf4j;
import org.springframework.http.HttpStatus;
@@ -36,4 +37,11 @@ public class ControllerAdvice {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
@ResponseBody
@ResponseStatus(value = HttpStatus.NOT_FOUND)
@ExceptionHandler(value = NotFoundException.class)
public ErrorMessage handleFileNotFoundException(NotFoundException e) {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
}
@@ -1,31 +1,42 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.file.management.v1.api.model.FileType;
import com.iqser.red.service.redaction.v1.model.*;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.AnnotationService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogMergeService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.stream.Collectors;
import org.apache.pdfbox.io.MemoryUsageSetting;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.redaction.v1.model.AnnotateRequest;
import com.iqser.red.service.redaction.v1.model.AnnotateResponse;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.AnnotationService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogMergeService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RestController
@@ -39,18 +50,25 @@ public class RedactionController implements RedactionResource {
private final PdfSegmentationService pdfSegmentationService;
private final RedactionStorageService redactionStorageService;
private final RedactionLogMergeService redactionLogMergeService;
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
public AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(annotateRequest.getDossierId(), annotateRequest.getFileId(), FileType.ORIGIN));
var redactionLog = redactionStorageService.getRedactionLog(annotateRequest.getDossierId(), annotateRequest.getFileId());
var mergedRedactionLog = getRedactionLog(RedactionRequest.builder()
.fileId(annotateRequest.getFileId())
.manualRedactions(annotateRequest.getManualRedactions())
.dossierId(annotateRequest.getDossierId())
.dossierTemplateId(annotateRequest.getDossierTemplateId())
.build());
var sectionsGrid = redactionStorageService.getSectionGrid(annotateRequest.getDossierId(), annotateRequest.getFileId());
try (PDDocument pdDocument = PDDocument.load(storedObjectStream, MemoryUsageSetting.setupTempFileOnly())) {
pdDocument.setAllSecurityToBeRemoved(true);
dictionaryService.updateDictionary(annotateRequest.getDossierTemplateId(), annotateRequest.getDossierId());
annotationService.annotate(pdDocument, redactionLog, sectionsGrid);
annotationService.annotate(pdDocument, mergedRedactionLog, sectionsGrid);
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
pdDocument.save(byteArrayOutputStream);
@@ -65,9 +83,10 @@ public class RedactionController implements RedactionResource {
@Override
public RedactionResult classify(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream);
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
@@ -85,14 +104,15 @@ public class RedactionController implements RedactionResource {
throw new RedactionException(e);
}
}
@Override
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream);
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
@@ -109,7 +129,6 @@ public class RedactionController implements RedactionResource {
throw new RedactionException(e);
}
}
@@ -120,12 +139,11 @@ public class RedactionController implements RedactionResource {
try {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, true);
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
} catch (Exception e) {
throw new RedactionException(e);
}
StringBuilder sb = new StringBuilder();
for (Page page : classifiedDoc.getPages()) {
for (AbstractTextContainer textContainer : page.getTextBlocks()) {
@@ -140,29 +158,53 @@ public class RedactionController implements RedactionResource {
}
@Override
public void testRules(@RequestBody String rules) {
droolsExecutionService.testRules(rules);
}
@Override
public RedactionLog getRedactionLog(RedactionRequest redactionRequest) {
log.info("Requested preview for: {}", redactionRequest);
dictionaryService.updateDictionary(redactionRequest.getDossierTemplateId(), redactionRequest.getDossierId());
log.debug("Requested preview for: {}", redactionRequest);
var redactionLog = redactionStorageService.getRedactionLog(redactionRequest.getDossierId(), redactionRequest.getFileId());
log.info("Loaded redaction log with computationalVersion: {}", redactionLog.getAnalysisVersion());
if (redactionLog.getAnalysisVersion() == 0) {
// old redaction logs are returned directly
return redactionLog;
} else {
return redactionLogMergeService.mergeRedactionLogData(redactionLog, redactionRequest.getDossierTemplateId(), redactionRequest.getManualRedactions());
if (redactionLog == null) {
throw new NotFoundException("RedactionLog not present");
}
log.info("Loaded redaction log with computationalVersion: {}", redactionLog.getAnalysisVersion());
SectionGrid sectionGrid = redactionStorageService.getSectionGrid(redactionRequest.getDossierId(), redactionRequest.getFileId());
if (sectionGrid.getSections().isEmpty()) {
log.info("SectionGrid does not have headlines set. Computing headlines now!");
var text = redactionStorageService.getText(redactionRequest.getDossierId(), redactionRequest.getFileId());
// enhance section grid with headline data
for (var sectionText : text.getSectionTexts()) {
sectionGrid.getSections()
.add(new SectionGrid.SectionGridSection(sectionText.getSectionNumber(), sectionText.getHeadline(), sectionText.getSectionAreas()
.stream()
.map(SectionArea::getPage)
.collect(Collectors.toSet()), sectionText.getSectionAreas()));
}
redactionStorageService.storeObject(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.SECTION_GRID, sectionGrid);
}
log.info("Loaded redaction log with computationalVersion: {}", redactionLog.getAnalysisVersion());
var merged = redactionLogMergeService.mergeRedactionLogData(redactionLog, sectionGrid, redactionRequest.getDossierTemplateId(), redactionRequest.getManualRedactions(), redactionRequest.getExcludedPages(), redactionRequest.getTypes(), redactionRequest.getColors());
merged.getRedactionLogEntry().removeIf(e -> e.isFalsePositive() && !redactionRequest.isIncludeFalsePositives());
return merged;
}
private RedactionResult convert(PDDocument document, int numberOfPages) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
@@ -176,4 +218,14 @@ public class RedactionController implements RedactionResource {
}
@Override
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions) {
var result = manualRedactionSurroundingTextService.addSurroundingText(dossierId, fileId, manualRedactions);
log.info("Added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", dossierId, fileId, result.getDuration());
return result.getManualRedactions();
}
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import com.iqser.red.service.redaction.v1.resources.RuleBuilderResource;
import com.iqser.red.service.redaction.v1.server.redaction.rulebuilder.RuleBuilderModelService;
import lombok.RequiredArgsConstructor;
import org.springframework.web.bind.annotation.RestController;
@RestController
@RequiredArgsConstructor
public class RuleBuilderController implements RuleBuilderResource {
private final RuleBuilderModelService ruleBuilderModelService;
@Override
public RuleBuilderModel getRuleBuilderModel() {
return ruleBuilderModelService.getRuleBuilderModel();
}
}
@@ -0,0 +1,9 @@
package com.iqser.red.service.redaction.v1.server.exception;
public class NotFoundException extends RuntimeException {
public NotFoundException(String message) {
super(message);
}
}
@@ -0,0 +1,377 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership.
* The ASF licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package com.iqser.red.service.redaction.v1.server.parsing;
import java.io.InputStream;
import java.io.IOException;
import java.util.Map;
import java.util.WeakHashMap;
import org.apache.commons.logging.Log;
import org.apache.commons.logging.LogFactory;
import org.apache.fontbox.ttf.TrueTypeFont;
import org.apache.fontbox.util.BoundingBox;
import org.apache.pdfbox.contentstream.PDFStreamEngine;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.font.encoding.GlyphList;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDCIDFont;
import org.apache.pdfbox.pdmodel.font.PDCIDFontType2;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.apache.pdfbox.pdmodel.font.PDSimpleFont;
import org.apache.pdfbox.pdmodel.font.PDTrueTypeFont;
import org.apache.pdfbox.pdmodel.font.PDType0Font;
import org.apache.pdfbox.pdmodel.font.PDType3Font;
import org.apache.pdfbox.pdmodel.graphics.state.PDGraphicsState;
import org.apache.pdfbox.text.TextPosition;
import org.apache.pdfbox.util.Matrix;
import org.apache.pdfbox.util.Vector;
import org.apache.pdfbox.contentstream.operator.DrawObject;
import org.apache.pdfbox.contentstream.operator.state.Concatenate;
import org.apache.pdfbox.contentstream.operator.state.Restore;
import org.apache.pdfbox.contentstream.operator.state.Save;
import org.apache.pdfbox.contentstream.operator.state.SetGraphicsStateParameters;
import org.apache.pdfbox.contentstream.operator.state.SetMatrix;
import org.apache.pdfbox.contentstream.operator.text.BeginText;
import org.apache.pdfbox.contentstream.operator.text.EndText;
import org.apache.pdfbox.contentstream.operator.text.SetFontAndSize;
import org.apache.pdfbox.contentstream.operator.text.SetTextHorizontalScaling;
import org.apache.pdfbox.contentstream.operator.text.ShowTextAdjusted;
import org.apache.pdfbox.contentstream.operator.text.ShowTextLine;
import org.apache.pdfbox.contentstream.operator.text.ShowTextLineAndSpace;
import org.apache.pdfbox.contentstream.operator.text.MoveText;
import org.apache.pdfbox.contentstream.operator.text.MoveTextSetLeading;
import org.apache.pdfbox.contentstream.operator.text.NextLine;
import org.apache.pdfbox.contentstream.operator.text.SetCharSpacing;
import org.apache.pdfbox.contentstream.operator.text.SetTextLeading;
import org.apache.pdfbox.contentstream.operator.text.SetTextRenderingMode;
import org.apache.pdfbox.contentstream.operator.text.SetTextRise;
import org.apache.pdfbox.contentstream.operator.text.SetWordSpacing;
import org.apache.pdfbox.contentstream.operator.text.ShowText;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.pdmodel.font.PDFontDescriptor;
/**
* LEGACY text calculations which are known to be incorrect but are depended on by PDFTextStripper.
*
* This class exists only so that we don't break the code of users who have their own subclasses of
* PDFTextStripper. It replaces the mostly empty implementation of showGlyph() in PDFStreamEngine
* with a heuristic implementation which is backwards compatible.
*
* DO NOT USE THIS CODE UNLESS YOU ARE WORKING WITH PDFTextStripper.
* THIS CODE IS DELIBERATELY INCORRECT, USE PDFStreamEngine INSTEAD.
*/
@SuppressWarnings({"PMD", "checkstyle:all"})
class LegacyPDFStreamEngine extends PDFStreamEngine
{
private static final Log LOG = LogFactory.getLog(LegacyPDFStreamEngine.class);
private int pageRotation;
private PDRectangle pageSize;
private Matrix translateMatrix;
private final GlyphList glyphList;
private final Map<COSDictionary, Float> fontHeightMap = new WeakHashMap<COSDictionary, Float>();
/**
* Constructor.
*/
LegacyPDFStreamEngine() throws IOException
{
addOperator(new BeginText());
addOperator(new Concatenate());
addOperator(new DrawObject()); // special text version
addOperator(new EndText());
addOperator(new SetGraphicsStateParameters());
addOperator(new Save());
addOperator(new Restore());
addOperator(new NextLine());
addOperator(new SetCharSpacing());
addOperator(new MoveText());
addOperator(new MoveTextSetLeading());
addOperator(new SetFontAndSize());
addOperator(new ShowText());
addOperator(new ShowTextAdjusted());
addOperator(new SetTextLeading());
addOperator(new SetMatrix());
addOperator(new SetTextRenderingMode());
addOperator(new SetTextRise());
addOperator(new SetWordSpacing());
addOperator(new SetTextHorizontalScaling());
addOperator(new ShowTextLine());
addOperator(new ShowTextLineAndSpace());
// load additional glyph list for Unicode mapping
String path = "/org/apache/pdfbox/resources/glyphlist/additional.txt";
InputStream input = GlyphList.class.getResourceAsStream(path);
glyphList = new GlyphList(GlyphList.getAdobeGlyphList(), input);
}
/**
* This will initialize and process the contents of the stream.
*
* @param page the page to process
* @throws java.io.IOException if there is an error accessing the stream.
*/
@Override
public void processPage(PDPage page) throws IOException
{
this.pageRotation = page.getRotation();
this.pageSize = page.getCropBox();
if (pageSize.getLowerLeftX() == 0 && pageSize.getLowerLeftY() == 0)
{
translateMatrix = null;
}
else
{
// translation matrix for cropbox
translateMatrix = Matrix.getTranslateInstance(-pageSize.getLowerLeftX(), -pageSize.getLowerLeftY());
}
super.processPage(page);
}
/**
* Called when a glyph is to be processed. The heuristic calculations here were originally
* written by Ben Litchfield for PDFStreamEngine.
*/
@Override
protected void showGlyph(Matrix textRenderingMatrix, PDFont font, int code,
String unicode,
Vector displacement)
throws IOException
{
//
// legacy calculations which were previously in PDFStreamEngine
//
// DO NOT USE THIS CODE UNLESS YOU ARE WORKING WITH PDFTextStripper.
// THIS CODE IS DELIBERATELY INCORRECT
//
PDGraphicsState state = getGraphicsState();
Matrix ctm = state.getCurrentTransformationMatrix();
float fontSize = state.getTextState().getFontSize();
float horizontalScaling = state.getTextState().getHorizontalScaling() / 100f;
Matrix textMatrix = getTextMatrix();
float displacementX = displacement.getX();
// the sorting algorithm is based on the width of the character. As the displacement
// for vertical characters doesn't provide any suitable value for it, we have to
// calculate our own
if (font.isVertical())
{
displacementX = font.getWidth(code) / 1000;
// there may be an additional scaling factor for true type fonts
TrueTypeFont ttf = null;
if (font instanceof PDTrueTypeFont)
{
ttf = ((PDTrueTypeFont)font).getTrueTypeFont();
}
else if (font instanceof PDType0Font)
{
PDCIDFont cidFont = ((PDType0Font)font).getDescendantFont();
if (cidFont instanceof PDCIDFontType2)
{
ttf = ((PDCIDFontType2)cidFont).getTrueTypeFont();
}
}
if (ttf != null && ttf.getUnitsPerEm() != 1000)
{
displacementX *= 1000f / ttf.getUnitsPerEm();
}
}
//
// legacy calculations which were previously in PDFStreamEngine
//
// DO NOT USE THIS CODE UNLESS YOU ARE WORKING WITH PDFTextStripper.
// THIS CODE IS DELIBERATELY INCORRECT
//
// (modified) combined displacement, this is calculated *without* taking the character
// spacing and word spacing into account, due to legacy code in TextStripper
float tx = displacementX * fontSize * horizontalScaling;
float ty = displacement.getY() * fontSize;
// (modified) combined displacement matrix
Matrix td = Matrix.getTranslateInstance(tx, ty);
// (modified) text rendering matrix
Matrix nextTextRenderingMatrix = td.multiply(textMatrix).multiply(ctm); // text space -> device space
float nextX = nextTextRenderingMatrix.getTranslateX();
float nextY = nextTextRenderingMatrix.getTranslateY();
// (modified) width and height calculations
float dxDisplay = nextX - textRenderingMatrix.getTranslateX();
Float fontHeight = fontHeightMap.get(font.getCOSObject());
if (fontHeight == null)
{
fontHeight = computeFontHeight(font);
fontHeightMap.put(font.getCOSObject(), fontHeight);
}
float dyDisplay = fontHeight * textRenderingMatrix.getScalingFactorY();
//
// start of the original method
//
// Note on variable names. There are three different units being used in this code.
// Character sizes are given in glyph units, text locations are initially given in text
// units, and we want to save the data in display units. The variable names should end with
// Text or Disp to represent if the values are in text or disp units (no glyph units are
// saved).
float glyphSpaceToTextSpaceFactor = 1 / 1000f;
if (font instanceof PDType3Font)
{
glyphSpaceToTextSpaceFactor = font.getFontMatrix().getScaleX();
}
float spaceWidthText = 0;
try
{
// to avoid crash as described in PDFBOX-614, see what the space displacement should be
spaceWidthText = font.getSpaceWidth() * glyphSpaceToTextSpaceFactor;
}
catch (Throwable exception)
{
LOG.warn(exception, exception);
}
if (spaceWidthText == 0)
{
spaceWidthText = font.getAverageFontWidth() * glyphSpaceToTextSpaceFactor;
// the average space width appears to be higher than necessary so make it smaller
spaceWidthText *= .80f;
}
if (spaceWidthText == 0)
{
spaceWidthText = 1.0f; // if could not find font, use a generic value
}
// the space width has to be transformed into display units
float spaceWidthDisplay = spaceWidthText * textRenderingMatrix.getScalingFactorX();
// use our additional glyph list for Unicode mapping
String unicodeMapping = font.toUnicode(code, glyphList);
// when there is no Unicode mapping available, Acrobat simply coerces the character code
// into Unicode, so we do the same. Subclasses of PDFStreamEngine don't necessarily want
// this, which is why we leave it until this point in PDFTextStreamEngine.
if (unicodeMapping == null)
{
if (font instanceof PDSimpleFont)
{
char c = (char) code;
unicodeMapping = new String(new char[] { c });
}
else
{
// Acrobat doesn't seem to coerce composite font's character codes, instead it
// skips them. See the "allah2.pdf" TestTextStripper file.
return;
}
}
// adjust for cropbox if needed
Matrix translatedTextRenderingMatrix;
if (translateMatrix == null)
{
translatedTextRenderingMatrix = textRenderingMatrix;
}
else
{
translatedTextRenderingMatrix = Matrix.concatenate(translateMatrix, textRenderingMatrix);
nextX -= pageSize.getLowerLeftX();
nextY -= pageSize.getLowerLeftY();
}
processTextPosition(new TextPosition(pageRotation, pageSize.getWidth(),
pageSize.getHeight(), translatedTextRenderingMatrix, nextX, nextY,
Math.abs(dyDisplay), dxDisplay,
Math.abs(spaceWidthDisplay), unicodeMapping, new int[] { code }, font,
fontSize,
(int)(fontSize * textMatrix.getScalingFactorX())));
}
/**
* Compute the font height. Override this if you want to use own calculations.
*
* @param font the font.
* @return the font height.
*
* @throws IOException if there is an error while getting the font bounding box.
*/
protected float computeFontHeight(PDFont font) throws IOException
{
BoundingBox bbox = font.getBoundingBox();
if (bbox.getLowerLeftY() < Short.MIN_VALUE)
{
// PDFBOX-2158 and PDFBOX-3130
// files by Salmat eSolutions / ClibPDF Library
bbox.setLowerLeftY(- (bbox.getLowerLeftY() + 65536));
}
// 1/2 the bbox is used as the height todo: why?
float glyphHeight = bbox.getHeight() / 2;
// sometimes the bbox has very high values, but CapHeight is OK
PDFontDescriptor fontDescriptor = font.getFontDescriptor();
if (fontDescriptor != null)
{
float capHeight = fontDescriptor.getCapHeight();
if (Float.compare(capHeight, 0) != 0 &&
(capHeight < glyphHeight || Float.compare(glyphHeight, 0) == 0))
{
glyphHeight = capHeight;
}
// PDFBOX-3464, PDFBOX-4480, PDFBOX-4553:
// sometimes even CapHeight has very high value, but Ascent and Descent are ok
float ascent = fontDescriptor.getAscent();
float descent = fontDescriptor.getDescent();
if (capHeight > ascent && ascent > 0 && descent < 0 &&
((ascent - descent) / 2 < glyphHeight || Float.compare(glyphHeight, 0) == 0))
{
glyphHeight = (ascent - descent) / 2;
}
}
// transformPoint from glyph space -> text space
float height;
if (font instanceof PDType3Font)
{
height = font.getFontMatrix().transformPoint(0, glyphHeight).y;
}
else
{
height = glyphHeight / 1000;
}
return height;
}
/**
* A method provided as an event interface to allow a subclass to perform some specific
* functionality when text needs to be processed.
*
* @param text The text to be processed.
*/
protected void processTextPosition(TextPosition text)
{
// subclasses can override to provide specific functionality
}
}
@@ -1,16 +1,33 @@
package com.iqser.red.service.redaction.v1.server.parsing;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import lombok.Getter;
import lombok.Setter;
import lombok.extern.slf4j.Slf4j;
import java.awt.geom.Point2D;
import java.awt.geom.Rectangle2D;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import org.apache.commons.lang3.reflect.FieldUtils;
import org.apache.pdfbox.contentstream.operator.Operator;
import org.apache.pdfbox.contentstream.operator.OperatorName;
import org.apache.pdfbox.contentstream.operator.color.*;
import org.apache.pdfbox.contentstream.operator.state.*;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColor;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColorN;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColorSpace;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceCMYKColor;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceGrayColor;
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceRGBColor;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColor;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColorN;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColorSpace;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceCMYKColor;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceGrayColor;
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceRGBColor;
import org.apache.pdfbox.contentstream.operator.state.SetFlatness;
import org.apache.pdfbox.contentstream.operator.state.SetLineCapStyle;
import org.apache.pdfbox.contentstream.operator.state.SetLineDashPattern;
import org.apache.pdfbox.contentstream.operator.state.SetLineJoinStyle;
import org.apache.pdfbox.contentstream.operator.state.SetLineMiterLimit;
import org.apache.pdfbox.contentstream.operator.state.SetLineWidth;
import org.apache.pdfbox.contentstream.operator.state.SetRenderingIntent;
import org.apache.pdfbox.contentstream.operator.text.SetFontAndSize;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSName;
@@ -19,16 +36,17 @@ import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.text.PDFTextStripper;
import org.apache.pdfbox.text.TextPosition;
import org.apache.pdfbox.util.Matrix;
import java.awt.geom.AffineTransform;
import java.awt.geom.Point2D;
import java.awt.geom.Rectangle2D;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.parsing.model.RedTextPosition;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import lombok.Getter;
import lombok.Setter;
import lombok.extern.slf4j.Slf4j;
@Slf4j
public class PDFLinesTextStripper extends PDFTextStripper {
@@ -106,11 +124,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
// The direction of vertical lines must always be from bottom to top for the table extraction algorithm.
if (pos.getY() > path_y) {
graphicsPath.add(new Ruling(new Point2D.Float(path_x, path_y), new Point2D.Float((float) pos.getX(), (float) pos
.getY())));
graphicsPath.add(new Ruling(new Point2D.Float(path_x, path_y), new Point2D.Float((float) pos.getX(), (float) pos.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float(path_x, (float) pos.getY()), new Point2D.Float((float) pos
.getX(), path_y)));
graphicsPath.add(new Ruling(new Point2D.Float(path_x, (float) pos.getY()), new Point2D.Float((float) pos.getX(), path_y)));
}
path_x = (float) pos.getX();
@@ -131,25 +147,19 @@ public class PDFLinesTextStripper extends PDFTextStripper {
Point2D p2 = transformPosition(x + width, y + height);
// Horizontal lines
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p2
.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p2
.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p2.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p2.getX(), (float) p2.getY())));
// Vertical lines, direction must always be from bottom to top for the table extraction algorithm.
if (p2.getY() > p1.getY()) {
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p1.getY()), new Point2D.Float((float) p2
.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p1.getY()), new Point2D.Float((float) p2.getX(), (float) p2.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p2.getY()), new Point2D.Float((float) p2
.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p2.getY()), new Point2D.Float((float) p2.getX(), (float) p1.getY())));
}
if (p2.getY() > p1.getY()) {
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p1
.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p1.getX(), (float) p2.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p1
.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p1.getX(), (float) p1.getY())));
}
}
break;
@@ -173,9 +183,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
graphicsPath.clear();
break;
case OperatorName.DRAW_OBJECT:
processImageOperation(arguments);
break;
// case OperatorName.DRAW_OBJECT:
// processImageOperation(arguments);
// break;
}
@@ -189,7 +199,7 @@ public class PDFLinesTextStripper extends PDFTextStripper {
COSName objectName = (COSName) arguments.get(0);
PDXObject xobject = getResources().getXObject(objectName);
if (xobject instanceof PDImageXObject) {
PDImageXObject image = (PDImageXObject)xobject;
PDImageXObject image = (PDImageXObject) xobject;
Matrix ctmNew = getGraphicsState().getCurrentTransformationMatrix();
Rectangle2D rect = new Rectangle2D.Float(ctmNew.getTranslateX(), ctmNew.getTranslateY(), ctmNew.getScaleX(), ctmNew.getScaleY());
@@ -198,7 +208,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
FieldUtils.writeField(image, "cachedImageSubsampling", -1, true);
if (rect.getHeight() > 2 && rect.getWidth() > 2) {
this.images.add(new PdfImage(image.getImage(), rect, pageNumber, image.getImage().getColorModel().hasAlpha()));
this.images.add(new PdfImage(image.getImage(), rect, pageNumber, image.getImage()
.getColorModel()
.hasAlpha()));
}
}
} catch (Exception e) {
@@ -207,8 +219,6 @@ public class PDFLinesTextStripper extends PDFTextStripper {
}
private float floatValue(COSBase value) {
if (value instanceof COSNumber) {
@@ -247,8 +257,16 @@ public class PDFLinesTextStripper extends PDFTextStripper {
public void writeString(String text, List<TextPosition> textPositions) throws IOException {
int startIndex = 0;
RedTextPosition previous = null;
for (int i = 0; i <= textPositions.size() - 1; i++) {
if (!textPositionSequences.isEmpty()) {
previous = textPositionSequences.get(textPositionSequences.size() - 1)
.getTextPositions()
.get(textPositionSequences.get(textPositionSequences.size() - 1).getTextPositions().size() - 1);
}
int charWidth = (int) textPositions.get(i).getWidthDirAdj();
if (charWidth < minCharWidth) {
minCharWidth = charWidth;
@@ -267,42 +285,58 @@ public class PDFLinesTextStripper extends PDFTextStripper {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0"))) {
.equals("\u00A0") || textPositions.get(i).getUnicode().equals("\t"))) {
startIndex++;
continue;
}
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
if (i > 0 && textPositions.get(i).getXDirAdj() < textPositions.get(i - 1).getXDirAdj()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i).getX() > textPositions.get(i - 1).getEndX() + 1) {
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i)
.getX() > textPositions.get(i - 1).getEndX() + 1) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0")) && i <= textPositions.size() - 2) {
.equals("\u00A0") || textPositions.get(i)
.getUnicode()
.equals("\t")) && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
.getUnicode()
.equals("\t")))) {
// Remove false sequence ends (whitespaces)
if (previous != null && sublist.get(0).getYDirAdj() == previous.getYDirAdj() && sublist.get(0)
.getXDirAdj() - (previous.getXDirAdj() + previous.getWidthDirAdj()) < 0.01) {
for (TextPosition t : sublist) {
textPositionSequences.get(textPositionSequences.size() - 1).add(t);
}
} else {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
}
startIndex = i + 1;
}
@@ -311,13 +345,23 @@ public class PDFLinesTextStripper extends PDFTextStripper {
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
.getUnicode()
.equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
.equals(" ") || sublist.get(sublist.size() - 1)
.getUnicode()
.equals("\u00A0") || sublist.get(sublist.size() - 1).getUnicode().equals("\t"))) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
.getUnicode()
.equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
.equals("\u00A0") || sublist.get(0).getUnicode().equals("\t")))) {
if (previous != null && sublist.get(0).getYDirAdj() == previous.getYDirAdj() && sublist.get(0)
.getXDirAdj() - (previous.getXDirAdj() + previous.getWidthDirAdj()) < 0.01) {
for (TextPosition t : sublist) {
textPositionSequences.get(textPositionSequences.size() - 1).add(t);
}
} else {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
}
super.writeString(text);
}
@@ -21,6 +21,8 @@ public class RedTextPosition {
private float YDirAdj;
private float width;
private float heightDir;
private float widthDirAdj;
private float dir;
// not used in reanalysis
@JsonIgnore
@@ -1,20 +1,24 @@
package com.iqser.red.service.redaction.v1.server.parsing.model;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import lombok.Data;
import lombok.NoArgsConstructor;
import org.apache.pdfbox.text.TextPosition;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Collectors;
import org.apache.pdfbox.text.TextPosition;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Data
@NoArgsConstructor
@JsonIgnoreProperties({ "empty" })
@JsonIgnoreProperties({"empty"})
public class TextPositionSequence implements CharSequence {
private int page;
@@ -23,12 +27,15 @@ public class TextPositionSequence implements CharSequence {
private float x1;
private float x2;
public TextPositionSequence(int page) {
this.page = page;
}
public static TextPositionSequence fromData(List<RedTextPosition> textPositions, int page) {
var textPositionSequence = new TextPositionSequence();
textPositionSequence.textPositions = textPositions;
textPositionSequence.page = page;
@@ -44,9 +51,6 @@ public class TextPositionSequence implements CharSequence {
}
@Override
public int length() {
@@ -129,6 +133,7 @@ public class TextPositionSequence implements CharSequence {
}
}
@JsonIgnore
public float getRotationAdjustedY() {
@@ -136,6 +141,13 @@ public class TextPositionSequence implements CharSequence {
}
@JsonIgnore
public float getRotationAdjustedX() {
return textPositions.get(0).getXDirAdj();
}
@JsonIgnore
public float getY1() {
@@ -181,10 +193,8 @@ public class TextPositionSequence implements CharSequence {
@JsonIgnore
public String getFont() {
return textPositions.get(0).getFontName()
.toLowerCase()
.replaceAll(",bold", "")
.replaceAll(",italic", "");
return textPositions.get(0).getFontName().toLowerCase().replaceAll(",bold", "").replaceAll(",italic", "");
}
@@ -205,27 +215,33 @@ public class TextPositionSequence implements CharSequence {
}
@JsonIgnore
public float getFontSize() {
return textPositions.get(0).getFontSizeInPt();
}
@JsonIgnore
public float getSpaceWidth() {
return textPositions.get(0).getWidthOfSpace();
}
@JsonIgnore
public int getRotation() {
return textPositions.get(0).getRotation();
}
@JsonIgnore
public Rectangle getRectangle() {
log.debug("Page: '{}', Word: '{}', Rotation: '{}'", page, toString(), textPositions.get(0).getRotation());
float height = getTextHeight();
float posXInit = getX1();
@@ -233,26 +249,59 @@ public class TextPositionSequence implements CharSequence {
float posYInit;
float posYEnd;
if (textPositions.get(0).getRotation() == 90) {
if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() != 0.0f) {
posXEnd = textPositions.get(0).getYDirAdj() + 2;
posYInit = getY1();
posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
} else if (textPositions.get(0).getRotation() == 270) {
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 270.0f) {
posYInit = textPositions.get(0).getPageHeight() - getX1();
posYEnd = textPositions.get(0).getPageHeight() - getX2() - textPositions.get(0)
.getWidth() - textPositions.get(textPositions.size() - 1).getWidth() - 1;
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + height;
} else if (textPositions.get(0).getRotation() == 270 && textPositions.get(0).getDir() == 0.0f) {
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = posYInit + 1;
posXInit = textPositions.get(0).getXDirAdj();
posXEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + 0.1f;
} else if (textPositions.get(0).getRotation() == 0 && textPositions.get(0).getDir() == 270f) {
posYInit = textPositions.get(0).getPageHeight() - getX1();
posYEnd = textPositions.get(0).getPageHeight() - getX2() - textPositions.get(0)
.getWidthDirAdj() - textPositions.get(textPositions.size() - 1).getWidthDirAdj() - 3;
posXInit = textPositions.get(0).getPageWidth() - textPositions.get(0).getYDirAdj() - 2;
posXEnd = textPositions.get(0).getPageWidth() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + height;
} else if (textPositions.get(0).getRotation() == 90 && textPositions.get(0).getDir() == 0.0f) {
posXInit = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getHeightDir();
posXEnd = textPositions.get(0).getXDirAdj();
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
} else if (textPositions.get(0).getRotation() == 0 && textPositions.get(0).getDir() == 90f) {
posYInit = getX1();
posYEnd = getX2() + textPositions.get(0).getWidthDirAdj() - textPositions.get(textPositions.size() - 1)
.getWidthDirAdj() - 3;
posXInit = textPositions.get(0).getYDirAdj() + 2;
posXEnd = textPositions.get(textPositions.size() - 1).getYDirAdj() - height;
} else {
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidthDirAdj() + 1;
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
}
return new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
var rectangle = new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
log.debug("Rectangle: {}", rectangle);
return rectangle;
}
}
@@ -0,0 +1,25 @@
package com.iqser.red.service.redaction.v1.server.queue;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_QUEUE;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
import com.fasterxml.jackson.core.JsonProcessingException;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RequiredArgsConstructor
public class MessageReceiver {
private final RedactionMessageReceiver redactionMessageReceiver;
@RabbitHandler
@RabbitListener(queues = REDACTION_QUEUE)
public void receiveAnalyzeRequest(String in) throws JsonProcessingException {
redactionMessageReceiver.receiveAnalyzeRequest(in, false);
}
}
@@ -1,8 +1,10 @@
package com.iqser.red.service.redaction.v1.server.queue;
import lombok.RequiredArgsConstructor;
import org.springframework.amqp.core.Queue;
import org.springframework.amqp.core.QueueBuilder;
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
@@ -10,11 +12,24 @@ import org.springframework.context.annotation.Configuration;
@RequiredArgsConstructor
public class MessagingConfiguration {
public static final String REDACTION_QUEUE = "redactionQueue";
public static final String REDACTION_DQL = "redactionDQL";
public static final String REDACTION_PRIORITY_QUEUE = "redactionPriorityQueue";
@Bean
@ConditionalOnProperty(prefix = "redaction-service", name = "priorityMode", havingValue = "false")
public MessageReceiver messageReceiver(RedactionMessageReceiver redactionMessageReceiver){
return new MessageReceiver(redactionMessageReceiver);
}
@Bean
@ConditionalOnProperty(prefix = "redaction-service", name = "priorityMode", havingValue = "true")
public PriorityMessageReceiver priorityMessageReceiver(RedactionMessageReceiver redactionMessageReceiver){
return new PriorityMessageReceiver(redactionMessageReceiver);
}
@Bean
public Queue redactionQueue() {
@@ -27,9 +42,21 @@ public class MessagingConfiguration {
}
@Bean
public Queue redactionPriorityQueue() {
return QueueBuilder.durable(REDACTION_PRIORITY_QUEUE)
.withArgument("x-dead-letter-exchange", "")
.withArgument("x-dead-letter-routing-key", REDACTION_DQL)
.maxPriority(2)
.build();
}
@Bean
public Queue redactionDeadLetterQueue() {
return QueueBuilder.durable(REDACTION_DQL).build();
}
}
@@ -0,0 +1,27 @@
package com.iqser.red.service.redaction.v1.server.queue;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_PRIORITY_QUEUE;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
import com.fasterxml.jackson.core.JsonProcessingException;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RequiredArgsConstructor
public class PriorityMessageReceiver {
private final RedactionMessageReceiver redactionMessageReceiver;
@RabbitHandler
@RabbitListener(queues = REDACTION_PRIORITY_QUEUE)
public void receiveAnalyzeRequest(String in) throws JsonProcessingException {
redactionMessageReceiver.receiveAnalyzeRequest(in, true);
}
}
@@ -1,19 +1,23 @@
package com.iqser.red.service.redaction.v1.server.queue;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_DQL;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_QUEUE;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
import org.springframework.stereotype.Service;
import com.fasterxml.jackson.core.JsonProcessingException;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.StructureAnalyzeRequest;
import com.iqser.red.service.redaction.v1.server.client.FileStatusProcessingUpdateClient;
import com.iqser.red.service.redaction.v1.server.redaction.service.ReanalyzeService;
import com.iqser.red.service.redaction.v1.server.redaction.service.AnalyzeService;
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.amqp.rabbit.annotation.RabbitHandler;
import org.springframework.amqp.rabbit.annotation.RabbitListener;
import org.springframework.stereotype.Service;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_DQL;
import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfiguration.REDACTION_QUEUE;
@Slf4j
@Service
@@ -21,26 +25,45 @@ import static com.iqser.red.service.redaction.v1.server.queue.MessagingConfigura
public class RedactionMessageReceiver {
private final ObjectMapper objectMapper;
private final ReanalyzeService reanalyzeService;
private final AnalyzeService analyzeService;
private final FileStatusProcessingUpdateClient fileStatusProcessingUpdateClient;
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
@RabbitHandler
@RabbitListener(queues = REDACTION_QUEUE)
public void receiveAnalyzeRequest(String in) throws JsonProcessingException {
public void receiveAnalyzeRequest(String in, boolean priority) throws JsonProcessingException {
var analyzeRequest = objectMapper.readValue(in, AnalyzeRequest.class);
log.info("Processing analyze request: {}", analyzeRequest);
AnalyzeResult result;
if (analyzeRequest.isReanalyseOnlyIfPossible()) {
result = reanalyzeService.reanalyze(analyzeRequest);
} else {
result = reanalyzeService.analyze(analyzeRequest);
}
log.info("Successfully analyzed {}", analyzeRequest);
log.info("Processing priority: {} analyze request for file: {}", priority, analyzeRequest.getFileId());
AnalyzeResult result = null;
switch (analyzeRequest.getMessageType()) {
case REANALYSE:
result = analyzeService.reanalyze(analyzeRequest);
log.info("Successfully reanalyzed dossier {} file {} took: {}", analyzeRequest.getDossierId(), analyzeRequest.getFileId(), result.getDuration());
break;
case STRUCTURE_ANALYSE:
result = analyzeService.analyzeDocumentStructure(new StructureAnalyzeRequest(analyzeRequest.getDossierId(), analyzeRequest.getFileId()));
log.info("Successfully analyzed structure dossier {} file {} took: {}", analyzeRequest.getDossierId(), analyzeRequest.getFileId(), result.getDuration());
break;
case ANALYSE:
result = analyzeService.analyze(analyzeRequest);
log.info("Successfully analyzed dossier {} file {} took: {}", analyzeRequest.getDossierId(), analyzeRequest.getFileId(), result.getDuration());
break;
case SURROUNDING_TEXT:
result = manualRedactionSurroundingTextService.addSurroundingText(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), analyzeRequest.getManualRedactions());
log.info("Successfully added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", analyzeRequest.getDossierId(), analyzeRequest.getFileId(), result.getDuration());
break;
}
result.setMessageType(analyzeRequest.getMessageType());
fileStatusProcessingUpdateClient.analysisSuccessful(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), result);
}
@RabbitHandler
@RabbitListener(queues = REDACTION_DQL)
public void receiveAnalyzeRequestDQL(String in) throws JsonProcessingException {
@@ -1,18 +1,16 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import lombok.Data;
import lombok.Getter;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import lombok.Data;
import lombok.Getter;
@Data
public class Dictionary {
public static final String RECOMMENDATION_PREFIX = "recommendation_";
@Getter
private List<DictionaryModel> dictionaryModels;
private Map<String, DictionaryModel> localAccessMap = new HashMap<>();
@@ -22,6 +20,7 @@ public class Dictionary {
public Dictionary(List<DictionaryModel> dictionaryModels, DictionaryVersion version) {
this.dictionaryModels = dictionaryModels;
this.dictionaryModels.forEach(dm -> localAccessMap.put(dm.getType(), dm));
this.version = version;
@@ -29,6 +28,7 @@ public class Dictionary {
public int getDictionaryRank(String type) {
if (!localAccessMap.containsKey(type)) {
return 0;
}
@@ -36,16 +36,6 @@ public class Dictionary {
}
public boolean isRecommendation(String type) {
DictionaryModel model = localAccessMap.get(type);
if (model != null) {
return model.isRecommendation();
}
return false;
}
public boolean hasLocalEntries() {
return dictionaryModels.stream().anyMatch(dm -> !dm.getLocalEntries().isEmpty());
@@ -58,16 +48,18 @@ public class Dictionary {
}
public DictionaryModel getType(String type) {
return localAccessMap.get(type);
}
public boolean containsValue(String type, String value) {
return localAccessMap.containsKey(type) && localAccessMap.get(type)
.getEntries()
.getValues(false)
.contains(value) || localAccessMap.containsKey(type) && localAccessMap.get(type)
.getLocalEntries()
.contains(value) || localAccessMap.containsKey(RECOMMENDATION_PREFIX + type) && localAccessMap.get(RECOMMENDATION_PREFIX + type)
.getEntries()
.contains(value) || localAccessMap.containsKey(RECOMMENDATION_PREFIX + type) && localAccessMap.get(RECOMMENDATION_PREFIX + type)
.getLocalEntries()
.getValues(true)
.contains(value);
}
@@ -81,6 +73,7 @@ public class Dictionary {
return false;
}
public boolean isCaseInsensitiveDictionary(String type) {
DictionaryModel dictionaryModel = localAccessMap.get(type);
@@ -0,0 +1,26 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.HashSet;
import java.util.Set;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class DictionaryEntries {
@Builder.Default
Set<DictionaryEntry> entries = new HashSet<>();
@Builder.Default
Set<DictionaryEntry> falsePositives = new HashSet<>();
@Builder.Default
Set<DictionaryEntry> falseRecommendations = new HashSet<>();
}
@@ -1,7 +1,7 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import com.iqser.red.service.configuration.v1.api.model.DictionaryEntry;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
import lombok.AllArgsConstructor;
import lombok.Data;
@@ -9,6 +9,7 @@ import java.io.Serializable;
import java.util.Set;
import java.util.stream.Collectors;
@Data
@AllArgsConstructor
public class DictionaryModel implements Serializable {
@@ -18,8 +19,9 @@ public class DictionaryModel implements Serializable {
private float[] color;
private boolean caseInsensitive;
private boolean hint;
private boolean recommendation;
private Set<DictionaryEntry> entries;
private Set<DictionaryEntry> falsePositives;
private Set<DictionaryEntry> falseRecommendations;
private Set<String> localEntries;
private boolean isDossierDictionary;
@@ -28,4 +30,14 @@ public class DictionaryModel implements Serializable {
.toSet());
}
public Set<String> getFalsePositiveValues() {
return falsePositives.stream().filter(e -> !e.isDeleted()).map(e -> e.getValue()).collect(Collectors
.toSet());
}
public Set<String> getFalseRecommendationValues() {
return falseRecommendations.stream().filter(e -> !e.isDeleted()).map(e -> e.getValue()).collect(Collectors
.toSet());
}
}
@@ -0,0 +1,23 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.HashSet;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class Entities {
@Builder.Default
private Set<Entity> entities = new HashSet<>();
@Builder.Default
private Set<Entity> nerEntities = new HashSet<>();
}
@@ -1,19 +1,22 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import com.iqser.red.service.redaction.v1.model.Engine;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.Data;
import lombok.EqualsAndHashCode;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
@Data
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
public class Entity implements ReasonHolder {
private final String word;
private final String type;
private String word;
private String type;
private boolean redaction;
private String redactionReason;
private String legalBasis;
@@ -39,8 +42,19 @@ public class Entity implements ReasonHolder {
private boolean isDossierDictionaryEntry;
private boolean ignored;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber, String legalBasis, boolean isDictionaryEntry, String textBefore, String textAfter, Integer start, Integer end, boolean isDossierDictionaryEntry) {
private Set<Engine> engines = new HashSet<>();
private Set<Entity> references = new HashSet<>();
private EntityType entityType;
public Entity(String word, String type, boolean redaction, String redactionReason,
List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber,
String legalBasis, boolean isDictionaryEntry, String textBefore, String textAfter, Integer start,
Integer end, boolean isDossierDictionaryEntry, Set<Engine> engines, Set<Entity> references, EntityType entityType) {
this.word = word;
this.type = type;
@@ -57,10 +71,14 @@ public class Entity implements ReasonHolder {
this.start = start;
this.end = end;
this.isDossierDictionaryEntry = isDossierDictionaryEntry;
this.engines = engines;
this.references = references;
this.entityType = entityType;
}
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber, boolean isDictionaryEntry, boolean isDossierDictionaryEntry) {
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber,
boolean isDictionaryEntry, boolean isDossierDictionaryEntry, Engine engine, EntityType entityType) {
this.word = word;
this.type = type;
@@ -70,6 +88,9 @@ public class Entity implements ReasonHolder {
this.sectionNumber = sectionNumber;
this.isDictionaryEntry = isDictionaryEntry;
this.isDossierDictionaryEntry = isDossierDictionaryEntry;
this.engines.add(engine);
this.entityType = entityType;
}
}
@@ -0,0 +1,5 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
public enum EntityType {
ENTITY, RECOMMENDATION, FALSE_POSITIVE, FALSE_RECOMMENDATION
}
@@ -20,6 +20,7 @@ public class Image implements ReasonHolder {
private int sectionNumber;
private String section;
private int page;
private boolean ignored;
private boolean hasTransparency;
}
@@ -0,0 +1,23 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@Data
@AllArgsConstructor
public class PageEntities {
@Builder.Default
private Map<Integer, List<Entity>> entitiesPerPage = new HashMap<>();
@Builder.Default
private Map<Integer, Set<Image>> imagesPerPage = new HashMap<>();
}
@@ -16,6 +16,7 @@ public class PdfImage {
private BufferedImage image;
@NonNull
private RedRectangle2D position;
@NonNull
private ImageType imageType;
private boolean isAppendedToParagraph;
private boolean hasTransparency;
@@ -1,16 +1,13 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import com.iqser.red.service.redaction.v1.model.FileAttribute;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
import lombok.Builder;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
import org.apache.commons.lang3.StringUtils;
import java.lang.annotation.ElementType;
import java.lang.annotation.Retention;
import java.lang.annotation.RetentionPolicy;
import java.lang.annotation.Target;
import java.util.ArrayList;
import java.util.Collection;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
@@ -20,7 +17,18 @@ import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import static com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary.RECOMMENDATION_PREFIX;
import org.apache.commons.lang3.StringUtils;
import com.iqser.red.service.redaction.v1.model.ArgumentType;
import com.iqser.red.service.redaction.v1.model.Engine;
import com.iqser.red.service.redaction.v1.model.FileAttribute;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
import lombok.Builder;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
@Data
@Slf4j
@@ -36,6 +44,8 @@ public class Section {
private Set<Entity> entities;
private Set<Entity> nerEntities;
// This still contains linebreaks etc.
private String text;
@@ -59,57 +69,206 @@ public class Section {
private List<FileAttribute> fileAttributes = new ArrayList<>();
public boolean fileAttributeByIdEquals(String id, String value){
return fileAttributes != null && fileAttributes.stream().filter(attribute -> id.equals(attribute.getId()) && value.equals(attribute.getValue())).findFirst().isPresent();
}
public void addAiEntities(String type, String asType) {
public boolean fileAttributeByPlaceholderEquals(String placeholder, String value){
return fileAttributes != null && fileAttributes.stream().filter(attribute -> placeholder.equals(attribute.getPlaceholder()) && value.equals(attribute.getValue())).findFirst().isPresent();
}
Set<Entity> entitiesOfType = nerEntities.stream().filter(nerEntity -> nerEntity.getType().equals(type)).collect(Collectors.toSet());
Set<String> values = entitiesOfType.stream().map(Entity::getWord).collect(Collectors.toSet());
Set<Entity> found = EntitySearchUtils.findEntities(searchText, values, dictionary.getType(asType), headline, sectionNumber, false, false, Engine.NER, true, true);
EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary);
public boolean fileAttributeByLabelEquals(String label, String value){
return fileAttributes != null && fileAttributes.stream().filter(attribute -> label.equals(attribute.getLabel()) && value.equals(attribute.getValue())).findFirst().isPresent();
Set<Entity> finalResult = new HashSet<>();
// Only keep Entities with correct offsets from AI Service.
for (Entity current : found) {
boolean foundSameOffsets = false;
for (Entity entity : entitiesOfType) {
if (entity.getStart().equals(current.getStart()) && entity.getEnd().equals(current.getEnd())) {
foundSameOffsets = true;
break;
}
}
if (foundSameOffsets) {
finalResult.add(current);
}
}
var nonOverlapping = EntitySearchUtils.findNonOverlappingMatchEntities(entities, finalResult);
EntitySearchUtils.addEntitiesWithHigherRank(entities, nonOverlapping, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
nerEntities.removeAll(entitiesOfType);
}
public boolean rowEquals(String headerName, String value) {
public void combineAiTypes(String startType, String combineTypes, int maxDistanceBetween, String asType, int minPartMatches, boolean allowDuplicateTypes) {
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
Set<String> combineSet = Set.of(combineTypes.split(","));
return tabularData != null && tabularData.containsKey(cleanHeaderName) && tabularData.get(cleanHeaderName)
.toString()
.equals(value);
List<Entity> sorted = nerEntities.stream().sorted(Comparator.comparing(Entity::getStart)).collect(Collectors.toList());
Set<Entity> found = new HashSet<>();
int start = -1;
int lastEnd = -1;
int numberOfMatchParts = 0;
Set<String> foundParts = new HashSet<>();
for (Entity entity : sorted) {
if (entity.getType().equals(startType) && start == -1) {
lastEnd = entity.getEnd();
start = entity.getStart();
foundParts.add(entity.getType());
numberOfMatchParts++;
} else if (!allowDuplicateTypes && foundParts.contains(entity.getType())) {
if (numberOfMatchParts >= minPartMatches) {
String value = searchText.substring(start, lastEnd);
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
}
start = -1;
lastEnd = -1;
numberOfMatchParts = 0;
foundParts = new HashSet<>();
if (entity.getType().equals(startType)) {
lastEnd = entity.getEnd();
start = entity.getStart();
foundParts.add(entity.getType());
numberOfMatchParts++;
}
} else if (entity.getType().equals(startType) && start != -1) {
if (numberOfMatchParts >= minPartMatches) {
String value = searchText.substring(start, lastEnd);
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
}
start = entity.getStart();
lastEnd = entity.getEnd();
numberOfMatchParts = 0;
foundParts = new HashSet<>();
foundParts.add(entity.getType());
numberOfMatchParts++;
} else if (start != -1 && combineSet.contains(entity.getType()) && entity.getStart() - lastEnd < maxDistanceBetween) {
lastEnd = entity.getEnd();
numberOfMatchParts++;
foundParts.add(entity.getType());
}
}
if (numberOfMatchParts >= minPartMatches) {
String value = searchText.substring(start, lastEnd);
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
}
if (!found.isEmpty()) {
var nonOverlapping = EntitySearchUtils.findNonOverlappingMatchEntities(entities, found);
EntitySearchUtils.addEntitiesWithHigherRank(entities, nonOverlapping, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
}
}
public boolean hasTableHeader(String headerName) {
@WhenCondition
public boolean fileAttributeByIdEquals(@Argument(ArgumentType.FILE_ATTRIBUTE) String id, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> id.equals(attribute.getId()) && value.equals(attribute.getValue()));
}
@WhenCondition
public boolean fileAttributeByPlaceholderEquals(@Argument(ArgumentType.FILE_ATTRIBUTE) String placeholder, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> placeholder.equals(attribute.getPlaceholder()) && value.equals(attribute.getValue()));
}
@WhenCondition
public boolean fileAttributeByLabelEquals(@Argument(ArgumentType.FILE_ATTRIBUTE) String label, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && value.equals(attribute.getValue()));
}
@WhenCondition
public boolean fileAttributeByIdEqualsIgnoreCase(@Argument(ArgumentType.FILE_ATTRIBUTE) String id, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> id.equals(attribute.getId()) && value.equalsIgnoreCase(attribute.getValue()));
}
@WhenCondition
public boolean fileAttributeByPlaceholderEqualsIgnoreCase(@Argument(ArgumentType.FILE_ATTRIBUTE) String placeholder, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream()
.anyMatch(attribute -> placeholder.equals(attribute.getPlaceholder()) && value.equalsIgnoreCase(attribute.getValue()));
}
@WhenCondition
public boolean fileAttributeByLabelEqualsIgnoreCase(@Argument(ArgumentType.FILE_ATTRIBUTE) String label, @Argument(ArgumentType.STRING) String value) {
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && value.equalsIgnoreCase(attribute.getValue()));
}
@WhenCondition
public boolean hasTableHeader(@Argument(ArgumentType.STRING) String headerName) {
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
return tabularData != null && tabularData.containsKey(cleanHeaderName);
}
public boolean matchesType(String type) {
@WhenCondition
public boolean aiMatchesType(@Argument(ArgumentType.TYPE) String type) {
return entities.stream().anyMatch(entity -> entity.getType().equals(type));
return nerEntities.stream().anyMatch(entity -> !entity.isIgnored() && entity.getType().equals(type));
}
public boolean matchesImageType(String type) {
@WhenCondition
public boolean matchesType(@Argument(ArgumentType.TYPE) String type) {
return images.stream().anyMatch(image -> image.getType().equals(type));
return entities.stream().anyMatch(entity -> !entity.isIgnored() && entity.getType().equals(type));
}
public boolean headlineContainsWord(String word) {
@WhenCondition
public boolean matchesImageType(@Argument(ArgumentType.TYPE) String type) {
return images.stream().anyMatch(image -> !image.isIgnored() && image.getType().equals(type));
}
@WhenCondition
public boolean headlineContainsWord(@Argument(ArgumentType.STRING) String word) {
return StringUtils.containsIgnoreCase(headline, word);
}
public void expandByRegEx(String type, String pattern, boolean patternCaseInsensitive, int group) {
@WhenCondition
public boolean rowEquals(@Argument(ArgumentType.STRING) String headerName, @Argument(ArgumentType.STRING) String value) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
return tabularData != null && tabularData.containsKey(cleanHeaderName) && tabularData.get(cleanHeaderName).toString().equals(value);
}
@ThenAction
public void expandByRegEx(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.REGEX) String suffixPattern,
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, @Argument(ArgumentType.INTEGER) int group) {
expandByRegEx(type, suffixPattern, patternCaseInsensitive, group, null);
}
@ThenAction
public void expandByRegEx(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.REGEX) String suffixPattern,
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, @Argument(ArgumentType.INTEGER) int group,
@Argument(ArgumentType.REGEX) String valuePattern) {
Pattern compiledValuePattern = null;
if (valuePattern != null) {
compiledValuePattern = Patterns.getCompiledPattern(valuePattern, patternCaseInsensitive);
}
Pattern compiledPattern = Patterns.getCompiledPattern(suffixPattern, patternCaseInsensitive);
Set<Entity> expanded = new HashSet<>();
for (Entity entity : entities) {
@@ -118,14 +277,21 @@ public class Section {
continue;
}
if (valuePattern != null) {
Matcher valueMatcher = compiledValuePattern.matcher(entity.getWord());
if (!valueMatcher.matches()) {
continue;
}
}
Matcher matcher = compiledPattern.matcher(entity.getTextAfter());
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
expanded.addAll(findEntities(entity.getWord() + match, type, false, entity.isRedaction(), entity.getMatchedRule(), entity
.getRedactionReason(), entity.getLegalBasis()));
if (StringUtils.isNotBlank(match)) {
Set<Entity> expandedEntities = findEntities(entity.getWord() + match, type, false, entity.isRedaction(), entity.getMatchedRule(), entity.getRedactionReason(), entity.getLegalBasis(), Engine.RULE, false);
expanded.addAll(EntitySearchUtils.findNonOverlappingMatchEntities(entities, expandedEntities));
}
}
}
@@ -135,106 +301,181 @@ public class Section {
}
public void redactImage(String type, int ruleNumber, String reason, String legalBasis) {
@ThenAction
public void redactImage(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason,
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
images.forEach(image -> {
if (image.getType().equals(type)) {
image.setRedaction(true);
image.setMatchedRule(ruleNumber);
image.setRedactionReason(reason);
image.setLegalBasis(legalBasis);
}
});
redactImage(type, ruleNumber, reason, legalBasis, true);
}
public void redact(String type, int ruleNumber, String reason, String legalBasis) {
@ThenAction
public void redactNotImage(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason) {
boolean hasRecommendationDictionary = dictionaryTypes.contains(RECOMMENDATION_PREFIX + type);
entities.forEach(entity -> {
if (entity.getType().equals(type) || hasRecommendationDictionary && entity.getType()
.equals(RECOMMENDATION_PREFIX + type)) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
redactImage(type, ruleNumber, reason, null, false);
}
public void redactNotImage(String type, int ruleNumber, String reason) {
@ThenAction
public void redact(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason,
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
images.forEach(image -> {
if (image.getType().equals(type)) {
image.setRedaction(false);
image.setMatchedRule(ruleNumber);
image.setRedactionReason(reason);
}
});
redact(type, ruleNumber, reason, legalBasis, true);
}
public void redactNot(String type, int ruleNumber, String reason) {
@ThenAction
public void redactNot(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason) {
boolean hasRecommendationDictionary = dictionaryTypes.contains(RECOMMENDATION_PREFIX + type);
entities.forEach(entity -> {
if (entity.getType().equals(type) || hasRecommendationDictionary && entity.getType()
.equals(RECOMMENDATION_PREFIX + type)) {
entity.setRedaction(false);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
redact(type, ruleNumber, reason, null, false);
}
public void expandToHintAnnotationByRegEx(String type, String pattern, boolean patternCaseInsensitive, int group,
String asType) {
@ThenAction
public void redactLineAfter(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere, @Argument(ArgumentType.STRING) String reason,
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, true);
}
@ThenAction
public void redactNotLineAfter(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere, @Argument(ArgumentType.STRING) String reason) {
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, null, false);
}
@ThenAction
public void redactByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true);
}
@ThenAction
public void redactNotByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.STRING) String reason) {
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false);
}
@ThenAction
public void redactBetween(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.STRING) String stop, @Argument(ArgumentType.TYPE) String asType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
redactBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, legalBasis, true);
}
@ThenAction
public void redactNotBetween(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.STRING) String stop, @Argument(ArgumentType.TYPE) String asType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
@Argument(ArgumentType.STRING) String reason) {
redactBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, null, false);
}
@ThenAction
public void redactLinesBetween(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.STRING) String stop, @Argument(ArgumentType.TYPE) String asType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, legalBasis, true);
}
@ThenAction
public void redactNotLinesBetween(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.STRING) String stop, @Argument(ArgumentType.TYPE) String asType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
@Argument(ArgumentType.STRING) String reason) {
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, null, false);
}
@ThenAction
public void redactCell(@Argument(ArgumentType.STRING) String cellHeader, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.TYPE) String type,
@Argument(ArgumentType.BOOLEAN) boolean addAsRecommendations, @Argument(ArgumentType.STRING) String reason,
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis);
}
@ThenAction
public void redactNotCell(@Argument(ArgumentType.STRING) String cellHeader, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.TYPE) String type,
@Argument(ArgumentType.BOOLEAN) boolean addAsRecommendations, @Argument(ArgumentType.STRING) String reason) {
annotateCell(cellHeader, ruleNumber, type, false, addAsRecommendations, reason, null);
}
@ThenAction
public void redactAndRecommendByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true);
}
@ThenAction
public void redactNotAndRecommendByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason) {
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false);
}
@ThenAction
public void addRecommendationByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Set<Entity> expanded = new HashSet<>();
for (Entity entity : entities) {
if (!entity.getType().equals(type) || entity.getTextAfter() == null) {
continue;
}
Matcher matcher = compiledPattern.matcher(entity.getTextAfter());
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
expanded.addAll(findEntities(entity.getWord() + match, asType, false, false, 0, null, null));
}
}
}
EntitySearchUtils.addEntitiesWithHigherRank(entities, expanded, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
}
public void addHintAnnotationByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
Matcher matcher = compiledPattern.matcher(text);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, false, 0, null, null);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(match);
}
}
}
public void redactIfPrecededBy(String prefix, String type, int ruleNumber, String reason, String legalBasis) {
@ThenAction
public void redactNotAndReference(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.REFERENCE_TYPE) String referenceType,
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason) {
Set<Entity> references = entities.stream().filter(entity -> entity.getType().equals(referenceType)).collect(Collectors.toSet());
entities.forEach(entity -> {
if (entity.getType().equals(type)) {
entity.setRedaction(false);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setReferences(references);
}
});
}
@ThenAction
public void redactIfPrecededBy(@Argument(ArgumentType.STRING) String prefix, @Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
entities.forEach(entity -> {
if (entity.getType().equals(type) && searchText.indexOf(prefix + entity.getWord()) != 1) {
@@ -247,41 +488,84 @@ public class Section {
}
public void addHintAnnotation(String value, String asType) {
@ThenAction
public void addRedaction(@Argument(ArgumentType.STRING) String value, @Argument(ArgumentType.TYPE) String asType, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
@Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null);
Set<Entity> found = findEntities(value.trim(), asType, true, true, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
}
public void addRedaction(String value, String asType, int ruleNumber, String reason, String legalBasis) {
public void ignore(String type) {
Set<Entity> found = findEntities(value.trim(), asType, true, true, ruleNumber, reason, legalBasis);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
entities.removeIf(entity -> entity.getType().equals(type) && entity.getEntityType().equals(EntityType.ENTITY));
}
public void redactLineAfter(String start, String asType, int ruleNumber, boolean redactEverywhere, String reason,
String legalBasis) {
public void ignoreRecommendations(String type) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
entities.removeIf(entity -> entity.getType().equals(type) && entity.getEntityType().equals(EntityType.RECOMMENDATION));
}
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
@ThenAction
public void expandToFalsePositiveByRegEx(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.STRING) String pattern,
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, @Argument(ArgumentType.INTEGER) int group) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Set<Entity> expanded = new HashSet<>();
for (Entity entity : entities) {
if (!entity.getType().equals(type) || entity.getTextAfter() == null) {
continue;
}
Matcher matcher = compiledPattern.matcher(entity.getTextAfter());
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
expanded.addAll(findEntities(entity.getWord() + match, type, false, false, 0, null, null, Engine.RULE, false));
}
}
}
EntitySearchUtils.addEntitiesWithHigherRank(entities, expanded, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
expanded.forEach(e -> e.setEntityType(EntityType.FALSE_POSITIVE));
}
@ThenAction
public void addHintAnnotationByRegEx(@Argument(ArgumentType.REGEX) String pattern, @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
@Argument(ArgumentType.INTEGER) int group, @Argument(ArgumentType.TYPE) String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, false, 0, null, null, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
}
}
}
public void recommendLineAfter(String start, String asType) {
@ThenAction
public void addHintAnnotation(@Argument(ArgumentType.STRING) String value, @Argument(ArgumentType.TYPE) String asType) {
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null, Engine.RULE, false);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
}
@ThenAction
public void recommendLineAfter(@Argument(ArgumentType.STRING) String start, @Argument(ArgumentType.TYPE) String asType) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
@@ -297,140 +581,42 @@ public class Section {
}
if (StringUtils.isNotBlank(cleanValue) && cleanValue.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>())
.add(cleanValue);
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(cleanValue);
}
}
}
}
public void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber,
String reason, String legalBasis) {
@ThenAction
public void highlightCell(@Argument(ArgumentType.STRING) String cellHeader, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.TYPE) String type) {
annotateCell(cellHeader, ruleNumber, type, false, false, null, null);
}
private void redactAndRecommendByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis,
boolean redaction) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, true, ruleNumber, reason, legalBasis);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
}
}
}
public void addRecommendationByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(text);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>()).add(match);
}
}
}
public void redactAndRecommendByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType,
int ruleNumber, String reason, String legalBasis) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>()).add(match);
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(match);
}
}
}
public void redactBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere,
String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(searchText, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
}
}
}
}
public void redactLinesBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere,
String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(text, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
String[] lines = value.split("\n");
for (String line : lines) {
if (line.trim().length() <= 2) {
return;
}
Set<Entity> found = findEntities(line.trim(), asType, false, true, ruleNumber, reason, legalBasis);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(line.trim());
}
}
}
}
}
}
public void highlightCell(String cellHeader, int ruleNumber, String type) {
annotateCell(cellHeader, ruleNumber, type, false, false, null, null);
}
public void redactCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations, String reason,
String legalBasis) {
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis);
}
public void redactNotCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations,
String reason) {
annotateCell(cellHeader, ruleNumber, type, false, addAsRecommendations, reason, null);
}
private Set<Entity> findEntities(String value, String asType, boolean caseInsensitive, boolean redacted,
int ruleNumber, String reason, String legalBasis) {
private Set<Entity> findEntities(String value, String asType, boolean caseInsensitive, boolean redacted, int ruleNumber, String reason, String legalBasis, Engine engine, boolean asRecommendation) {
String text = caseInsensitive ? searchText.toLowerCase() : searchText;
String searchValue = caseInsensitive ? value.toLowerCase() : value;
Set<Entity> found = EntitySearchUtils.find(text, Set.of(searchValue), asType, headline, sectionNumber, true, false);
Set<Entity> found = EntitySearchUtils.findEntities(text, Set.of(searchValue), dictionary.getType(asType), headline, sectionNumber, false, false, engine, false, asRecommendation);
found.forEach(entity -> {
if (redacted) {
@@ -445,8 +631,33 @@ public class Section {
}
private void annotateCell(String cellHeader, int ruleNumber, String type, boolean redact,
boolean addAsRecommendations, String reason, String legalBasis) {
private void redact(String type, int ruleNumber, String reason, String legalBasis, boolean redaction) {
entities.forEach(entity -> {
if (entity.getType().equals(type)) {
entity.setRedaction(redaction);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
private void redactImage(String type, int ruleNumber, String reason, String legalBasis, boolean redaction) {
images.forEach(image -> {
if (image.getType().equals(type)) {
image.setRedaction(redaction);
image.setMatchedRule(ruleNumber);
image.setRedactionReason(reason);
image.setLegalBasis(legalBasis);
}
});
}
private void annotateCell(String cellHeader, int ruleNumber, String type, boolean redact, boolean addAsRecommendations, String reason, String legalBasis) {
String cleanHeaderName = cellHeader.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
@@ -456,7 +667,7 @@ public class Section {
} else {
String word = value.toString();
Entity entity = new Entity(word, type, value.getRowSpanStart(), value.getRowSpanStart() + word.length(), headline, sectionNumber, false, false);
Entity entity = new Entity(word, type, value.getRowSpanStart(), value.getRowSpanStart() + word.length(), headline, sectionNumber, false, false, Engine.RULE, EntityType.ENTITY);
entity.setRedaction(redact);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
@@ -483,17 +694,119 @@ public class Section {
while (matcher.find()) {
String match = matcher.group().trim();
if (match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + type, (x) -> new HashSet<>())
.add(match);
localDictionaryAdds.computeIfAbsent(type, (x) -> new HashSet<>()).add(match);
String lastname = match.split(" ")[0];
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + type, (x) -> new HashSet<>())
.add(lastname);
localDictionaryAdds.computeIfAbsent(type, (x) -> new HashSet<>()).add(lastname);
}
}
}
}
}
private void redactLineAfter(String start, String asType, int ruleNumber, boolean redactEverywhere, String reason, String legalBasis, boolean redaction) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
}
}
}
}
private void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis, boolean redaction) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
}
}
}
private void redactBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere, String reason, String legalBasis, boolean redaction) {
String[] values = StringUtils.substringsBetween(searchText, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
}
}
}
}
private void redactLinesBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere, String reason, String legalBasis, boolean redaction) {
String[] values = StringUtils.substringsBetween(text, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
String[] lines = value.split("\n");
for (String line : lines) {
if (line.trim().length() <= 2) {
return;
}
Set<Entity> found = findEntities(line.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(line.trim());
}
}
}
}
}
}
@Retention(RetentionPolicy.RUNTIME)
@Target(ElementType.METHOD)
public @interface WhenCondition {
}
@Retention(RetentionPolicy.RUNTIME)
@Target(ElementType.METHOD)
public @interface ThenAction {
}
@Retention(RetentionPolicy.RUNTIME)
@Target(ElementType.PARAMETER)
public @interface Argument {
ArgumentType value() default ArgumentType.STRING;
}
}
@@ -0,0 +1,14 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import java.util.HashMap;
import java.util.Map;
import lombok.Data;
@Data
public class Classification {
private Map<String, Float> probabilities = new HashMap<>();
private String label;
}
@@ -0,0 +1,10 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class FilterGeometry {
private ImageSize imageSize;
private ImageFormat imageFormat;
}
@@ -0,0 +1,11 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class Filters {
private FilterGeometry geometry;
private Probability probability;
private boolean allPassed;
}
@@ -0,0 +1,9 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class Geometry {
private float width;
private float height;
}
@@ -0,0 +1,12 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class ImageFormat {
private float quotient;
private boolean tooTall;
private boolean tooWide;
}
@@ -0,0 +1,12 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class ImageMetadata {
private Classification classification;
private Position position;
private Geometry geometry;
private Filters filters;
}
@@ -0,0 +1,15 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import java.util.ArrayList;
import java.util.List;
import lombok.Data;
@Data
public class ImageServiceResponse {
private String dossierId;
private String fileId;
private List<ImageMetadata> data = new ArrayList<>();
}
@@ -0,0 +1,12 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class ImageSize {
private float quotient;
private boolean tooLarge;
private boolean tooSmall;
}
@@ -0,0 +1,12 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class Position {
private float x1;
private float x2;
private float y1;
private float y2;
private int pageNumber;
}
@@ -0,0 +1,10 @@
package com.iqser.red.service.redaction.v1.server.redaction.model.image;
import lombok.Data;
@Data
public class Probability {
private boolean unconfident;
}
@@ -0,0 +1,36 @@
package com.iqser.red.service.redaction.v1.server.redaction.rulebuilder;
import com.iqser.red.service.redaction.v1.model.Argument;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import com.iqser.red.service.redaction.v1.model.RuleElement;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import org.springframework.stereotype.Service;
import java.lang.reflect.Method;
import java.util.Arrays;
import java.util.List;
import java.util.stream.Collectors;
@Service
public class RuleBuilderModelService {
public RuleBuilderModel getRuleBuilderModel() {
var whenConditions = Arrays.stream(Section.class.getDeclaredMethods()).filter(m -> m.isAnnotationPresent(Section.WhenCondition.class)).collect(Collectors.toList());
var thenActions = Arrays.stream(Section.class.getDeclaredMethods()).filter(m -> m.isAnnotationPresent(Section.ThenAction.class)).collect(Collectors.toList());
RuleBuilderModel ruleBuilderModel = new RuleBuilderModel();
ruleBuilderModel.setWhenClauses(whenConditions.stream().map(c -> new RuleElement(c.getName(), toArguments(c))).collect(Collectors.toList()));
ruleBuilderModel.setThenConditions(thenActions.stream().map(c -> new RuleElement(c.getName(), toArguments(c))).collect(Collectors.toList()));
return ruleBuilderModel;
}
private List<Argument> toArguments(Method c) {
return Arrays.stream(c.getParameters())
.map(parameter -> new Argument(parameter.getName(), parameter.getAnnotation(Section.Argument.class).value()))
.collect(Collectors.toList());
}
}
@@ -1,68 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.RedactionChangeLog;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import lombok.NoArgsConstructor;
import lombok.RequiredArgsConstructor;
import org.springframework.stereotype.Service;
@Service
@RequiredArgsConstructor
public class AnalyzeResponseService {
private final RedactionServiceSettings redactionServiceSettings;
public AnalyzeResult createAnalyzeResponse(String dossierId, String fileId, long duration, int pageCount,
RedactionLog redactionLog, RedactionChangeLog redactionChangeLog) {
boolean hasHints = redactionLog.getRedactionLogEntry()
.stream()
.filter(entry -> !entry.isExcluded())
.anyMatch(entry -> entry.isHint() && !entry.getType().equals("false_positive"));
boolean hasRequests = redactionLog.getRedactionLogEntry()
.stream()
.filter(entry -> !entry.isExcluded())
.anyMatch(entry -> entry.isManual() && entry.getStatus()
.equals(com.iqser.red.service.redaction.v1.model.Status.REQUESTED));
boolean hasRedactions = redactionLog.getRedactionLogEntry()
.stream()
.filter(entry -> !entry.isExcluded())
.anyMatch(entry -> entry.isRedacted() && !entry.isManual() || entry.isManual() && entry.getStatus()
.equals(com.iqser.red.service.redaction.v1.model.Status.APPROVED));
boolean hasImages = redactionLog.getRedactionLogEntry()
.stream()
.filter(entry -> !entry.isExcluded())
.anyMatch(entry -> entry.isHint() && entry.getType().equals("image") || entry.isImage());
boolean hasUpdates = redactionChangeLog != null && redactionChangeLog.getRedactionLogEntry() != null && !redactionChangeLog
.getRedactionLogEntry()
.isEmpty() && redactionChangeLog.getRedactionLogEntry()
.stream()
.anyMatch(entry -> !entry.getType().equals("false_positive"));
return AnalyzeResult.builder()
.dossierId(dossierId)
.fileId(fileId)
.duration(duration)
.numberOfPages(pageCount)
.hasHints(hasHints)
.hasRedactions(hasRedactions)
.hasRequests(hasRequests)
.hasImages(hasImages)
.hasUpdates(hasUpdates)
.analysisVersion(redactionServiceSettings.getAnalysisVersion())
.rulesVersion(redactionLog.getRulesVersion())
.dictionaryVersion(redactionLog.getDictionaryVersion())
.legalBasisVersion(redactionLog.getLegalBasisVersion())
.dossierDictionaryVersion(redactionLog.getDossierDictionaryVersion())
.build();
}
}
@@ -0,0 +1,313 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import static com.iqser.red.service.redaction.v1.server.redaction.service.ImportedRedactionService.IMPORTED_REDACTION_TYPE;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import org.kie.api.runtime.KieContainer;
import org.springframework.stereotype.Service;
import org.springframework.web.bind.annotation.RequestBody;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualForceRedaction;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualLegalBasisChange;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.model.StructureAnalyzeRequest;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.SectionText;
import com.iqser.red.service.redaction.v1.server.classification.model.Text;
import com.iqser.red.service.redaction.v1.server.client.LegalBasisClient;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryIncrement;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryVersion;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import com.iqser.red.service.redaction.v1.server.redaction.model.PageEntities;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.redaction.model.RedRectangle2D;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.segmentation.ImageService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import lombok.RequiredArgsConstructor;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class AnalyzeService {
private final DictionaryService dictionaryService;
private final DroolsExecutionService droolsExecutionService;
private final EntityRedactionService entityRedactionService;
private final RedactionLogCreatorService redactionLogCreatorService;
private final RedactionStorageService redactionStorageService;
private final PdfSegmentationService pdfSegmentationService;
private final RedactionChangeLogService redactionChangeLogService;
private final LegalBasisClient legalBasisClient;
private final RedactionServiceSettings redactionServiceSettings;
private final SectionTextBuilderService sectionTextBuilderService;
private final SectionGridCreatorService sectionGridCreatorService;
private final ImageService imageService;
private final ImportedRedactionService importedRedactionService;
public AnalyzeResult analyzeDocumentStructure(StructureAnalyzeRequest analyzeRequest) {
long startTime = System.currentTimeMillis();
var pageCount = 0;
Document classifiedDoc;
try {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.ORIGIN));
Map<Integer, List<PdfImage>> pdfImages = null;
if (redactionServiceSettings.isEnableImageClassification()) {
pdfImages = imageService.convertImages(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
}
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, pdfImages);
pageCount = classifiedDoc.getPages().size();
} catch (Exception e) {
throw new RedactionException(e);
}
List<SectionText> sectionTexts = sectionTextBuilderService.buildSectionText(classifiedDoc);
sectionGridCreatorService.createSectionGrid(classifiedDoc, pageCount);
Text text = new Text(pageCount, sectionTexts);
// enhance section grid with headline data
sectionTexts.forEach(sectionText -> classifiedDoc.getSectionGrid()
.getSections()
.add(new SectionGrid.SectionGridSection(sectionText.getSectionNumber(), sectionText.getHeadline(), sectionText.getSectionAreas()
.stream()
.map(SectionArea::getPage)
.collect(Collectors.toSet()), sectionText.getSectionAreas())));
log.info("Store text and section grid for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
redactionStorageService.storeObject(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.TEXT, text);
redactionStorageService.storeObject(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.SECTION_GRID, classifiedDoc.getSectionGrid());
return AnalyzeResult.builder()
.dossierId(analyzeRequest.getDossierId())
.fileId(analyzeRequest.getFileId())
.duration(System.currentTimeMillis() - startTime)
.numberOfPages(text.getNumberOfPages())
.analysisVersion(redactionServiceSettings.getAnalysisVersion())
.build();
}
@SneakyThrows
public AnalyzeResult reanalyze(@RequestBody AnalyzeRequest analyzeRequest) {
long startTime = System.currentTimeMillis();
var redactionLog = redactionStorageService.getRedactionLog(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
var text = redactionStorageService.getText(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
// not yet ready for reanalysis
if (redactionLog == null || text == null || text.getNumberOfPages() == 0) {
return analyze(analyzeRequest);
}
DictionaryIncrement dictionaryIncrement = dictionaryService.getDictionaryIncrements(analyzeRequest.getDossierTemplateId(), new DictionaryVersion(redactionLog.getDictionaryVersion(), redactionLog.getDossierDictionaryVersion()), analyzeRequest.getDossierId());
Set<Integer> sectionsToReanalyse = !analyzeRequest.getSectionsToReanalyse()
.isEmpty() ? analyzeRequest.getSectionsToReanalyse() : findSectionsToReanalyse(dictionaryIncrement, redactionLog, text, analyzeRequest);
if (sectionsToReanalyse.isEmpty()) {
return finalizeAnalysis(analyzeRequest, startTime, redactionLog, text, dictionaryIncrement.getDictionaryVersion(), true);
}
NerEntities nerEntities;
if (redactionServiceSettings.isNerServiceEnabled()) {
nerEntities = redactionStorageService.getNerEntities(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
} else {
nerEntities = NerEntities.builder().build();
}
List<SectionText> reanalysisSections = text.getSectionTexts()
.stream()
.filter(sectionText -> sectionsToReanalyse.contains(sectionText.getSectionNumber()))
.collect(Collectors.toList());
KieContainer kieContainer = droolsExecutionService.updateRules(analyzeRequest.getDossierTemplateId());
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId());
PageEntities pageEntities = entityRedactionService.findEntities(dictionary, reanalysisSections, kieContainer, analyzeRequest, nerEntities);
var newRedactionLogEntries = redactionLogCreatorService.createRedactionLog(pageEntities, text.getNumberOfPages(), analyzeRequest.getDossierTemplateId());
var importedRedactionFilteredEntries = importedRedactionService.processImportedRedactions(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId(), analyzeRequest.getFileId(), newRedactionLogEntries, false);
redactionLog.getRedactionLogEntry()
.removeIf(entry -> sectionsToReanalyse.contains(entry.getSectionNumber()) && !entry.getType()
.equals(IMPORTED_REDACTION_TYPE));
redactionLog.getRedactionLogEntry().addAll(importedRedactionFilteredEntries);
return finalizeAnalysis(analyzeRequest, startTime, redactionLog, text, dictionaryIncrement.getDictionaryVersion(), true);
}
public AnalyzeResult analyze(AnalyzeRequest analyzeRequest) {
long startTime = System.currentTimeMillis();
var text = redactionStorageService.getText(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
NerEntities nerEntities;
if (redactionServiceSettings.isNerServiceEnabled()) {
nerEntities = redactionStorageService.getNerEntities(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
} else {
nerEntities = NerEntities.builder().build();
}
dictionaryService.updateDictionary(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId());
KieContainer kieContainer = droolsExecutionService.updateRules(analyzeRequest.getDossierTemplateId());
long rulesVersion = droolsExecutionService.getRulesVersion(analyzeRequest.getDossierTemplateId());
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId());
PageEntities pageEntities = entityRedactionService.findEntities(dictionary, text.getSectionTexts(), kieContainer, analyzeRequest, nerEntities);
List<RedactionLogEntry> redactionLogEntries = redactionLogCreatorService.createRedactionLog(pageEntities, text.getNumberOfPages(), analyzeRequest.getDossierTemplateId());
var legalBasis = legalBasisClient.getLegalBasisMapping(analyzeRequest.getDossierTemplateId());
var redactionLog = new RedactionLog(redactionServiceSettings.getAnalysisVersion(), analyzeRequest.getAnalysisNumber(), redactionLogEntries, legalBasis, dictionary.getVersion()
.getDossierTemplateVersion(), dictionary.getVersion()
.getDossierVersion(), rulesVersion, legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId()));
var importedRedactionFilteredEntries = importedRedactionService.processImportedRedactions(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId(), analyzeRequest.getFileId(), redactionLog.getRedactionLogEntry(), true);
redactionLog.setRedactionLogEntry(importedRedactionFilteredEntries);
return finalizeAnalysis(analyzeRequest, startTime, redactionLog, text, dictionary.getVersion(), false);
}
private Set<Integer> findSectionsToReanalyse(DictionaryIncrement dictionaryIncrement, RedactionLog redactionLog,
Text text, AnalyzeRequest analyzeRequest) {
long start = System.currentTimeMillis();
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
Set<Integer> sectionsToReanalyse = new HashSet<>();
Map<Integer, Set<Image>> imageEntries = new HashMap<>();
for (RedactionLogEntry entry : redactionLog.getRedactionLogEntry()) {
if (entry.isLocalManualRedaction() || relevantManuallyModifiedAnnotationIds.contains(entry.getId())) {
sectionsToReanalyse.add(entry.getSectionNumber());
}
if (entry.isImage()) {
imageEntries.computeIfAbsent(entry.getSectionNumber(), x -> new HashSet<>()).add(convert(entry));
}
}
for (SectionText sectionText : text.getSectionTexts()) {
if (EntitySearchUtils.sectionContainsAny(sectionText.getText(), dictionaryIncrement.getValues())) {
sectionsToReanalyse.add(sectionText.getSectionNumber());
}
}
log.info("Should reanalyze {} sections for request: {}, took: {}", sectionsToReanalyse.size(), analyzeRequest, System.currentTimeMillis() - start);
return sectionsToReanalyse;
}
private AnalyzeResult finalizeAnalysis(@RequestBody AnalyzeRequest analyzeRequest, long startTime,
RedactionLog redactionLog, Text text, DictionaryVersion dictionaryVersion,
boolean isReanalysis) {
redactionLog.setDictionaryVersion(dictionaryVersion.getDossierTemplateVersion());
redactionLog.setDossierDictionaryVersion(dictionaryVersion.getDossierVersion());
excludeExcludedPages(redactionLog, analyzeRequest.getExcludedPages());
var redactionLogChange = redactionChangeLogService.computeChanges(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), redactionLog, analyzeRequest.getAnalysisNumber());
redactionStorageService.storeObject(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.REDACTION_LOG, redactionLogChange.getRedactionLog());
long duration = System.currentTimeMillis() - startTime;
return AnalyzeResult.builder()
.dossierId(analyzeRequest.getDossierId())
.fileId(analyzeRequest.getFileId())
.duration(duration)
.numberOfPages(text.getNumberOfPages())
.hasUpdates(redactionLogChange.isHasChanges())
.analysisVersion(redactionServiceSettings.getAnalysisVersion())
.analysisNumber(analyzeRequest.getAnalysisNumber())
.rulesVersion(redactionLog.getRulesVersion())
.dictionaryVersion(redactionLog.getDictionaryVersion())
.legalBasisVersion(redactionLog.getLegalBasisVersion())
.dossierDictionaryVersion(redactionLog.getDossierDictionaryVersion())
.wasReanalyzed(isReanalysis)
.manualRedactions(analyzeRequest.getManualRedactions())
.build();
}
private Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
if (manualRedactions == null) {
return new HashSet<>();
}
return Stream.concat(manualRedactions.getLegalBasisChanges()
.stream()
.map(ManualLegalBasisChange::getAnnotationId), Stream.concat(manualRedactions.getImageRecategorization()
.stream()
.map(ManualImageRecategorization::getAnnotationId), Stream.concat(manualRedactions.getIdsToRemove()
.stream()
.map(IdRemoval::getAnnotationId), manualRedactions.getForceRedactions()
.stream()
.map(ManualForceRedaction::getAnnotationId)))).collect(Collectors.toSet());
}
public Image convert(RedactionLogEntry entry) {
Rectangle position = entry.getPositions().get(0);
return Image.builder()
.type(entry.getType())
.position(new RedRectangle2D(position.getTopLeft().getX(), position.getTopLeft()
.getY(), position.getWidth(), position.getHeight()))
.sectionNumber(entry.getSectionNumber())
.section(entry.getSection())
.page(position.getPage())
.hasTransparency(entry.isImageHasTransparency())
.build();
}
private void excludeExcludedPages(RedactionLog redactionLog, Set<Integer> excludedPages) {
if (excludedPages != null && !excludedPages.isEmpty()) {
redactionLog.getRedactionLogEntry().forEach(entry -> entry.getPositions().forEach(pos -> {
if (excludedPages.contains(pos.getPage())) {
entry.setExcluded(true);
}
}));
}
}
}
@@ -1,5 +1,7 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.Comment;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.*;
import lombok.RequiredArgsConstructor;
import org.apache.pdfbox.pdmodel.PDDocument;
@@ -14,21 +16,16 @@ import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationText;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.springframework.stereotype.Service;
import java.awt.Color;
import java.awt.*;
import java.io.IOException;
import java.util.ArrayList;
import java.util.GregorianCalendar;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.*;
import java.util.stream.Collectors;
@Service
@RequiredArgsConstructor
public class AnnotationService {
private final DictionaryService dictionaryService;
public void annotate(PDDocument document, RedactionLog redactionLog, SectionGrid sectionGrid) throws IOException {
@@ -91,7 +88,7 @@ public class AnnotationService {
if (redactionLogEntry.getComments() != null) {
for (Comment comment : redactionLogEntry.getComments()) {
PDAnnotationText txtAnnot = new PDAnnotationText();
txtAnnot.setAnnotationName(comment.getId());
txtAnnot.setAnnotationName(String.valueOf(comment.getId()));
txtAnnot.setInReplyTo(annotation); // Reference to highlight annotation
txtAnnot.setName(PDAnnotationText.NAME_COMMENT);
txtAnnot.setCreationDate(GregorianCalendar.from(comment.getDate().toZonedDateTime()));
@@ -108,10 +105,10 @@ public class AnnotationService {
private String createAnnotationContent(RedactionLogEntry redactionLogEntry) {
if (redactionLogEntry.isManual()) {
if (redactionLogEntry.isLocalManualRedaction()) {
return "\nManual Redaction\n\nIn Section : \"" + redactionLogEntry.getSection() + "\"";
}
return "\nRule " + redactionLogEntry.getMatchedRule() + " matched\n\n" + redactionLogEntry.getReason() + "\n\nLegal basis:" + redactionLogEntry
return redactionLogEntry.getType() + " \nRule " + redactionLogEntry.getMatchedRule() + " matched\n\n" + redactionLogEntry.getReason() + "\n\nLegal basis:" + redactionLogEntry
.getLegalBasis() + "\n\nIn section: \"" + redactionLogEntry.getSection() + "\"";
}
@@ -170,12 +167,12 @@ public class AnnotationService {
private float[] toQuadPoint(Rectangle rectangle, PDRectangle mediaBox, PDRectangle cropBox) {
var x1 = rectangle.getTopLeft().getX() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
var y1 = rectangle.getTopLeft().getY() + (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
var x1 = rectangle.getTopLeft().getX() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftX() + (cropBox.toString().equals(mediaBox.toString()) ? cropBox.getLowerLeftX() : 0f);
var y1 = rectangle.getTopLeft().getY() + (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY()) + (cropBox.toString().equals(mediaBox.toString()) ? cropBox.getLowerLeftY() : 0f);
var x2 = rectangle.getTopLeft()
.getX() + rectangle.getWidth() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
.getX() + rectangle.getWidth() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftX() + (cropBox.toString().equals(mediaBox.toString()) ? cropBox.getLowerLeftX() : 0f);
var y2 = rectangle.getTopLeft()
.getY() + rectangle.getHeight() - (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
.getY() + rectangle.getHeight() - (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY()) + (cropBox.toString().equals(mediaBox.toString()) ? cropBox.getLowerLeftY() : 0f);
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
@@ -1,9 +1,7 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryEntry;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.*;
@@ -14,11 +12,11 @@ import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.SerializationUtils;
import org.springframework.stereotype.Service;
import java.awt.Color;
import java.awt.*;
import java.util.List;
import java.util.*;
import java.util.stream.Collectors;
import static com.iqser.red.service.configuration.v1.api.resource.DictionaryResource.GLOBAL_DOSSIER;
@Slf4j
@Service
@@ -33,17 +31,17 @@ public class DictionaryService {
public DictionaryVersion updateDictionary(String dossierTemplateId, String dossierId) {
log.info("Updating dictionary data for: {} / {}", dossierTemplateId, dossierId);
long dossierTemplateDictionaryVersion = dictionaryClient.getVersion(dossierTemplateId, GLOBAL_DOSSIER);
log.info("Updating dictionary data for dossierTemplate {} and dossier {}", dossierTemplateId, dossierId);
long dossierTemplateDictionaryVersion = dictionaryClient.getVersion(dossierTemplateId);
var dossierTemplateDictionary = dictionariesByDossierTemplate.get(dossierTemplateId);
if (dossierTemplateDictionary == null || dossierTemplateDictionaryVersion > dossierTemplateDictionary.getDictionaryVersion()) {
updateDictionaryEntry(dossierTemplateId, dossierTemplateDictionaryVersion, GLOBAL_DOSSIER);
updateDictionaryEntry(dossierTemplateId, dossierTemplateDictionaryVersion, getVersion(dossierTemplateDictionary), null);
}
long dossierDictionaryVersion = dictionaryClient.getVersion(dossierTemplateId, dossierId);
long dossierDictionaryVersion = dictionaryClient.getVersionForDossier(dossierId);
var dossierDictionary = dictionariesByDossier.get(dossierId);
if (dossierDictionary == null || dossierDictionaryVersion > dossierDictionary.getDictionaryVersion()) {
updateDictionaryEntry(dossierTemplateId, dossierDictionaryVersion, dossierId);
updateDictionaryEntry(dossierTemplateId, dossierDictionaryVersion, getVersion(dossierDictionary), dossierId);
}
return DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateDictionaryVersion).dossierVersion(dossierDictionaryVersion).build();
@@ -62,9 +60,19 @@ public class DictionaryService {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
});
if(dictionariesByDossier.containsKey(dossierId)) {
if (dictionariesByDossier.containsKey(dossierId)) {
dictionaryModels = dictionariesByDossier.get(dossierId).getDictionary();
dictionaryModels.forEach(dictionaryModel -> {
dictionaryModel.getEntries().forEach(dictionaryEntry -> {
@@ -72,6 +80,16 @@ public class DictionaryService {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
});
}
@@ -79,18 +97,56 @@ public class DictionaryService {
}
private void updateDictionaryEntry(String dossierTemplateId, long version, String dossierId) {
private void updateDictionaryEntry(String dossierTemplateId, long version, Long currentVersion, String dossierId) {
try {
DictionaryRepresentation dictionaryRepresentation = new DictionaryRepresentation();
TypeResponse typeResponse = dictionaryClient.getAllTypes(dossierTemplateId, dossierId);
if (typeResponse != null && CollectionUtils.isNotEmpty(typeResponse.getTypes())) {
var typeResponse = dossierId == null ? dictionaryClient.getAllTypesForDossierTemplate(dossierTemplateId) : dictionaryClient.getAllTypesForDossier(dossierId);
if (CollectionUtils.isNotEmpty(typeResponse)) {
List<DictionaryModel> dictionary = typeResponse.getTypes()
List<DictionaryModel> dictionary = typeResponse
.stream()
.map(t -> new DictionaryModel(t.getType(), t.getRank(), convertColor(t.getHexColor()), t.isCaseInsensitive(), t
.isHint(), t.isRecommendation(), convertEntries(t, dossierId), new HashSet<>(), !dossierId.equals(GLOBAL_DOSSIER)))
.map(t -> {
Optional<DictionaryModel> oldModel;
if (dossierId == null) {
var representation = dictionariesByDossierTemplate.get(dossierTemplateId);
oldModel = representation != null ? representation.getDictionary().stream().filter(f -> f.getType().equals(t.getType())).findAny() : Optional.empty();
} else {
var representation = dictionariesByDossier.get(dossierId);
oldModel = representation != null ? representation.getDictionary().stream().filter(f -> f.getType().equals(t.getType())).findAny() : Optional.empty();
}
Set<DictionaryEntry> entries = new HashSet<>();
Set<DictionaryEntry> falsePositives = new HashSet<>();
Set<DictionaryEntry> falseRecommendations = new HashSet<>();
DictionaryEntries newEntries = getEntries(t.getId(), currentVersion);
var newValues = newEntries.getEntries().stream().map(v -> v.getValue()).collect(Collectors.toSet());
var newFalsePositivesValues = newEntries.getFalsePositives().stream().map(v -> v.getValue()).collect(Collectors.toSet());
var newFalseRecommendationsValues = newEntries.getFalseRecommendations().stream().map(v -> v.getValue()).collect(Collectors.toSet());
// add old entries from existing DictionaryModel
oldModel.ifPresent(dictionaryModel -> entries.addAll(dictionaryModel.getEntries().stream().filter(
f -> !newValues.contains(f.getValue())).collect(Collectors.toList())
));
oldModel.ifPresent(dictionaryModel -> falsePositives.addAll(dictionaryModel.getFalsePositives().stream().filter(
f -> !newFalsePositivesValues.contains(f.getValue())).collect(Collectors.toList())
));
oldModel.ifPresent(dictionaryModel -> falseRecommendations.addAll(dictionaryModel.getFalseRecommendations().stream().filter(
f -> !newFalseRecommendationsValues.contains(f.getValue())).collect(Collectors.toList())
));
// Add Increments
entries.addAll(newEntries.getEntries());
falsePositives.addAll(newEntries.getFalsePositives());
falseRecommendations.addAll(newEntries.getFalseRecommendations());
return new DictionaryModel(t.getType(), t.getRank(), convertColor(t.getHexColor()), t.isCaseInsensitive(), t
.isHint(), entries, falsePositives, falseRecommendations, new HashSet<>(), dossierId != null);
})
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
.collect(Collectors.toList());
@@ -106,7 +162,7 @@ public class DictionaryService {
dictionaryRepresentation.setDictionaryVersion(version);
dictionaryRepresentation.setDictionary(dictionary);
if(dossierId.equals(GLOBAL_DOSSIER)) {
if (dossierId == null) {
dictionariesByDossierTemplate.put(dossierTemplateId, dictionaryRepresentation);
} else {
dictionariesByDossier.put(dossierId, dictionaryRepresentation);
@@ -119,29 +175,20 @@ public class DictionaryService {
}
public void updateExternalDictionary(Dictionary dictionary, String dossierTemplateId) {
private DictionaryEntries getEntries(String typeId, Long fromVersion) {
dictionary.getDictionaryModels().forEach(dm -> {
if (dm.isRecommendation() && !dm.getLocalEntries().isEmpty()) {
dictionaryClient.addEntries(dm.getType(), dossierTemplateId, new ArrayList<>(dm.getLocalEntries()), false, GLOBAL_DOSSIER);
long externalVersion = dictionaryClient.getVersion(dossierTemplateId, GLOBAL_DOSSIER);
if (externalVersion == dictionary.getVersion().getDossierTemplateVersion() + 1) {
dictionary.getVersion().setDossierTemplateVersion(externalVersion);
}
}
});
}
var type = dictionaryClient.getDictionaryForType(typeId, fromVersion);
Set<DictionaryEntry> entries = type.getEntries() != null ? new HashSet<>(type.getEntries()) : new HashSet<>();
Set<DictionaryEntry> falsePositives = type.getFalsePositiveEntries() != null ? new HashSet<>(type.getFalsePositiveEntries()) : new HashSet<>();
Set<DictionaryEntry> falseRecommendations = type.getFalseRecommendationEntries() != null ? new HashSet<>(type.getFalseRecommendationEntries()) : new HashSet<>();
private Set<DictionaryEntry> convertEntries(TypeResult t, String dossierId) {
Set<DictionaryEntry> entries = new HashSet<>(dictionaryClient.getDictionaryForType(t.getType(), t.getDossierTemplateId(), dossierId)
.getEntries());
if (t.isCaseInsensitive()) {
if (type.isCaseInsensitive()) {
entries.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
falsePositives.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
falseRecommendations.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
}
return entries;
return new DictionaryEntries(entries, falsePositives, falseRecommendations);
}
@@ -164,7 +211,6 @@ public class DictionaryService {
public float[] getColor(String type, String dossierTemplateId) {
log.info("requested : {} / {}",type,dossierTemplateId);
DictionaryModel model = dictionariesByDossierTemplate.get(dossierTemplateId).getLocalAccessMap().get(type);
if (model != null) {
return model.getColor();
@@ -183,16 +229,6 @@ public class DictionaryService {
}
public boolean isRecommendation(String type, String dossierTemplateId) {
DictionaryModel model = dictionariesByDossierTemplate.get(dossierTemplateId).getLocalAccessMap().get(type);
if (model != null) {
return model.isRecommendation();
}
return false;
}
public Dictionary getDeepCopyDictionary(String dossierTemplateId, String dossierId) {
List<DictionaryModel> copy = new ArrayList<>();
@@ -204,7 +240,7 @@ public class DictionaryService {
//TODO merge dictionaries if they have same names
long dossierDictionaryVersion = -1;
if(dictionariesByDossier.containsKey(dossierId)) {
if (dictionariesByDossier.containsKey(dossierId)) {
var dossierRepresentation = dictionariesByDossier.get(dossierId);
dossierRepresentation.getDictionary().forEach(dm -> {
copy.add(SerializationUtils.clone(dm));
@@ -233,4 +269,11 @@ public class DictionaryService {
return dictionariesByDossierTemplate.get(dossierTemplateId).getRequestAddColor();
}
private Long getVersion(DictionaryRepresentation dictionaryRepresentation) {
if (dictionaryRepresentation == null) {
return null;
} else {
return dictionaryRepresentation.getDictionaryVersion();
}
}
}
@@ -1,6 +1,5 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
@@ -76,13 +75,13 @@ public class DroolsExecutionService {
try {
RulesResponse rules = rulesClient.getRules(dossierTemplateId);
if (rules == null || StringUtils.isEmpty(rules.getRules())) {
var rules = rulesClient.getRules(dossierTemplateId);
if (rules == null || StringUtils.isEmpty(rules.getValue())) {
throw new RuntimeException("Rules cannot be empty.");
}
KieServices kieServices = KieServices.Factory.get();
KieModule kieModule = getKieModule(dossierTemplateId, rules.getRules(), kieServices);
KieModule kieModule = getKieModule(dossierTemplateId, rules.getValue(), kieServices);
var container = kieContainers.get(dossierTemplateId);
if (container != null) {
@@ -1,64 +1,169 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.redaction.v1.model.*;
import com.iqser.red.service.redaction.v1.server.classification.model.*;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.*;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.commons.collections4.CollectionUtils;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.runtime.KieContainer;
import org.springframework.stereotype.Service;
import java.util.*;
import java.util.concurrent.atomic.AtomicInteger;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.Engine;
import com.iqser.red.service.redaction.v1.server.classification.model.SectionText;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entities;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import com.iqser.red.service.redaction.v1.server.redaction.model.PageEntities;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.redaction.model.SectionSearchableTextPair;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class EntityRedactionService {
private final DictionaryService dictionaryService;
private final RedactionServiceSettings redactionServiceSettings;
private final DroolsExecutionService droolsExecutionService;
private final SurroundingWordsService surroundingWordsService;
public void processDocument(Document classifiedDoc, String dossierTemplateId, ManualRedactions manualRedactions,
String dossierId, List<FileAttribute> fileAttributes) {
public PageEntities findEntities(Dictionary dictionary, List<SectionText> sectionTexts, KieContainer kieContainer,
AnalyzeRequest analyzeRequest, NerEntities nerEntities) {
dictionaryService.updateDictionary(dossierTemplateId, dossierId);
KieContainer container = droolsExecutionService.updateRules(dossierTemplateId);
long rulesVersion = droolsExecutionService.getRulesVersion(dossierTemplateId);
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(dossierTemplateId, dossierId);
Set<Entity> documentEntities = new HashSet<>(findEntities(classifiedDoc, container, manualRedactions, dictionary, false, null, fileAttributes));
Map<Integer, Set<Image>> imagesPerPage = new HashMap<>();
Set<Entity> entities = findEntities(sectionTexts, dictionary, kieContainer, analyzeRequest, false, null, imagesPerPage, nerEntities);
if (dictionary.hasLocalEntries()) {
Map<Integer, Set<Entity>> hintsPerSectionNumber = getHintsPerSection(documentEntities, dictionary);
Set<Entity> foundByLocal = findEntities(classifiedDoc, container, manualRedactions, dictionary, true, hintsPerSectionNumber, fileAttributes);
EntitySearchUtils.addEntitiesWithHigherRank(documentEntities, foundByLocal, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(documentEntities);
Map<Integer, Set<Entity>> hintsPerSectionNumber = getHintsPerSection(entities, dictionary);
Set<Entity> foundByLocal = findEntities(sectionTexts, dictionary, kieContainer, analyzeRequest, true, hintsPerSectionNumber, imagesPerPage, nerEntities);
EntitySearchUtils.addEntitiesWithHigherRank(entities, foundByLocal, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
}
classifiedDoc.setEntities(convertToEnititesPerPage(documentEntities));
Map<Integer, List<Entity>> entitiesPerPage = convertToEntitiesPerPage(entities);
EntitySearchUtils.removeEntitiesContainedInRedactedLogos(imagesPerPage, entitiesPerPage);
dictionaryService.updateExternalDictionary(dictionary, dossierTemplateId);
classifiedDoc.setDictionaryVersion(dictionary.getVersion());
classifiedDoc.setRulesVersion(rulesVersion);
return new PageEntities(entitiesPerPage, imagesPerPage);
}
public Map<Integer, List<Entity>> convertToEnititesPerPage(Set<Entity> entities){
public Set<Entity> findEntities(List<SectionText> reanalysisSections, Dictionary dictionary,
KieContainer kieContainer, AnalyzeRequest analyzeRequest, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber,
Map<Integer, Set<Image>> imagesPerPage, NerEntities nerEntities) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (SectionText reanalysisSection : reanalysisSections) {
Entities entities = findEntities(reanalysisSection.getSearchableText(), reanalysisSection.getHeadline(), reanalysisSection
.getSectionNumber(), dictionary, local, nerEntities, reanalysisSection.getCellStarts());
if (reanalysisSection.getCellStarts() != null && !reanalysisSection.getCellStarts().isEmpty()) {
surroundingWordsService.addSurroundingText(entities.getEntities(), reanalysisSection.getSearchableText(), dictionary, reanalysisSection
.getCellStarts());
} else {
surroundingWordsService.addSurroundingText(entities.getEntities(), reanalysisSection.getSearchableText(), dictionary);
}
if (!local && analyzeRequest.getManualRedactions() != null) {
var approvedForceRedactions = analyzeRequest.getManualRedactions().getForceRedactions().stream()
.filter(fr -> fr.getStatus() == AnnotationStatus.APPROVED)
.filter(fr -> fr.getRequestDate() != null)
.collect(Collectors.toList());
// only approved id removals, that haven't been forced back afterwards
var idsToRemove = analyzeRequest.getManualRedactions().getIdsToRemove().stream()
.filter(idr -> idr.getStatus() == AnnotationStatus.APPROVED && !idr.isRemoveFromDictionary())
.filter(idr -> idr.getRequestDate() != null)
.filter(idr -> approvedForceRedactions.stream().noneMatch(forceRedact -> forceRedact.getRequestDate().isAfter(idr.getRequestDate())))
.map(IdRemoval::getAnnotationId).collect(Collectors.toSet());
if (reanalysisSection.getImages() != null && !reanalysisSection.getImages()
.isEmpty() && analyzeRequest.getManualRedactions().getImageRecategorization() != null) {
for (Image image : reanalysisSection.getImages()) {
String imageId = IdBuilder.buildId(image.getPosition(), image.getPage());
for (ManualImageRecategorization imageRecategorization : analyzeRequest.getManualRedactions()
.getImageRecategorization()) {
if (imageRecategorization.getStatus()
.equals(AnnotationStatus.APPROVED) && imageRecategorization.getAnnotationId()
.equals(imageId)) {
image.setType(imageRecategorization.getType());
}
}
if (idsToRemove.contains(imageId)) {
image.setIgnored(true);
}
}
}
entities.getEntities().forEach(entity -> entity.getPositionSequences().forEach(ps -> {
if (idsToRemove.contains(ps.getId())) {
entity.setIgnored(true);
}
}));
}
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(false)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(reanalysisSection.getSectionNumber()) ? Stream
.concat(entities.getEntities().stream(), hintsPerSectionNumber.get(reanalysisSection.getSectionNumber())
.stream())
.collect(Collectors.toSet()) : entities.getEntities())
.nerEntities(entities.getNerEntities())
.text(reanalysisSection.getSearchableText().getAsStringWithLinebreaks())
.searchText(reanalysisSection.getSearchableText().toString())
.headline(reanalysisSection.getHeadline())
.sectionNumber(reanalysisSection.getSectionNumber())
.tabularData(reanalysisSection.getTabularData())
.searchableText(reanalysisSection.getSearchableText())
.dictionary(dictionary)
.images(reanalysisSection.getImages())
.fileAttributes(analyzeRequest.getFileAttributes())
.build(), reanalysisSection.getSearchableText()));
}
Set<Entity> entities = new HashSet<>();
sectionSearchableTextPairs.forEach(sectionSearchableTextPair -> {
Section analysedSection = droolsExecutionService.executeRules(kieContainer, sectionSearchableTextPair.getSection());
EntitySearchUtils.removeEntitiesContainedInLarger(analysedSection.getEntities());
entities.addAll(analysedSection.getEntities());
if (!local) {
for (Image image : analysedSection.getImages()) {
imagesPerPage.computeIfAbsent(image.getPage(), (a) -> new HashSet<>()).add(image);
}
addLocalValuesToDictionary(analysedSection, dictionary);
}
});
return entities;
}
private Map<Integer, List<Entity>> convertToEntitiesPerPage(Set<Entity> entities) {
Map<Integer, List<Entity>> entitiesPerPage = new HashMap<>();
for (Entity entity : entities) {
Map<Integer, List<EntityPositionSequence>> sequenceOnPage = new HashMap<>();
@@ -68,95 +173,33 @@ public class EntityRedactionService {
}
for (Map.Entry<Integer, List<EntityPositionSequence>> entry : sequenceOnPage.entrySet()) {
entitiesPerPage
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
entitiesPerPage.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber(), entity
.getLegalBasis(), entity.isDictionaryEntry(), entity.getTextBefore(), entity.getTextAfter(), entity
.getStart(), entity.getEnd(), entity.isDossierDictionaryEntry()));
.getStart(), entity.getEnd(), entity.isDossierDictionaryEntry(), entity.getEngines(), entity
.getReferences(), entity.getEntityType()));
}
}
return entitiesPerPage;
}
public Map<Integer, Set<Entity>> getHintsPerSection(Set<Entity> entities, Dictionary dictionary){
private Map<Integer, Set<Entity>> getHintsPerSection(Set<Entity> entities, Dictionary dictionary) {
Map<Integer, Set<Entity>> hintsPerSectionNumber = new HashMap<>();
entities.stream().forEach(entity -> {
if (dictionary.isHint(entity.getType()) && entity.isDictionaryEntry()) {
hintsPerSectionNumber.computeIfAbsent(entity.getSectionNumber(), (x) -> new HashSet<>())
.add(entity);
hintsPerSectionNumber.computeIfAbsent(entity.getSectionNumber(), (x) -> new HashSet<>()).add(entity);
}
});
return hintsPerSectionNumber;
}
private Set<Entity> findEntities(Document classifiedDoc, KieContainer kieContainer,
ManualRedactions manualRedactions, Dictionary dictionary, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber,
List<FileAttribute> fileAttributes) {
Set<Entity> documentEntities = new HashSet<>();
private void addLocalValuesToDictionary(Section analysedSection, Dictionary dictionary) {
AtomicInteger sectionNumber = new AtomicInteger(1);
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
List<Table> tables = paragraph.getTables();
for (Table table : tables) {
if (table.getColCount() == 2) {
sectionSearchableTextPairs.addAll(processTableAsOneText(classifiedDoc, table, sectionNumber, dictionary, local, hintsPerSectionNumber, fileAttributes));
} else {
sectionSearchableTextPairs.addAll(processTablePerRow(classifiedDoc, table, sectionNumber, dictionary, local, hintsPerSectionNumber, fileAttributes));
}
sectionNumber.incrementAndGet();
}
sectionSearchableTextPairs.add(processText(classifiedDoc, paragraph.getSearchableText(), paragraph.getTextBlocks(), paragraph
.getHeadline(), manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber, paragraph
.getImages(), fileAttributes));
sectionNumber.incrementAndGet();
}
for (Header header : classifiedDoc.getHeaders()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, header.getSearchableText(), header.getTextBlocks(), "Header", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber, new ArrayList<>(), fileAttributes));
sectionNumber.incrementAndGet();
}
for (Footer footer : classifiedDoc.getFooters()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, footer.getSearchableText(), footer.getTextBlocks(), "Footer", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber, new ArrayList<>(), fileAttributes));
sectionNumber.incrementAndGet();
}
for (UnclassifiedText unclassifiedText : classifiedDoc.getUnclassifiedTexts()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, unclassifiedText.getSearchableText(), unclassifiedText
.getTextBlocks(), "", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber, new ArrayList<>(), fileAttributes));
sectionNumber.incrementAndGet();
}
sectionSearchableTextPairs.forEach(sectionSearchableTextPair -> {
Section analysedSection = droolsExecutionService.executeRules(kieContainer, sectionSearchableTextPair.getSection());
documentEntities.addAll(analysedSection.getEntities());
for (Image image : analysedSection.getImages()) {
classifiedDoc.getImages().computeIfAbsent(image.getPage(), (a) -> new HashSet<>()).add(image);
}
addLocalValuesToDictionary(analysedSection, dictionary);
});
return documentEntities;
}
public void addLocalValuesToDictionary(Section analysedSection, Dictionary dictionary){
analysedSection.getLocalDictionaryAdds().keySet().forEach(key -> {
if (dictionary.isRecommendation(key)) {
analysedSection.getLocalDictionaryAdds().get(key).forEach(value -> {
if (!dictionary.containsValue(key, value)) {
dictionary.getLocalAccessMap().get(key).getLocalEntries().add(value);
}
});
} else {
analysedSection.getLocalDictionaryAdds().get(key).forEach(value -> {
if (dictionary.getLocalAccessMap().get(key) == null) {
@@ -169,258 +212,62 @@ public class EntityRedactionService {
dictionary.getLocalAccessMap().get(key).getLocalEntries().add(value);
});
}
});
}
private List<SectionSearchableTextPair> processTablePerRow(Document classifiedDoc, Table table,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber,
List<FileAttribute> fileAttributes) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
Map<String, CellValue> tabularData = new HashMap<>();
int start = 0;
List<Integer> cellStarts = new ArrayList<>();
SectionText sectionText = new SectionText();
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
SectionArea sectionArea = new SectionArea(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight(), cell.getTextBlocks()
.get(0)
.getSequences()
.get(0)
.getPage());
sectionText.getSectionAreas().add(sectionArea);
sectionText.getTextBlocks().addAll(cell.getTextBlocks());
int cellStart = start;
if (!cell.isHeaderCell()) {
cell.getHeaderCells().forEach(headerCell -> {
StringBuilder headerBuilder = new StringBuilder();
headerCell.getTextBlocks().forEach(textBlock -> headerBuilder.append(textBlock.getText()));
String headerName = headerBuilder.toString()
.replaceAll("\n", "")
.replaceAll(" ", "")
.replaceAll("-", "");
sectionArea.setHeader(headerName);
tabularData.put(headerName, new CellValue(cell.getTextBlocks(), cellStart));
});
}
for (TextBlock textBlock : cell.getTextBlocks()) {
// TODO avoid cell overlap merging.
searchableRow.addAll(textBlock.getSequences());
}
cellStarts.add(cellStart);
start = start + cell.toString().trim().length() + 1;
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, searchableRow, dictionary, cellStarts);
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(rowEntities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber.intValue())
.tabularData(tabularData)
.searchableText(searchableRow)
.dictionary(dictionary)
.fileAttributes(fileAttributes)
.build(), searchableRow));
if (!local) {
sectionText.setText(searchableRow.toString());
sectionText.setHeadline(table.getHeadline());
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(true);
sectionText.setTabularData(tabularData);
sectionText.setCellStarts(cellStarts);
classifiedDoc.getSectionText().add(sectionText);
}
sectionNumber.incrementAndGet();
}
return sectionSearchableTextPairs;
}
private List<SectionSearchableTextPair> processTableAsOneText(Document classifiedDoc, Table table,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber,
List<FileAttribute> fileAttributes) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
SearchableText entireTableText = new SearchableText();
SectionText sectionText = new SectionText();
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
if (!local) {
SectionArea sectionArea = new SectionArea(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight(), cell.getTextBlocks()
.get(0)
.getSequences()
.get(0)
.getPage());
sectionText.getTextBlocks().addAll(cell.getTextBlocks());
sectionText.getSectionAreas().add(sectionArea);
}
for (TextBlock textBlock : cell.getTextBlocks()) {
entireTableText.addAll(textBlock.getSequences());
}
}
}
Set<Entity> rowEntities = findEntities(entireTableText, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, entireTableText, dictionary);
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(rowEntities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : rowEntities)
.text(entireTableText.getAsStringWithLinebreaks())
.searchText(entireTableText.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber.intValue())
.searchableText(entireTableText)
.dictionary(dictionary)
.fileAttributes(fileAttributes)
.build(), entireTableText));
if (!local) {
sectionText.setText(entireTableText.toString());
sectionText.setHeadline(table.getHeadline());
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(true);
classifiedDoc.getSectionText().add(sectionText);
}
return sectionSearchableTextPairs;
}
private SectionSearchableTextPair processText(Document classifiedDoc, SearchableText searchableText,
List<TextBlock> paragraphTextBlocks, String headline,
ManualRedactions manualRedactions, AtomicInteger sectionNumber,
Dictionary dictionary, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber,
List<PdfImage> images, List<FileAttribute> fileAttributes) {
if (!local) {
SectionText sectionText = new SectionText();
for (TextBlock paragraphTextBlock : paragraphTextBlocks) {
SectionArea sectionArea = new SectionArea(new Point(paragraphTextBlock.getMinX(), paragraphTextBlock.getMinY()), paragraphTextBlock
.getWidth(), paragraphTextBlock.getHeight(), paragraphTextBlock.getPage());
sectionText.getSectionAreas().add(sectionArea);
}
sectionText.setText(searchableText.toString());
sectionText.setHeadline(headline);
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(false);
sectionText.setImages(images.stream()
.map(image -> convertAndRecategorize(image, sectionNumber.intValue(), headline, manualRedactions))
.collect(Collectors.toSet()));
sectionText.setTextBlocks(paragraphTextBlocks);
classifiedDoc.getSectionText().add(sectionText);
}
Set<Entity> entities = findEntities(searchableText, headline, sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(entities, searchableText, dictionary);
return new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(entities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(headline)
.sectionNumber(sectionNumber.intValue())
.searchableText(searchableText)
.dictionary(dictionary)
.images(images.stream()
.map(image -> convertAndRecategorize(image, sectionNumber.intValue(), headline, manualRedactions))
.collect(Collectors.toSet()))
.fileAttributes(fileAttributes)
.build(), searchableText);
}
public Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber,
Dictionary dictionary, boolean local) {
private Entities findEntities(SearchableText searchableText, String headline, int sectionNumber,
Dictionary dictionary, boolean local, NerEntities nerEntities,
List<Integer> cellStarts) {
Set<Entity> found = new HashSet<>();
String searchableString = searchableText.toString();
if (StringUtils.isEmpty(searchableString)) {
return found;
return new Entities(new HashSet<>(), new HashSet<>());
}
String lowercaseInputString = searchableString.toLowerCase();
for (DictionaryModel model : dictionary.getDictionaryModels()) {
if (model.isCaseInsensitive()) {
found.addAll(EntitySearchUtils.find(lowercaseInputString, model.getValues(local), model.getType(), headline, sectionNumber, local, model
.isDossierDictionary()));
EntitySearchUtils.addOrAddEngine(found, EntitySearchUtils.findEntities(lowercaseInputString, model.getValues(local), model, headline, sectionNumber, !local, model.isDossierDictionary(), local ? Engine.RULE : Engine.DICTIONARY, false, local ? true : false));
} else {
found.addAll(EntitySearchUtils.find(searchableString, model.getValues(local), model.getType(), headline, sectionNumber, local, model
.isDossierDictionary()));
EntitySearchUtils.addOrAddEngine(found, EntitySearchUtils.findEntities(searchableString, model.getValues(local), model, headline, sectionNumber, !local, model.isDossierDictionary(), local ? Engine.RULE : Engine.DICTIONARY, false, local ? true : false));
}
}
return EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary);
Set<Entity> nerFound = new HashSet<>();
if (!local) {
nerFound.addAll(getNerValues(sectionNumber, nerEntities, cellStarts, headline));
}
return new Entities(EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary), nerFound);
}
private Image convertAndRecategorize(PdfImage pdfImage, int sectionNumber, String headline, ManualRedactions manualRedactions) {
private Set<Entity> getNerValues(int sectionNumber, NerEntities nerEntities, List<Integer> cellStarts,
String headline) {
Image image = Image.builder()
.type(pdfImage.getImageType().equals(ImageType.OTHER) ? "image" : pdfImage.getImageType()
.name()
.toLowerCase(Locale.ROOT))
.position(pdfImage.getPosition())
.sectionNumber(sectionNumber)
.section(headline)
.page(pdfImage.getPage())
.hasTransparency(pdfImage.isHasTransparency())
.build();
Set<Entity> entities = new HashSet<>();
String imageId = IdBuilder.buildId(image.getPosition(), image.getPage());
if (manualRedactions != null && manualRedactions.getImageRecategorizations() != null) {
for (ManualImageRecategorization imageRecategorization : manualRedactions.getImageRecategorizations()) {
if (imageRecategorization.getStatus().equals(Status.APPROVED) && imageRecategorization.getId()
.equals(imageId)) {
image.setType(imageRecategorization.getType());
if (redactionServiceSettings.isNerServiceEnabled() && nerEntities.getData().containsKey(sectionNumber)) {
nerEntities.getData().get(sectionNumber).forEach(res -> {
if (cellStarts == null || cellStarts.isEmpty()) {
entities.add(new Entity(res.getValue(), res.getType(), res.getStartOffset(), res.getEndOffset(), headline, sectionNumber, false, false, Engine.NER, EntityType.RECOMMENDATION));
} else {
boolean intersectsCellStart = false;
for (Integer cellStart : cellStarts) {
if (res.getStartOffset() < cellStart && cellStart < res.getEndOffset()) {
intersectsCellStart = true;
break;
}
}
if (!intersectsCellStart) {
entities.add(new Entity(res.getValue(), res.getType(), res.getStartOffset(), res.getEndOffset(), headline, sectionNumber, false, false, Engine.NER, EntityType.RECOMMENDATION));
}
}
}
});
}
return image;
return entities;
}
}
@@ -1,71 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.client.ImageClassificationClient;
import com.iqser.red.service.redaction.v1.server.client.ImageClassificationResponse;
import com.iqser.red.service.redaction.v1.server.client.MockMultipartFile;
import com.iqser.red.service.redaction.v1.server.redaction.model.ImageType;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import javax.imageio.ImageIO;
import java.io.ByteArrayOutputStream;
@Slf4j
@Service
@RequiredArgsConstructor
public class ImageClassificationService {
private final ImageClassificationClient imageClassificationClient;
private final RedactionServiceSettings settings;
public void classifyImages(Page page) {
page.getImages().forEach(image -> {
if (settings.isEnableImageClassification() && !isEntirePageImage(image, page)) {
long start = System.currentTimeMillis();
try (ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
ImageIO.write(image.getImage(), "png", baos);
var mockFile = new MockMultipartFile("file", "Image.png", "image/png", baos.toByteArray());
ImageClassificationResponse response = imageClassificationClient.classify(mockFile);
image.setImageType(ImageType.valueOf(response.getCategory()));
} catch (Exception e) {
log.error("Could not classify image", e);
image.setImageType(ImageType.OTHER);
}
log.info("Image classification took: " + (System.currentTimeMillis() - start));
} else {
image.setImageType(ImageType.OTHER);
}
image.getImage().flush();
image.setImage(null);
if (image.getImageType().equals(ImageType.OTHER)) {
page.getTextBlocks().forEach(textblock -> {
if (image.getPosition()
.contains(textblock.getMinX(), textblock.getMinY(), textblock.getWidth(), textblock.getHeight())) {
image.setImageType(ImageType.OCR);
}
});
}
});
}
private boolean isEntirePageImage(PdfImage image, Page page){
double imageArea = image.getPosition().getHeight() * image.getPosition().getWidth();
if(imageArea / page.getCropBoxArea() >= settings.getMaxImageCropboxRatio()){
log.info("Skipping image classification because images is almost as large as the entire page");
return true;
}
return false;
}
}
@@ -0,0 +1,102 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.util.Iterator;
import java.util.List;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.ImportedRedaction;
import com.iqser.red.service.redaction.v1.model.ImportedRedactions;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class ImportedRedactionService {
public static final String IMPORTED_REDACTION_TYPE = "imported_redaction";
private final DictionaryService dictionaryService;
private final RedactionStorageService redactionStorageService;
public List<RedactionLogEntry> processImportedRedactions(String dossierTemplateId, String dossierId, String fileId,
List<RedactionLogEntry> redactionLogEntries, boolean addImportedRedactions) {
var importedRedactions = redactionStorageService.getImportedRedactions(dossierId, fileId);
if (importedRedactions == null) {
return redactionLogEntries;
}
redactionLogEntries.removeIf(redactionLogEntry -> hasIntersections(redactionLogEntry, importedRedactions));
if (addImportedRedactions) {
return addImportedRedactionsRedactionLogEntries(dossierTemplateId, redactionLogEntries, importedRedactions);
}
return redactionLogEntries;
}
private List<RedactionLogEntry> addImportedRedactionsRedactionLogEntries(String dossierTemplateId,
List<RedactionLogEntry> redactionLogEntries,
ImportedRedactions importedRedactions) {
for (List<ImportedRedaction> importedRedactionsValues : importedRedactions.getImportedRedactions().values()) {
for (ImportedRedaction importedRedaction : importedRedactionsValues) {
RedactionLogEntry redactionLogEntry = RedactionLogEntry.builder()
.id(importedRedaction.getId())
.type(IMPORTED_REDACTION_TYPE)
.imported(true)
.redacted(true)
.positions(importedRedaction.getPositions())
.color(getColor("imported_redaction", dossierTemplateId))
.build();
redactionLogEntries.add(redactionLogEntry);
}
}
return redactionLogEntries;
}
private boolean hasIntersections(RedactionLogEntry redactionLogEntry, ImportedRedactions importedRedactions) {
for (Rectangle rectangle : redactionLogEntry.getPositions()) {
if (importedRedactions.getImportedRedactions().containsKey(rectangle.getPage())) {
var importedRedactionsOnPage = importedRedactions.getImportedRedactions().get(rectangle.getPage());
for (ImportedRedaction importedRedaction : importedRedactionsOnPage) {
for (Rectangle importedRedactionPosition : importedRedaction.getPositions()) {
if (intersects(importedRedactionPosition, rectangle)) {
return true;
}
}
}
}
}
return false;
}
private boolean intersects(Rectangle importedPosition, Rectangle redactionLogPosition) {
return redactionLogPosition.getTopLeft()
.getX() + redactionLogPosition.getWidth() > importedPosition.getTopLeft()
.getX() && redactionLogPosition.getTopLeft()
.getY() + redactionLogPosition.getHeight() > importedPosition.getTopLeft()
.getY() && redactionLogPosition.getTopLeft().getX() < importedPosition.getTopLeft()
.getX() + importedPosition.getWidth() && redactionLogPosition.getTopLeft()
.getY() < importedPosition.getTopLeft().getY() + importedPosition.getHeight();
}
private float[] getColor(String type, String dossierTemplateId) {
return dictionaryService.getColor(type, dossierTemplateId);
}
}
@@ -0,0 +1,153 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.util.ArrayList;
import java.util.List;
import java.util.Set;
import org.apache.commons.lang3.tuple.Pair;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualRedactionEntry;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualResizeRedaction;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.Engine;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.server.classification.model.SectionText;
import com.iqser.red.service.redaction.v1.server.classification.model.Text;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class ManualRedactionSurroundingTextService {
private final RedactionStorageService redactionStorageService;
private final SurroundingWordsService surroundingWordsService;
public AnalyzeResult addSurroundingText(String dossierId, String fileId, ManualRedactions manualRedactions) {
long startTime = System.currentTimeMillis();
Text text = redactionStorageService.getText(dossierId, fileId);
List<ManualRedactionEntry> processedAddRedactions = new ArrayList<>();
List<ManualResizeRedaction> processedResizeRedactions = new ArrayList<>();
for (SectionText sectionText : text.getSectionTexts()) {
if (manualRedactions.getEntriesToAdd().isEmpty() && manualRedactions.getResizeRedactions().isEmpty()) {
break;
}
for (SectionArea sectionArea : sectionText.getSectionAreas()) {
var addItty = manualRedactions.getEntriesToAdd().iterator();
while (addItty.hasNext()) {
var manualAddRedaction = addItty.next();
if (sectionContainsEntry(sectionArea, manualAddRedaction.getPositions())) {
var surroundingText = findSurroundingText(sectionText, manualAddRedaction.getValue(), manualAddRedaction.getPositions());
manualAddRedaction.setTextBefore(surroundingText.getLeft());
manualAddRedaction.setTextAfter(surroundingText.getRight());
processedAddRedactions.add(manualAddRedaction);
addItty.remove();
}
}
var resizeItty = manualRedactions.getResizeRedactions().iterator();
while (resizeItty.hasNext()) {
var manualResizeRedaction = resizeItty.next();
if (sectionContainsEntry(sectionArea, manualResizeRedaction.getPositions())) {
var surroundingText = findSurroundingText(sectionText, manualResizeRedaction.getValue(), manualResizeRedaction.getPositions());
manualResizeRedaction.setTextBefore(surroundingText.getLeft());
manualResizeRedaction.setTextAfter(surroundingText.getRight());
processedResizeRedactions.add(manualResizeRedaction);
resizeItty.remove();
}
}
}
}
manualRedactions.getEntriesToAdd().addAll(processedAddRedactions);
manualRedactions.getResizeRedactions().addAll(processedResizeRedactions);
return AnalyzeResult.builder()
.dossierId(dossierId)
.fileId(fileId)
.manualRedactions(manualRedactions)
.duration(System.currentTimeMillis() - startTime)
.build();
}
private Pair<String, String> findSurroundingText(SectionText sectionText, String value,
List<Rectangle> toFindPositions) {
Set<Entity> entities = EntitySearchUtils.find(sectionText.getText(), Set.of(value), "dummy", sectionText.getHeadline(), sectionText.getSectionNumber(), false, false, Engine.DICTIONARY, false, EntityType.ENTITY);
Set<Entity> entitiesWithPositions = EntitySearchUtils.clearAndFindPositions(entities, sectionText.getSearchableText(), null);
Entity correctEntity = getEntityOnCorrectPosition(entitiesWithPositions, toFindPositions);
if (correctEntity == null) {
log.warn("Could not calculate surrounding text");
return Pair.of("", "");
}
if (sectionText.getCellStarts() != null && !sectionText.getCellStarts().isEmpty()) {
surroundingWordsService.addSurroundingText(Set.of(correctEntity), sectionText.getSearchableText(), null, sectionText.getCellStarts());
} else {
surroundingWordsService.addSurroundingText(Set.of(correctEntity), sectionText.getSearchableText(), null);
}
return Pair.of(correctEntity.getTextBefore(), correctEntity.getTextAfter());
}
private boolean sectionContainsEntry(SectionArea sectionArea, List<Rectangle> positions) {
for (Rectangle position : positions) {
if (sectionArea.contains(position)) {
return true;
}
}
return false;
}
private Entity getEntityOnCorrectPosition(Set<Entity> entitiesWithPositions, List<Rectangle> toFindPositions) {
for (Entity entityWithPos : entitiesWithPositions) {
for (EntityPositionSequence entityPositionSequence : entityWithPos.getPositionSequences()) {
for (TextPositionSequence textPositionSequence : entityPositionSequence.getSequences()) {
for (Rectangle manualRedactionRectangle : toFindPositions) {
if (intersects(manualRedactionRectangle, textPositionSequence.getRectangle())) {
return entityWithPos;
}
}
}
}
}
return null;
}
public boolean intersects(Rectangle manualPosition,
com.iqser.red.service.redaction.v1.model.Rectangle textPositionRectangle) {
return textPositionRectangle.getTopLeft()
.getX() + textPositionRectangle.getWidth() > manualPosition.getTopLeftX() && textPositionRectangle.getTopLeft()
.getY() + textPositionRectangle.getHeight() > manualPosition.getTopLeftY() && textPositionRectangle.getTopLeft()
.getX() < manualPosition.getTopLeftX() + manualPosition.getWidth() && textPositionRectangle.getTopLeft()
.getY() < manualPosition.getTopLeftY() + manualPosition.getHeight();
}
}

Some files were not shown because too many files have changed in this diff Show More