Compare commits

...
Author SHA1 Message Date
Kilian Schuettler 32c2087383 added documentation for EntityCreationService, RedactionEntity to prompts 2023-08-09 10:14:39 +02:00
Kilian Schuettler d0126d36a7 proomping 2 2023-08-08 10:57:53 +02:00
Kilian Schuettler b2621fb4b3 proompting 2023-08-08 10:57:51 +02:00
deiflaender 12fcc6ca6d DM-165: Updated documine rules 2023-08-07 13:38:28 +02:00
Dominique Eifländer bfc8e16c94 Merge branch 'DM-165' into 'master'
DM-165: Renamed isOnPage to onPage to prevent not working drools optimizations

Closes DM-165

See merge request redactmanager/redaction-service!81
2023-08-07 13:28:01 +02:00
deiflaender ea36f31df4 DM-165: Renamed isOnPage to onPage to prevent not working drools optimizations 2023-08-07 13:19:47 +02:00
Renovate Bot a9c02540c2 Merge branch 'renovate/master-com.iqser.red.commons-storage-commons-2.x' into 'master'
fix(deps): update dependency com.iqser.red.commons:storage-commons to v2.24.0 (master)

See merge request redactmanager/redaction-service!80
2023-08-05 09:07:00 +02:00
Renovate Bot 92137a9ebd fix(deps): update dependency com.iqser.red.commons:storage-commons to v2.24.0 (master) 2023-08-05 09:07:00 +02:00
Renovate Bot 65c2f2d1e2 Merge branch 'renovate/master-com.iqser.red.commons-storage-commons-2.x' into 'master'
fix(deps): update dependency com.iqser.red.commons:storage-commons to v2.23.0 (master)

See merge request redactmanager/redaction-service!78
2023-08-04 15:06:52 +02:00
Renovate Bot 1669efaa29 fix(deps): update dependency com.iqser.red.commons:storage-commons to v2.23.0 (master) 2023-08-04 15:06:52 +02:00
Kilian Schüttler 607f6db67f Merge branch 'CYB-001' into 'master'
CYB-001: cyberport custom

Closes CYB-001

See merge request redactmanager/redaction-service!79
2023-08-04 10:19:13 +02:00
Kilian Schüttler 26da18fb26 CYB-001: cyberport custom 2023-08-04 10:19:13 +02:00
Andrei Isvoran 6ebd6704e0 RED-7290 - fix checkstyle 2023-08-04 09:41:26 +03:00
Andrei Isvoran 12649d99da RED-7290 - Update platform-common-dependency version 2023-08-03 18:15:57 +03:00
Andrei Isvoran 6300e6721e Merge remote-tracking branch 'origin/master'
# Conflicts:
#	redaction-service-v1/redaction-service-server-v1/pom.xml
2023-08-03 18:15:27 +03:00
Andrei Isvoran 4ba8d77d79 RED-7290 - Update platform-common-dependency version 2023-08-03 18:15:15 +03:00
Timo Bejan 867c16b6ab Merge branch 'RED-7052-2' into 'master'
RED-7052 Solved KieContainer race condition

Closes RED-7052

See merge request redactmanager/redaction-service!77
2023-08-03 09:41:50 +02:00
Timo Bejan bd30e301e0 RED-7052 Solved KieContainer race condition 2023-08-02 23:17:51 +03:00
Renovate Bot 9b5ccd14e4 Merge branch 'renovate/master-spring-boot' into 'master'
fix(deps): update spring boot to v3.1.2 (master)

See merge request redactmanager/redaction-service!74
2023-08-02 21:07:12 +02:00
Renovate Bot cdff1a0bc0 fix(deps): update spring boot to v3.1.2 (master) 2023-08-02 21:07:12 +02:00
Renovate Bot 8e63be5752 Merge branch 'renovate/master-jacksonversion' into 'master'
fix(deps): update jacksonversion to v2.15.2 (master)

See merge request redactmanager/redaction-service!72
2023-08-02 18:06:43 +02:00
Renovate Bot 3a8668e1ad fix(deps): update jacksonversion to v2.15.2 (master) 2023-08-02 18:06:43 +02:00
Renovate Bot d896276a89 Merge branch 'renovate/master-droolsversion' into 'master'
fix(deps): update droolsversion to v8.42.0.final (master)

See merge request redactmanager/redaction-service!71
2023-08-02 15:06:50 +02:00
Renovate Bot 1cad97b515 fix(deps): update droolsversion to v8.42.0.final (master) 2023-08-02 15:06:50 +02:00
Renovate Bot 8c14b86306 Merge branch 'renovate/master-org.kie-kie-spring-7.x' into 'master'
fix(deps): update dependency org.kie:kie-spring to v7.74.1.final (master)

See merge request redactmanager/redaction-service!70
2023-08-02 12:06:35 +02:00
Renovate Bot 141f9e37d4 fix(deps): update dependency org.kie:kie-spring to v7.74.1.final (master) 2023-08-02 12:06:35 +02:00
Renovate Bot 17a04afd1e Merge branch 'renovate/master-com.iqser.red.commons-spring-commons-2.x' into 'master'
fix(deps): update dependency com.iqser.red.commons:spring-commons to v2.6.0 (master)

See merge request redactmanager/redaction-service!68
2023-08-02 09:07:16 +02:00
Renovate Bot cbdaf76265 fix(deps): update dependency com.iqser.red.commons:spring-commons to v2.6.0 (master) 2023-08-02 09:07:16 +02:00
Renovate Bot 5ddf7e9fdc Merge branch 'renovate/master-com.iqser.red.commons-metric-commons-2.x' into 'master'
fix(deps): update dependency com.iqser.red.commons:metric-commons to v2.3.0 (master)

See merge request redactmanager/redaction-service!67
2023-08-02 06:06:59 +02:00
Renovate Bot c8dfd39dad fix(deps): update dependency com.iqser.red.commons:metric-commons to v2.3.0 (master) 2023-08-02 06:06:59 +02:00
Renovate Bot 26e1ae1ff6 Merge branch 'renovate/master-com.iqser.red.commons-dictionary-merge-commons-1.x' into 'master'
fix(deps): update dependency com.iqser.red.commons:dictionary-merge-commons to v1.5.0 (master)

See merge request redactmanager/redaction-service!66
2023-08-02 03:06:29 +02:00
Renovate Bot 1416fc82bc fix(deps): update dependency com.iqser.red.commons:dictionary-merge-commons to v1.5.0 (master) 2023-08-02 03:06:29 +02:00
Renovate Bot 43ebba2c09 Merge branch 'renovate/master-plugins-(non-major)' into 'master'
chore(deps): update plugins (non-major) (master)

See merge request redactmanager/redaction-service!65
2023-08-02 00:06:25 +02:00
Renovate Bot 70d45c5b86 chore(deps): update plugins (non-major) (master) 2023-08-02 00:06:25 +02:00
Renovate Bot 6b205bf537 Merge branch 'renovate/master-spring-core' into 'master'
fix(deps): update dependency org.springframework:spring-web to v6.0.11 (master)

See merge request redactmanager/redaction-service!63
2023-08-01 21:06:53 +02:00
Renovate Bot 378fa66a1e fix(deps): update dependency org.springframework:spring-web to v6.0.11 (master) 2023-08-01 21:06:53 +02:00
Renovate Bot 686a3dad04 Merge branch 'renovate/master-spring-cloud' into 'master'
fix(deps): update dependency org.springframework.cloud:spring-cloud-starter-openfeign to v4.0.4 (master)

See merge request redactmanager/redaction-service!62
2023-08-01 18:07:30 +02:00
Renovate Bot 7ca196e45c fix(deps): update dependency org.springframework.cloud:spring-cloud-starter-openfeign to v4.0.4 (master) 2023-08-01 18:07:30 +02:00
Kilian Schüttler f29ff06263 Merge branch 'testGradle' into 'master'
Test gradle

See merge request redactmanager/redaction-service!61
2023-08-01 12:09:00 +02:00
Kilian Schüttler af9b581b4f Test gradle 2023-08-01 12:09:00 +02:00
Timo Bejan d9f7b3f516 Update pom.xml 2023-07-27 08:50:12 +02:00
Dominique Eifländer 7ad84a0ae1 Merge branch 'RED-5253' into 'master'
RED-5253: Improved headline detection for DocuMine 2

Closes RED-5253

See merge request redactmanager/redaction-service!60
2023-07-24 16:25:00 +02:00
deiflaender e930a124ba RED-5253: Improved headline detection for DocuMine 2 2023-07-24 16:19:34 +02:00
Dominique Eifländer 8f64331c16 Merge branch 'RED-5253' into 'master'
RED-5253: Improved headline detection for DocuMine

Closes RED-5253

See merge request redactmanager/redaction-service!59
2023-07-24 12:15:18 +02:00
deiflaender 02b6c05b14 RED-5253: Improved headline detection for DocuMine 2023-07-24 12:08:51 +02:00
Kilian Schüttler 6bc97c7e58 Merge branch 'DM-334' into 'master'
DM-334: OCR (hint) appears as skipped

Closes DM-334

See merge request redactmanager/redaction-service!58
2023-07-21 14:30:19 +02:00
Kilian Schuettler 2d7202ab54 DM-334: OCR (hint) appears as skipped 2023-07-21 13:07:58 +02:00
Kilian Schuettler 294855f91f hotfix increase threshhold 2023-07-20 16:06:08 +02:00
Kilian Schüttler 9e723f57bb Merge branch 'hotfix' into 'master'
hotfix: fix rule creating endless loop with dossier-redactions

See merge request redactmanager/redaction-service!57
2023-07-20 15:36:46 +02:00
Kilian Schuettler 25762e71b7 hotfix: fix rule creating endless loop with dossier-redactions 2023-07-20 15:30:40 +02:00
Dominique Eifländer 3bc11b7d33 Merge branch 'DM-307' into 'master'
DM-307 Improved headline detection

Closes DM-307

See merge request redactmanager/redaction-service!56
2023-07-18 15:29:21 +02:00
deiflaender 33a4562938 DM-307 Improved headline detection 2023-07-18 15:24:01 +02:00
Kilian Schuettler 9b081e8739 RED-7156: filter out dictionary add manual redaction entries 2023-07-18 14:46:19 +02:00
Kilian Schuettler 68fe33ab70 RED-7156: fix manual redaction page index 2023-07-18 13:53:07 +02:00
Kilian Schuettler 8d66e52cad RED-7156: fix pmd 2023-07-18 10:58:41 +02:00
Kilian Schüttler b80ec83195 Merge branch 'RED-7156' into 'master'
RED-7156: some files stuck in error state

Closes RED-7156

See merge request redactmanager/redaction-service!54
2023-07-18 10:27:51 +02:00
Kilian Schüttler 65f97a6bb9 RED-7156: some files stuck in error state 2023-07-18 10:27:51 +02:00
Kevin Tumma 091895044a Merge branch 'renovate/configure' into 'master'
Configure Renovate

See merge request redactmanager/redaction-service!48
2023-07-17 15:32:31 +02:00
Kevin Tumma 5bf6187a74 Configure Renovate 2023-07-17 15:32:31 +02:00
Dominique Eifländer 6fbbb19b87 Merge branch 'DM-307' into 'master'
DM-307: Undo lineheight experiment

Closes DM-307

See merge request redactmanager/redaction-service!53
2023-07-17 11:42:11 +02:00
deiflaender 66ab995e71 DM-307: Undo lineheight experiment 2023-07-17 11:34:19 +02:00
Dominique Eifländer d337825846 Merge branch 'DM307' into 'master'
DM-307: Improved table merging

See merge request redactmanager/redaction-service!52
2023-07-14 15:57:30 +02:00
deiflaender f366c63985 DM-307: Improved table merging 2023-07-14 15:51:34 +02:00
Kilian Schüttler a7d481f861 Merge branch 'DM-307' into 'master'
DM-307: documine fixes

Closes DM-307

See merge request redactmanager/redaction-service!51
2023-07-14 14:52:10 +02:00
Kilian Schüttler fc233cb56d DM-307: documine fixes 2023-07-14 14:52:09 +02:00
Dominique Eifländer be6fe0b0ca Merge branch 'DM-307' into 'master'
DM-307: Improved paragraph splitting

Closes DM-307

See merge request redactmanager/redaction-service!50
2023-07-14 14:19:28 +02:00
deiflaender 56796ca00e DM-307: Improved paragraph splitting 2023-07-14 14:05:30 +02:00
Dominique Eifländer 2fdaf6d917 Merge branch 'DM-305' into 'master'
DM-305: Improved DocuMine rules

Closes DM-305

See merge request redactmanager/redaction-service!49
2023-07-14 12:31:31 +02:00
deiflaender 18979d0f33 DM-305: Improved DocuMine rules 2023-07-14 12:26:06 +02:00
Dominique Eifländer c5668b738d Merge branch 'DM-307' into 'master'
DM-307: Implement across column function

Closes DM-307

See merge request redactmanager/redaction-service!47
2023-07-14 10:57:14 +02:00
Dominique Eifländer 6565fa1446 DM-307: Implement across column function 2023-07-14 10:57:14 +02:00
Kilian Schüttler 45fe200521 Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!46
2023-07-12 18:40:04 +02:00
Kilian Schüttler 83776b6685 RED-6009: Document Tree Structure 2023-07-12 18:40:04 +02:00
Kilian Schüttler 63f38a8708 Merge branch 'DM-318' into 'master'
DM-318: remove all images

Closes DM-318

See merge request redactmanager/redaction-service!45
2023-07-12 17:07:39 +02:00
Kilian Schuettler 12eee9482d DM-318: remove all images 2023-07-12 17:01:12 +02:00
Kilian Schüttler 36fcc88671 Merge branch 'experimental_features' into 'master'
DM-305: port rules to new schema

See merge request redactmanager/redaction-service!44
2023-07-12 13:44:43 +02:00
Kilian Schüttler 196083df6b DM-305: port rules to new schema 2023-07-12 13:44:43 +02:00
Dominique Eifländer ff749ab88d Merge branch 'DM-307' into 'master'
Resolve DM-307

Closes DM-307

See merge request redactmanager/redaction-service!43
2023-07-11 09:35:18 +02:00
Dominique Eifländer 5ca501da74 Resolve DM-307 2023-07-11 09:35:18 +02:00
Kilian Schüttler dfa66e99db Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/results

Closes RED-6929

See merge request redactmanager/redaction-service!42
2023-07-10 15:14:08 +02:00
Kilian Schüttler a3af762daf RED-6929: fix acceptance tests/results 2023-07-10 15:14:08 +02:00
Dominique Eifländer ca8fd18a0e Merge branch 'DM-305' into 'master'
DM-305: Improved anyHeadlineContains

Closes DM-305

See merge request redactmanager/redaction-service!41
2023-07-10 13:26:28 +02:00
deiflaender 9db58a8b80 DM-305: Improved anyHeadlineContains 2023-07-10 13:18:48 +02:00
Dominique Eifländer 2c937cc6fb Merge branch 'DM-307' into 'master'
DM-307: Implemented betweenStringsInclusive

Closes DM-307

See merge request redactmanager/redaction-service!40
2023-07-07 14:53:40 +02:00
deiflaender 0f5e0ecbad DM-307: Implemented betweenStringsInclusive 2023-07-07 14:47:04 +02:00
Kilian Schüttler df1b4d5178 Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance rules/tests

Closes RED-6929

See merge request redactmanager/redaction-service!39
2023-07-06 18:27:17 +02:00
Kilian Schuettler 9d7180923d RED-6929: fix acceptance rules/tests 2023-07-06 18:22:26 +02:00
Dominique Eifländer 0a47e7fae9 Merge branch 'DM-307' into 'master'
DM-307: Fixed applyWithLineBreaks 3

Closes DM-307

See merge request redactmanager/redaction-service!38
2023-07-06 16:57:13 +02:00
deiflaender 897243be2f DM-307: Fixed applyWithLineBreaks 3 2023-07-06 16:52:03 +02:00
deiflaender 210186bfd7 DM-307: Fixed applyWithLineBreaks 2023-07-06 15:26:06 +02:00
Dominique Eifländer a1ebcc2abf Merge branch 'DM-307' into 'master'
DM-307: Changed rule to applyWithLineBreaks, fixes applyWithLineBreaks

Closes DM-307

See merge request redactmanager/redaction-service!37
2023-07-06 14:27:15 +02:00
deiflaender fa5bc86e0e DM-307: Changed rule to applyWithLineBreaks, fixes applyWithLineBreaks 2023-07-06 14:21:36 +02:00
deiflaender 7171afe5a3 DM-307: Fixed pmd error 2023-07-06 10:27:03 +02:00
Dominique Eifländer 8a011e14bc Merge branch 'DM-307-6' into 'master'
DM-307: Added BodyTextFrameService logic for SCM prototype to fix some missing...

Closes DM-307

See merge request redactmanager/redaction-service!36
2023-07-06 09:53:06 +02:00
Dominique Eifländer edc5833bce DM-307: Added BodyTextFrameService logic for SCM prototype to fix some missing... 2023-07-06 09:53:06 +02:00
Kilian Schüttler 390bb7d381 Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!35
2023-07-06 01:39:39 +02:00
Kilian Schüttler e4d44b8e17 RED-6929: fix acceptance tests/rules 2023-07-06 01:39:39 +02:00
Kilian Schüttler 8b031fa459 Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!34
2023-07-06 00:54:00 +02:00
Kilian Schüttler 4c4885e80c RED-6929: fix acceptance tests/rules 2023-07-06 00:53:59 +02:00
Kilian Schüttler e847aa7ebf Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!33
2023-07-05 23:06:04 +02:00
Kilian Schuettler 97d32912fc RED-6929: fix acceptance tests/rules 2023-07-05 22:58:54 +02:00
Kilian Schüttler ee65044578 Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!32
2023-07-05 21:50:46 +02:00
Kilian Schüttler 0d53edba32 RED-6929: fix acceptance tests/rules 2023-07-05 21:50:45 +02:00
Kilian Schüttler ffb0482ab1 Merge branch 'RED-7082' into 'master'
RED-7082: getBBox() Performance Improvement

Closes RED-7082

See merge request redactmanager/redaction-service!31
2023-07-04 20:19:08 +02:00
Kilian Schüttler 4d00b01514 RED-7082: getBBox() Performance Improvement 2023-07-04 20:19:08 +02:00
Dominique Eifländer ce66d9104d Merge branch 'DM-305' into 'master'
DM-305: Adapted DocuMine rules to latest changes

Closes DM-305

See merge request redactmanager/redaction-service!30
2023-07-04 12:41:26 +02:00
deiflaender 1edfc1b44d DM-305: Adapted DocuMine rules to latest changes 2023-07-04 12:33:25 +02:00
Kilian Schüttler ab282227a8 Merge branch 'RED-6929' into 'master'
RED-6929: Fix Acceptance Tests/Rules

Closes RED-6929

See merge request redactmanager/redaction-service!29
2023-07-03 21:41:34 +02:00
Kilian Schüttler d5bb7d8a0a RED-6929: Fix Acceptance Tests/Rules 2023-07-03 21:41:34 +02:00
Kilian Schüttler cc41838b11 Merge branch 'RED-6929' into 'master'
RED-6929: Fix Acceptance Tests/Rules

Closes RED-6929

See merge request redactmanager/redaction-service!28
2023-07-03 20:19:25 +02:00
Kilian Schüttler 23d14db2d9 RED-6929: Fix Acceptance Tests/Rules 2023-07-03 20:19:25 +02:00
Kilian Schüttler f54727ec8d Merge branch 'RED-6929' into 'master'
RED-6929: Fix Acceptance Tests/Rules

Closes RED-6929

See merge request redactmanager/redaction-service!27
2023-07-03 17:10:08 +02:00
Kilian Schüttler 5625d1ff01 RED-6929: Fix Acceptance Tests/Rules 2023-07-03 17:10:08 +02:00
Dominique Eifländer 41d065afaa Merge branch 'DM-307-5' into 'master'
DM-307: Fixed file in error state because table has no rows

Closes DM-307

See merge request redactmanager/redaction-service!26
2023-06-30 14:10:40 +02:00
deiflaender 62fc31eba5 DM-307: Fixed file in error state because table has no rows 2023-06-30 14:02:50 +02:00
Dominique Eifländer 54111beb76 Merge branch 'DM-307-4' into 'master'
DM-307: Rules that lead to files in error state because section has no paragrapghs

Closes DM-307

See merge request redactmanager/redaction-service!25
2023-06-30 11:51:46 +02:00
deiflaender 27c64acbea DM-307: Rules that lead to files in error state because section has no paragrapghs 2023-06-30 11:43:49 +02:00
Dominique Eifländer a4b86a874b Merge branch 'DM-307-4' into 'master'
DM-307: Fixed bug in rule function byRegexWithLinebreaks

Closes DM-307

See merge request redactmanager/redaction-service!24
2023-06-29 13:22:25 +02:00
deiflaender b1729d9dd6 DM-307: Fixed bug in rule function byRegexWithLinebreaks 2023-06-29 13:14:31 +02:00
Dominique Eifländer 84ac4abf2b Merge branch 'DM-307-3' into 'master'
DM-307: Implemented rule function byRegexWithLinebreaks

Closes DM-307

See merge request redactmanager/redaction-service!23
2023-06-29 12:03:24 +02:00
deiflaender a129953bca DM-307: Implemented rule function byRegexWithLinebreaks 2023-06-29 11:49:33 +02:00
Timo Bejan 92207ed4cc Merge branch 'RED-6686-2' into 'master'
RED-6686 tenant commons update

Closes RED-6686

See merge request redactmanager/redaction-service!22
2023-06-27 23:25:38 +02:00
Timo Bejan 91f8c43465 RED-6686 tenant commons update 2023-06-28 00:16:22 +03:00
Timo Bejan dfea860b01 Merge branch 'RED-6686-2' into 'master'
RED-6686 tenant commons update

Closes RED-6686

See merge request redactmanager/redaction-service!21
2023-06-27 22:50:19 +02:00
Timo Bejan 22261f61f6 RED-6686 tenant commons update 2023-06-27 23:41:26 +03:00
Dominique Eifländer 0fa9f01273 Merge branch 'DM-307-2' into 'master'
DM-307: Enabled to configure custom blockification for DocuMine

Closes DM-307

See merge request redactmanager/redaction-service!20
2023-06-27 17:09:14 +02:00
deiflaender 7230d44a26 DM-307: Enabled to configure custom blockification for DocuMine 2023-06-27 17:02:05 +02:00
Dominique Eifländer 9ad4508ea0 Merge branch 'DM-307-2' into 'master'
DM-307: Added rules function to check if section has paragraphs

Closes DM-307

See merge request redactmanager/redaction-service!19
2023-06-27 16:29:13 +02:00
deiflaender 83329dd0c1 DM-307: Added rules function to check if section has paragraphs 2023-06-27 15:48:53 +02:00
Dominique Eifländer 6c38dbb2d3 Merge branch 'DM-307' into 'master'
DM-307: Enabled to configure DocuMine Paragraph classifications

Closes DM-307

See merge request redactmanager/redaction-service!18
2023-06-27 11:57:29 +02:00
deiflaender b1b0e3efa2 DM-307: Enabled to configure DocuMine Paragraph classifications 2023-06-27 11:47:16 +02:00
Timo Bejan 0f4effa68b Merge branch 'RED-6686' into 'master'
RED-6686 Extract Tenant and user-management code into a separate service.

See merge request redactmanager/redaction-service!2
2023-06-26 22:02:55 +02:00
Timo Bejan faac39c796 RED-6686 Extract Tenant and user-management code into a separate service. 2023-06-26 22:02:55 +02:00
Dominique Eifländer 6e9c32678a Merge branch 'DM-305-2' into 'master'
DM-305: Improved rules for DocuMine

Closes DM-305

See merge request redactmanager/redaction-service!16
2023-06-26 16:48:41 +02:00
deiflaender a36dcc5db1 DM-305: Improved rules for DocuMine 2023-06-26 16:40:35 +02:00
Dominique Eifländer 78a42ed86a Merge branch 'DM-305' into 'master'
DM-305: Implemented anyHeadlineContainsString for DocuMine and fixed some DocuMine Rules

Closes DM-305

See merge request redactmanager/redaction-service!15
2023-06-26 14:07:55 +02:00
deiflaender 3447ee1856 DM-305: Implemented anyHeadlineContainsString for DocuMine and fixed some DocuMine Rules 2023-06-26 14:00:23 +02:00
Dominique Eifländer 04656acd9d Merge branch 'DM-305' into 'master'
DM-305: Added valueEqualsAnyOf to FileAttribute and test for DocuMine

Closes DM-305

See merge request redactmanager/redaction-service!14
2023-06-23 14:30:13 +02:00
deiflaender aed3256d24 DM-305: Added valueEqualsAnyOf to FileAttribute and test for DocuMine 2023-06-23 14:20:32 +02:00
Kilian Schüttler b5bfa68b2d Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!13
2023-06-22 18:05:08 +02:00
Kilian Schüttler beb1e8b6b1 RED-6929: fix acceptance tests/rules 2023-06-22 18:05:08 +02:00
Kilian Schüttler 48d9d9a9a4 Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!12
2023-06-22 14:42:20 +02:00
Kilian Schüttler cd942188b9 RED-6929: fix acceptance tests/rules 2023-06-22 14:42:20 +02:00
Kilian Schüttler 39790c0c8f Merge branch 'RED-6929' into 'master'
RED-6929: fix acceptance tests/rules

Closes RED-6929

See merge request redactmanager/redaction-service!11
2023-06-22 13:52:45 +02:00
Kilian Schüttler 7c6f8210e7 RED-6929: fix acceptance tests/rules 2023-06-22 13:52:44 +02:00
Kilian Schüttler 847a50c32a Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!10
2023-06-21 13:21:06 +02:00
Kilian Schüttler 529269e66b RED-6009: Document Tree Structure 2023-06-21 13:21:06 +02:00
Kilian Schüttler 7915b235d2 Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!9
2023-06-20 16:58:59 +02:00
Kilian Schüttler ad0b0f3f8d RED-6009: Document Tree Structure 2023-06-20 16:58:59 +02:00
Corina Olariu 778286e3b8 RED-6734 - Get merged dossier and template dictionaries
- update to the removing of logic of merge dictionaries to dictionary-merge commons
2023-06-20 15:27:11 +03:00
Dominique Eifländer 57ba08d5b4 Merge branch 'RED-6935' into 'master'
RED-6935: Fixed null values in file attributes

Closes RED-6935

See merge request redactmanager/redaction-service!8
2023-06-20 12:17:44 +02:00
deiflaender cb03b9653c RED-6935: Fixed null values in file attributes 2023-06-20 12:09:17 +02:00
Dominique Eifländer 9f921f3002 Merge branch 'RED-6009' into 'master'
RED-6009: Sort TextpositionSequences with tolerance

Closes RED-6009

See merge request redactmanager/redaction-service!7
2023-06-19 11:51:12 +02:00
deiflaender 77a420c849 RED-6009: Sort TextpositionSequences with tolerance 2023-06-19 11:40:23 +02:00
Kilian Schüttler d0264248fb Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!6
2023-06-15 22:11:48 +02:00
Kilian Schüttler de96e29b09 RED-6009: Document Tree Structure 2023-06-15 22:11:48 +02:00
Kilian Schüttler f290df40b1 Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!5
2023-06-15 21:44:25 +02:00
Kilian Schüttler bc605aec8c RED-6009: Document Tree Structure 2023-06-15 21:44:25 +02:00
Kilian Schüttler 7694c11dd6 Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!4
2023-06-15 21:07:48 +02:00
Kilian Schüttler 2a87eede6d RED-6009: Document Tree Structure 2023-06-15 21:07:47 +02:00
Dominique Eifländer 108da249fa Merge branch 'RED-6072' into 'master'
RED-6072 - As Operation I want to see why files are in an ERROR state

Closes RED-6072

See merge request redactmanager/redaction-service!3
2023-06-15 14:07:50 +02:00
Corina Olariu c508f8c0ea RED-6072 - As Operation I want to see why files are in an ERROR state 2023-06-15 14:07:49 +02:00
Kilian Schüttler c54dc0ab43 Merge branch 'RED-6009' into 'master'
RED-6009: Document Tree Structure

Closes RED-6009

See merge request redactmanager/redaction-service!1
2023-06-14 09:49:31 +02:00
Kilian Schuettler 1f9e151092 RED-6009: Document Tree Structure
* squashed commits
2023-06-06 19:37:27 +02:00
Christoph Schabert a6a6fd8180 Update file pom.xml 2023-06-01 11:21:14 +02:00
Christoph Schabert aa0167525b Update 7 files
- /bamboo-specs/src/main/java/buildjob/PlanSpec.java
- /bamboo-specs/src/main/resources/scripts/build-java.sh
- /bamboo-specs/src/main/resources/scripts/sonar-java.sh
- /bamboo-specs/src/test/java/buildjob/PlanSpecTest.java
- /bamboo-specs/pom.xml
- /pom.xml
- /.gitlab-ci.yml
2023-06-01 11:11:24 +02:00
Corina Olariu cdb0a418f8 Pull request #537: RED-6744
Merge in RED/redaction-service from RED-6744 to master

* commit '809349f65d7aa4af9ac5ca2fa75d80548d2f475b':
  RED-6744 - Merge dossier and template dictionaries in redaction-service - reformat - add junit test
  RED-6744 - Merge dossier and template dictionaries in redaction-service - remove unnecessary annotations
  RED-6744 - erge dossier and template dictionaries in redaction-service - extend DictionaryEntry from persistence to use it in merging dictionaries and to override the hascode and equals - use this DictionaryEntryModel instead of DictionaryEntry
2023-05-31 10:19:32 +02:00
devplant 809349f65d RED-6744 - Merge dossier and template dictionaries in redaction-service
- reformat
- add junit test
2023-05-26 10:17:48 +03:00
devplant 9cfbf62bc5 RED-6744 - Merge dossier and template dictionaries in redaction-service
- remove unnecessary annotations
2023-05-24 12:23:04 +03:00
devplant 2c6e397247 RED-6744 - erge dossier and template dictionaries in redaction-service
- extend DictionaryEntry from persistence to use it in merging dictionaries and to override the hascode and equals
- use this DictionaryEntryModel instead of DictionaryEntry
2023-05-23 17:05:53 +03:00
Thomas Beyer a6be7c9cfd Pull request #534: RED-6619 5
Merge in RED/redaction-service from RED-6619_5 to master

* commit 'f574b414c471cb6f349a5fd44a1ad337c77521ac':
  RED-6619 - change pom to fforesight again
  RED-6619 - revert changes in pom
  RED-6619 - change from iqser to knecon
  RED-6619 - change nexus-url from iqser to knecon
2023-05-03 09:43:35 +02:00
Thomas Beyer f574b414c4 RED-6619 - change pom to fforesight again 2023-05-03 08:55:42 +02:00
Thomas Beyer a427e676f7 RED-6619 - revert changes in pom 2023-05-02 16:41:55 +02:00
Thomas Beyer a8dd0ccd47 RED-6619 - change from iqser to knecon 2023-05-02 15:51:28 +02:00
Thomas Beyer fca35c97ca RED-6619 - change nexus-url from iqser to knecon 2023-05-02 15:10:09 +02:00
Thomas Beyer 4bf686a432 Pull request #532: RED-6619 1
Merge in RED/redaction-service from RED-6619_1 to master

* commit 'aff7074b4012627a4e8b64b2f4a779b18feed50a':
  RED-6619 - renamed variables
  RED-6619 - reformat code
  RED-6619 - delete unnecessary import
  RED-6619 - moved not the logic to a boolean, but the 1 into a constant
  RED-6619 - fix integration-tests by adding versions and move the hasMinimumSize-logic into own boolea
  RED-6619 - added missing ' (typo)
  RED-6619 - add logic to ignore found table-cells with height or width < 1. Also: Fix the tests and add new segmentation-tests and 1 redaction-integration-test. Renamed the latter to fit maven regexp
  RED-6619 - add tests for table-extraction
2023-05-02 10:54:30 +02:00
Thomas Beyer aff7074b40 RED-6619 - renamed variables 2023-04-28 16:54:16 +02:00
Thomas Beyer 2e37cd5669 RED-6619 - reformat code 2023-04-28 12:32:11 +02:00
Thomas Beyer 794a160115 RED-6619 - delete unnecessary import 2023-04-28 12:22:11 +02:00
Thomas Beyer e2d3972e16 RED-6619 - moved not the logic to a boolean, but the 1 into a constant 2023-04-28 12:19:14 +02:00
Thomas Beyer 4a76e89ab8 RED-6619 - fix integration-tests by adding versions and move the hasMinimumSize-logic into own boolea 2023-04-28 12:13:54 +02:00
Thomas Beyer fcc4085321 RED-6619 - added missing ' (typo) 2023-04-27 17:52:03 +02:00
Thomas Beyer 8490690001 RED-6619 - add logic to ignore found table-cells with height or width < 1.
Also: Fix the tests and add new segmentation-tests and 1 redaction-integration-test. Renamed the latter to fit maven regexp
2023-04-27 17:49:36 +02:00
Thomas Beyer 6f783a9f00 RED-6619 - add tests for table-extraction 2023-04-27 14:04:22 +02:00
Corina Olariu 2646407805 Pull request #531: RED-5694 - Upgrade spring-boot to 3.0
Merge in RED/redaction-service from RED-5694-storage to master

* commit '9a933e9563769dbbc38756151ef7d33fde859194':
  RED-5694 - Upgrade spring-boot to 3.0 - import storageAutoConfiguration to Application
2023-04-21 07:21:48 +02:00
devplant 9a933e9563 RED-5694 - Upgrade spring-boot to 3.0
- import storageAutoConfiguration to Application
2023-04-20 13:03:39 +03:00
Corina Olariu 999973a00f Pull request #530: RED-5694
Merge in RED/redaction-service from RED-5694 to master

* commit '7d4b3f40d3a09493459d6f71e53275fc1da7adad':
  RED-5694 - Upgrade spring-boot to 3.0 - exclude storageAutoConfiguration with @ComponentScan
  RED-5694 - Upgrade spring-boot to 3.0 - remove comment code
  RED-5694 - Upgrade spring-boot to 3.0 - remove comment code
  RED-5694 - Upgrade spring-boot to 3.0 - update exclusion of StorageAutoConfiguration
  RED-5694 - Upgrade spring-boot to 3.0 - update after the merge of master
  RED-5694 - Upgrade spring-boot to 3.0 - update platfrom-dependency and other dependencies with latest versions - remove dslplatform dependency
  RED-5694: Imported new version of platform-dependency and fixed a couple of issues.
2023-04-05 13:48:04 +02:00
devplant 7d4b3f40d3 RED-5694 - Upgrade spring-boot to 3.0
- exclude storageAutoConfiguration with @ComponentScan
2023-04-05 14:09:02 +03:00
devplant d366d9207a RED-5694 - Upgrade spring-boot to 3.0
- remove comment code
2023-04-05 13:48:28 +03:00
devplant e22e16aef1 RED-5694 - Upgrade spring-boot to 3.0
- remove comment code
2023-04-05 13:43:45 +03:00
devplant 077677e02d RED-5694 - Upgrade spring-boot to 3.0
- update exclusion of StorageAutoConfiguration
2023-04-05 13:22:17 +03:00
devplant 09de0dc2c6 RED-5694 - Upgrade spring-boot to 3.0
- update after the merge of master
2023-04-05 12:06:12 +03:00
devplant c3b29e4ebc Merge branch 'master' of https://git.iqser.com/scm/red/redaction-service into RED-5694
# Conflicts:
#	redaction-service-v1/redaction-service-server-v1/src/test/java/com/iqser/red/service/redaction/v1/server/RedactionIntegrationTest.java
2023-04-05 11:37:33 +03:00
devplant adcf17c5f0 RED-5694 - Upgrade spring-boot to 3.0
- update platfrom-dependency and other dependencies with latest versions
- remove dslplatform dependency
2023-04-05 10:58:18 +03:00
Thomas Beyer 49860ab95d Pull request #528: RED-6411 4.5.0 1
Merge in RED/redaction-service from RED-6411_4.5.0_1 to master

* commit '2a56187294c125931d4b998ee1f812ea426e2b58':
  RED-6411 - optimize code
  RED-6411 - optimize method
  RED-6411 - rename testclass because of regexp
  RED-6411 - fix import of configuration
  RED-6411 - add test for this case (with the current rules-file) and move/abstract logic from the RedactionIntegrationTest into AbstractRedactionIntegrationTest
  RED-6411 - extend logic when replacing entities with higher rank
  RED-6411 - remove false positives before calling EntitySearchUtils.addOrAddEngine()
2023-03-29 15:03:13 +02:00
Thomas Beyer 2a56187294 RED-6411 - optimize code 2023-03-29 09:35:55 +02:00
Thomas Beyer 710747d00b RED-6411 - optimize method 2023-03-28 11:28:43 +02:00
Thomas Beyer 4e76c007b8 RED-6411 - rename testclass because of regexp 2023-03-28 10:41:37 +02:00
Thomas Beyer 6e5b64ca94 RED-6411 - fix import of configuration 2023-03-28 10:35:27 +02:00
Thomas Beyer 157b20a1de RED-6411 - add test for this case (with the current rules-file) and move/abstract logic from the RedactionIntegrationTest into AbstractRedactionIntegrationTest 2023-03-28 10:16:17 +02:00
Thomas Beyer 7644afc3aa RED-6411 - extend logic when replacing entities with higher rank 2023-03-27 12:49:33 +02:00
Dominique Eiflaender 89e09f780d Pull request #525: RED-6224: Multitenancy for rules cache
Merge in RED/redaction-service from RED-6224 to master

* commit 'f03523ca8f49767b22d551023866dc9948c06e9f':
  RED-6224: Multitenancy for rules cache
2023-03-27 09:44:46 +02:00
deiflaender f03523ca8f RED-6224: Multitenancy for rules cache 2023-03-27 09:37:48 +02:00
Thomas Beyer 459af9e33b RED-6411 - remove false positives before calling EntitySearchUtils.addOrAddEngine() 2023-03-27 09:05:13 +02:00
Viktor Seifert 3626dbd0a9 RED-5694: Imported new version of platform-dependency and fixed a couple of issues.
* Updated imports for some javax.* packages.
* Added improved config for annotation processors (not dependent on implementation details).
* Updated SectionText class to not use wildcard imports, since the cause problems for lombok + dsl-json.
2023-03-20 18:06:32 +01:00
Dominique Eiflaender cec18a472f Pull request #524: RED-6224: Made dictionary cache multitenancy ready
Merge in RED/redaction-service from RED-6224 to master

* commit '339833e2d6f4f85757f7fdb84c2006f17d766f38':
  RED-6224: Fixed pr findings
  RED-6224: Made dictionary cache multitenancy ready
2023-03-17 16:17:57 +01:00
deiflaender 339833e2d6 RED-6224: Fixed pr findings 2023-03-17 16:07:20 +01:00
deiflaender ec0de5b6a2 RED-6224: Made dictionary cache multitenancy ready 2023-03-17 13:54:22 +01:00
Viktor Seifert 4b42c8d13f Pull request #523: RED-6250
Merge in RED/redaction-service from RED-6250 to master

* commit 'fe5121b1edf784d5cc57abd9cd8e27f70dea0f4a':
  RED-6250: Switched to new metrics-commons version to not have to instantiate a factory (reduces code duplication)
  RED-6250: Changed function-timer metric to use code from metric commons
2023-03-16 10:33:38 +01:00
Viktor Seifert fe5121b1ed RED-6250: Switched to new metrics-commons version to not have to instantiate a factory (reduces code duplication) 2023-03-15 18:16:53 +01:00
Viktor Seifert 519bb9cb52 RED-6250: Changed function-timer metric to use code from metric commons 2023-03-15 16:40:04 +01:00
Timo Bejan 6c0886e2c4 Pull request #522: RED-6162 - bumped version
Merge in RED/redaction-service from RED-6162 to master

* commit '9f396cccdf4b6762904c12923c6ba278398869ca':
  RED-6162 - bumped version
2023-03-10 21:44:17 +01:00
Dominique Eiflaender 9f70fb75dd Pull request #521: RED-4645: Multitenancy for storage
Merge in RED/redaction-service from RED-4645 to master

* commit 'c13d148ae7bc4a8f9e4d9355fc93e66e97603eac':
  RED-4645: Multitenancy for storage
2023-03-10 16:44:18 +01:00
deiflaender c13d148ae7 RED-4645: Multitenancy for storage 2023-03-10 16:21:54 +01:00
Timo Bejan 9f396cccdf RED-6162 - bumped version 2023-03-10 15:58:29 +02:00
Timo Bejan 5860daa6fb Pull request #520: RED-6162 - bumped version
Merge in RED/redaction-service from RED-6162 to master

* commit '08d20e30e347c79ab4667a524ab814e160778074':
  RED-6162 - bumped version
2023-03-10 14:56:01 +01:00
Timo Bejan 08d20e30e3 RED-6162 - bumped version 2023-03-10 15:50:41 +02:00
Timo Bejan c3e0364d80 Pull request #519: RED-6162 Redaction Gateway - Persistence Service Merge Updates
Merge in RED/redaction-service from RED-6162 to master

* commit '56f98549e4b8f5baf4ab479b81290c98e38a53cd':
  RED-6182 - version bump
  RED-6182 - version bump
  RED-6162 - persistence update - reverse dependency cleanup
  RED-6162 - persistence update
2023-03-10 11:27:18 +01:00
Timo Bejan 56f98549e4 RED-6182 - version bump 2023-03-10 08:46:34 +02:00
Timo Bejan 5ca2e5af23 RED-6182 - version bump 2023-03-10 08:46:34 +02:00
Timo Bejan 99b25a4ccb RED-6162 - persistence update - reverse dependency cleanup 2023-03-10 08:46:34 +02:00
Timo Bejan 4c153fc5f8 RED-6162 - persistence update 2023-03-10 08:46:34 +02:00
Corina Olariu e42ecad4e4 Pull request #518: RED-4988 Check jacoco version in poms and update to a current compatible version
Merge in RED/redaction-service from RED-4988-cisa to master

* commit 'a73373ba42e6098e093fcd58a458902d4ca428fe':
  RED-4988 Check jacoco version in poms and update to a current compatible version - add -DknownExploitedEnabled=false for dependency-check:aggregate
2023-03-08 12:08:03 +01:00
devplant a73373ba42 RED-4988 Check jacoco version in poms and update to a current compatible version
- add -DknownExploitedEnabled=false for dependency-check:aggregate
2023-03-08 12:56:09 +02:00
Viktor Seifert cab54f6cec Pull request #517: RED-6264
Merge in RED/redaction-service from RED-6264 to master

* commit 'f9d61a57c1813d34219c6442108af34b411094f7':
  RED-6264: Corrected metric name and added clarifying comment
  RED-6264: Implemented metric to show analyze time correlated with pages
2023-03-03 09:47:51 +01:00
Viktor Seifert f9d61a57c1 RED-6264: Corrected metric name and added clarifying comment 2023-03-02 14:42:14 +01:00
Viktor Seifert c6837d41af RED-6264: Implemented metric to show analyze time correlated with pages 2023-03-01 16:39:22 +01:00
Viktor Seifert 8ce093090e Pull request #516: RED-6204
Merge in RED/redaction-service from RED-6204 to master

* commit '6e73cac99c99fc333b9ec29b4a16aad48ffba149':
  RED-6204: Added annotation to suppress a false-positive warning in SonarQube
  RED-6204: Corrected typo in comment
  RED-6204: Simplified conversion of File to Path
  RED-6204: Corrected temp file deletion in PdfSegmentationService.
  RED-6204: Upgraded to newest platform-dependency and migrated tests to Junit5
2023-02-24 17:29:43 +01:00
Viktor Seifert 6e73cac99c RED-6204: Added annotation to suppress a false-positive warning in SonarQube 2023-02-24 17:19:52 +01:00
Viktor Seifert dcf040b000 RED-6204: Corrected typo in comment 2023-02-24 15:18:00 +01:00
Viktor Seifert 196cd5934b RED-6204: Simplified conversion of File to Path 2023-02-24 14:01:57 +01:00
Viktor Seifert 9931b56b92 Merge branch 'master' into RED-6204 2023-02-24 13:55:19 +01:00
Viktor Seifert 6ae4d16bb7 RED-6204: Corrected temp file deletion in PdfSegmentationService.
Previously files were not deleted correctly, because the code tried to delete the file while a file-stream was still open.
2023-02-24 13:32:27 +01:00
Viktor Seifert 4b21418163 RED-6204: Upgraded to newest platform-dependency and migrated tests to Junit5 2023-02-24 12:13:43 +01:00
Viktor Seifert f7ec180710 Pull request #515: RED-6204
Merge in RED/redaction-service from RED-6204 to master

* commit '3fad6381ce71f36083d6f545e1b9c47cecda3ef1':
  RED-6204: Removed redundant parenthesis
  RED-6204: Removed redundant variable assignment (sonar issue) & simplified code
  RED-6204: Remove unused import (sonar issue)
  RED-6204: Moved code to its own class for metrics.
  RED-6204: Moved code to its own class for metrics.
  RED-6204: Remove AspectJ mode setting, since it would require a couple of AspectJ dependencies for a very limited use case
  RED-6204: Switched to AspectJ to enable proxies on private methods
2023-02-24 10:00:01 +01:00
Viktor Seifert 3fad6381ce RED-6204: Removed redundant parenthesis 2023-02-23 16:57:58 +01:00
Viktor Seifert 3804acf3c1 RED-6204: Removed redundant variable assignment (sonar issue) & simplified code 2023-02-23 16:49:21 +01:00
Viktor Seifert 8ed976b0f7 RED-6204: Remove unused import (sonar issue) 2023-02-23 16:36:17 +01:00
Viktor Seifert 8fac22b8ab RED-6204: Moved code to its own class for metrics.
Moved a private method to find sections to its own class, so that it can produce a separate metric value.
2023-02-23 11:57:36 +01:00
Viktor Seifert 00ef0eb677 RED-6204: Moved code to its own class for metrics.
Moved a private method to find entities to its own class, so that it can produce a separate metric value.
2023-02-22 17:45:30 +01:00
Viktor Seifert f9b7ad4e3e RED-6204: Remove AspectJ mode setting, since it would require a couple of AspectJ dependencies for a very limited use case 2023-02-22 16:33:04 +01:00
Viktor Seifert d65ff273a6 RED-6204: Switched to AspectJ to enable proxies on private methods 2023-02-22 12:39:47 +01:00
Dominique Eiflaender 04343334d8 Pull request #514: RED-6164: Fixed calculation of image is ocr on scanned pages with cv analysis found tables
Merge in RED/redaction-service from RED-6164 to master

* commit '0cf867b97c2a240f2bbdca479ebdc36f64ed3bbb':
  RED-6164: Fixed calculation of image is ocr on scanned pages with cv analysis found tables
2023-02-17 14:36:09 +01:00
deiflaender 0cf867b97c RED-6164: Fixed calculation of image is ocr on scanned pages with cv analysis found tables 2023-02-17 12:13:39 +01:00
Dominique Eiflaender f8a0a911bc Pull request #513: RED-5664: Enabled to redact words that start or end with seperator, needed for japan documents
Merge in RED/redaction-service from RED-5664 to master

* commit '1fca62f5782a7b2512092389c13195415bd5df9c':
  RED-5664: Enabled to redact words that start or end with seperator, needed for japan documents
2023-02-13 10:38:19 +01:00
deiflaender 1fca62f578 RED-5664: Enabled to redact words that start or end with seperator, needed for japan documents 2023-02-13 10:33:22 +01:00
Timo Bejan 0e925f2f24 Pull request #512: RED-4609 - adjusted some metrics, added tests for metrics
Merge in RED/redaction-service from RED-4609 to master

* commit 'e23432096cb0ed486a294ed210646bbd8682350e':
  RED-4609 - adjusted some metrics, added tests for metrics
2023-02-09 09:54:19 +01:00
Timo Bejan e23432096c RED-4609 - adjusted some metrics, added tests for metrics 2023-02-08 19:12:47 +02:00
Dominique Eiflaender c1e2b8da29 Pull request #511: RED-5276: Fixed strange behavior of text parsing for tables example document
Merge in RED/redaction-service from RED-5276-1 to master

* commit '16b04b5918a3c8cac0070125d0e984bcd55b9b70':
  RED-5276: Fixed strange behavior of text parsing for tables example document
2023-01-31 11:16:07 +01:00
deiflaender 16b04b5918 RED-5276: Fixed strange behavior of text parsing for tables example document 2023-01-31 11:03:55 +01:00
Philipp Schramm c16b6d41d5 Pull request #510: RED-5248: Fix handling of temp files
Merge in RED/redaction-service from RED-5248 to master

* commit 'b839c4e3aea67c94fa29d476f63673fc232dcdab':
  RED-5248: Fix handling of temp files
2023-01-30 12:24:54 +01:00
Philipp Schramm b839c4e3ae RED-5248: Fix handling of temp files 2023-01-30 11:51:14 +01:00
Philipp Schramm 6fd6caa8ad Pull request #509: RED-5917: Wrong value set for Signature and Logo after Resize
Merge in RED/redaction-service from RED-5917 to master

* commit '41282c0edd1cbc98357ba04e33731d5b079064b8':
  RED-5917: Wrong value set for Signature and Logo after Resize
2023-01-19 13:00:30 +01:00
Philipp Schramm 41282c0edd RED-5917: Wrong value set for Signature and Logo after Resize 2023-01-19 11:25:26 +01:00
Timo Bejan 3c79e65345 Pull request #508: RED-5981 remove from dictionary pending analhysis
Merge in RED/redaction-service from RED-5981 to master

* commit 'd2eeaa91a6857b779996e5f5e9254bd4a460ac63':
  RED-5981 remove from dictionary pending analhysis
2023-01-15 09:15:54 +01:00
Timo Bejan d2eeaa91a6 RED-5981 remove from dictionary pending analhysis 2023-01-15 16:05:21 +08:00
Dominique Eiflaender 53a375b832 Pull request #507: RED-5276: Imporved table calculation, support spanned rows and colmns
Merge in RED/redaction-service from RED-5276-1 to master

* commit 'd233c18d335d17d3c79590dd3a295eaa89881de2':
  RED-5276: Imporved table calculation, support spanned rows and colmns
2022-12-23 11:25:43 +01:00
deiflaender d233c18d33 RED-5276: Imporved table calculation, support spanned rows and colmns 2022-12-23 11:20:05 +01:00
Philipp Schramm b74673ae63 Pull request #506: RED-5249: Marked utils classes with @UtilityClass
Merge in RED/redaction-service from RED-5249 to master

* commit 'faa702d3f499b89dfd9a7dedf8c16af89128c77d':
  RED-5249: Marked utils classes with @UtilityClass
2022-12-15 09:49:06 +01:00
Philipp Schramm faa702d3f4 RED-5249: Marked utils classes with @UtilityClass 2022-12-14 15:27:23 +01:00
Corina Olariu 05bf95d62b Pull request #505: RED-5748 - Redaction Service Error after helm upgrade
Merge in RED/redaction-service from RED-5748 to master

* commit '776de8392a78f7b26e864e37f4be607d1ff7e48e':
  RED-5748 - Redaction Service Error after helm upgrade - in case of deleted types throw a NotFoundException providing the info about the type not found instead of NullPointerException
2022-12-14 11:04:10 +01:00
devplant 776de8392a RED-5748 - Redaction Service Error after helm upgrade
- in case of deleted types throw a NotFoundException providing the info about the type not found instead of NullPointerException
2022-12-13 16:39:23 +02:00
Viktor Seifert 660abb318f Pull request #504: RED-5670: Update platform version to include security fixes from spring-boot
Merge in RED/redaction-service from RED-5670 to master

* commit '8b74142d319718cc448c9ddbbb40fa1b30c29b18':
  RED-5670: Update platform version to include security fixes from spring-boot
2022-11-25 16:24:54 +01:00
Viktor Seifert 8b74142d31 RED-5670: Update platform version to include security fixes from spring-boot 2022-11-25 16:14:34 +01:00
Corina Olariu 467a242a3d Pull request #503: RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED
Merge in RED/redaction-service from RED-5469 to master

* commit '87b11842a6606cd24155b8e9b0a82932267512bb':
  RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED - renamed from cvServiceEnabled to cvTableParsingEnabled
2022-11-18 11:25:10 +01:00
devplant 87b11842a6 RED-5469 - Rename CVSERVICEENABLED to CVTABLEPARSINGENABLED
- renamed from cvServiceEnabled to cvTableParsingEnabled
2022-11-17 14:23:06 +02:00
Timo Bejan 8f56d50322 Pull request #501: RED-5562 more symbols
Merge in RED/redaction-service from RED-5562-mst to master

* commit '38ce801f2d9e8d2cc83fea4e3ca56d73738f7bb1':
  RED-5562 more symbols
2022-11-15 11:10:01 +01:00
Timo Bejan 38ce801f2d RED-5562 more symbols 2022-11-15 11:29:17 +02:00
Timo Bejan 91e227248d Pull request #498: RED-5562 japanse space characters
Merge in RED/redaction-service from RED-5562-mst to master

* commit '623b8df5e6136d2ae6649e5016933abdc40454cc':
  RED-5562 improved pattern compile
  RED-5562 japanse space characters - fixed PMD
  RED-5562 japanse space characters
2022-11-15 10:05:33 +01:00
Timo Bejan 623b8df5e6 RED-5562 improved pattern compile 2022-11-15 01:38:33 +02:00
Timo Bejan 20ab65afd2 RED-5562 japanse space characters - fixed PMD 2022-11-15 00:57:54 +02:00
Timo Bejan 1ce47b7fbc RED-5562 japanse space characters 2022-11-14 23:31:31 +02:00
Dominique Eiflaender 5feb6891e2 Pull request #496: RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section
Merge in RED/redaction-service from RSS-164-2 to master

* commit '02bdbbc2d1f6bff0c907240f61726a78a0b8f318':
  RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section
2022-11-04 13:37:20 +01:00
deiflaender 02bdbbc2d1 RSS-164: Added new rule function for redactLineAfterAcrossColumns with param to return only exactMatch in the section 2022-11-04 13:32:09 +01:00
Dominique Eiflaender 19e607e8a8 Pull request #495: RSS-177: Added required rule functions for scm poc
Merge in RED/redaction-service from RSS-177 to master

* commit '5c38150d34d8558f0e0e19ecd340d459a132426d':
  RSS-177: Added required rule functions for scm poc
2022-11-03 14:22:40 +01:00
deiflaender 5c38150d34 RSS-177: Added required rule functions for scm poc 2022-11-03 12:50:29 +01:00
Dominique Eiflaender 18487c639b Pull request #494: RSS-145: Added rules to add new FileAttributes
Merge in RED/redaction-service from RSS-145 to master

* commit 'e2234dc52a8a27a285c0e20b8326bd7129fc7dbf':
  RSS-145: Fixed Immutable list exception when merging existing and added fileattributes
  RSS-145: Added rules to add new FileAttributes
2022-10-28 13:07:20 +02:00
deiflaender e2234dc52a RSS-145: Fixed Immutable list exception when merging existing and added fileattributes 2022-10-28 12:55:52 +02:00
deiflaender 503df78f88 RSS-145: Added rules to add new FileAttributes 2022-10-28 12:33:45 +02:00
Philipp Schramm 5527eaec4e Pull request #492: RSS-118: Refactored method in Section for adding AI entries
Merge in RED/redaction-service from RSS-118 to master

* commit 'b4f079c3c2a5b26828a81b1c950d0f7e40c6314a':
  RSS-118: Refactored method in Section for adding AI entries
2022-10-27 14:20:49 +02:00
Philipp Schramm b4f079c3c2 RSS-118: Refactored method in Section for adding AI entries 2022-10-27 14:15:42 +02:00
Dominique Eiflaender 1e6e5e2154 Pull request #490: RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages
Merge in RED/redaction-service from RED-5381 to master

* commit '17bdcf8d2469429106234d9ccfdb9f2db17d9b84':
  RED-5381: Fixed pr findings
  RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages
2022-10-21 13:00:08 +02:00
deiflaender 17bdcf8d24 RED-5381: Fixed pr findings 2022-10-21 12:40:10 +02:00
deiflaender aa43453206 RED-5381: Fixed calculation of textblocks and body text frame for rotated text and rotated pages 2022-10-21 11:42:39 +02:00
Philipp Schramm ddbf80e4a6 Pull request #486: RED-5232: Fixed NPE
Merge in RED/redaction-service from RED-5232 to master

* commit '2e3d4ad361ee9453ebcea3645014bce0fd4af22c':
  RED-5232: Fixed NPE
2022-10-17 15:55:30 +02:00
Philipp Schramm 2e3d4ad361 RED-5232: Fixed NPE 2022-10-17 15:51:21 +02:00
Ali Oezyetimoglu b8320bd000 Pull request #485: RED-5232: Code reformatting
Merge in RED/redaction-service from RED-5232 to master

* commit '76fda2b573d0fa1aaf73d4ccc7af9496ac257fa1':
  RED-5232: Code reformatting
2022-10-17 15:04:39 +02:00
Ali Oezyetimoglu 76fda2b573 RED-5232: Code reformatting 2022-10-17 14:58:26 +02:00
Dominique Eiflaender 69540bcd5e Pull request #482: RED-5295: Added redactWordPartByRegEx rule function
Merge in RED/redaction-service from RED-5295 to master

* commit 'e0dd06c6bf64bccff41f425cd934ebbe934384cf':
  RED-5295: Added redactWordPartByRegEx rule function
2022-10-12 10:49:11 +02:00
Dominique Eiflaender 8ab2738cd0 Pull request #484: RED-5275: Calculate precision and recall for headline detection
Merge in RED/redaction-service from RED-5275 to master

* commit '8d88b19915f0eaf83dc4174635585081f0468db2':
  RED-5275: Calculate precision and recall for headline detection
2022-10-12 10:48:56 +02:00
deiflaender 8d88b19915 RED-5275: Calculate precision and recall for headline detection 2022-10-11 11:05:19 +02:00
Philipp Schramm 9d88925ff1 Pull request #483: RED-5028: Integrated cv table service
Merge in RED/redaction-service from RED-5028 to master

* commit 'f6bc49d42c65a8580a5558891cabd4738af01d87':
  RED-5028: Integrated cv table service
2022-10-11 09:03:07 +02:00
Philipp Schramm f6bc49d42c RED-5028: Integrated cv table service 2022-10-11 08:07:55 +02:00
deiflaender e0dd06c6bf RED-5295: Added redactWordPartByRegEx rule function 2022-10-05 11:32:33 +02:00
Dominique Eiflaender 97209a3508 Pull request #481: RED-5275: Fixed not all headline were found because headline contains newlines
Merge in RED/redaction-service from RED-5275 to master

* commit '074205aa4d4532d5143463ab80199be08e1599a1':
  RED-5275: Fixed not all headline were found because headline contains newlines
2022-09-28 15:16:08 +02:00
deiflaender 074205aa4d RED-5275: Fixed not all headline were found because headline contains newlines 2022-09-28 15:12:49 +02:00
Dominique Eiflaender 2f88d8083c Pull request #480: RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule
Merge in RED/redaction-service from RED-5275 to master

* commit 'a510a8bb9f6796da4652b45fdefbf3c5f3f2ac92':
  RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule
2022-09-28 13:32:12 +02:00
deiflaender a510a8bb9f RED-5275: Fixed ArrayOutOfBounds if headline is empty in redactHeadline rule 2022-09-28 13:25:03 +02:00
Dominique Eiflaender c027924b19 Pull request #478: RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco
Merge in RED/redaction-service from RED-4988 to master

* commit '313cf68044513f354d297162b1789ce9ce1f171c':
  RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco
2022-09-26 16:20:41 +02:00
deiflaender 313cf68044 RED-4988: Updated drools to latest final version, updated jacoco, removed drools from jacoco 2022-09-26 16:07:25 +02:00
Dominique Eiflaender 3c3666cd06 Pull request #476: RED-5139: Bugfix if false positives are added, for rule added entries
Merge in RED/redaction-service from RED-5139-d to master

* commit '5aba8fe88151ba6576e3ad17fd9f2ba78fc889f6':
  RED-5139: Bugfix if false positives are added, for rule added entries
2022-09-26 14:49:09 +02:00
deiflaender 5aba8fe881 RED-5139: Bugfix if false positives are added, for rule added entries 2022-09-26 14:39:09 +02:00
Dominique Eiflaender ec60d0eab4 Pull request #475: RED-5151: Do not retry messages on oom errors
Merge in RED/redaction-service from RED-5151 to master

* commit '0e71613ffcce147b2ae659e47ad6ab1c6405f0e7':
  RED-5151: Do not retry messages on oom errors
2022-09-26 14:01:22 +02:00
deiflaender 0e71613ffc RED-5151: Do not retry messages on oom errors 2022-09-26 12:52:39 +02:00
Dominique Eiflaender 142f5256ae Pull request #473: RSS-114: Do not skipOverride on same types
Merge in RED/redaction-service from RSS-114 to master

* commit 'ceee64ce59d4434f953ea220f9a733877ecde08d':
  RSS-114: Do not skipOverride on same types
2022-09-16 13:56:36 +02:00
deiflaender ceee64ce59 RSS-114: Do not skipOverride on same types 2022-09-16 10:57:54 +02:00
Dominique Eiflaender d2f2cd975c Pull request #472: RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes
Merge in RED/redaction-service from RSS-105 to master

* commit '5503dbfffc10ef9f51da7c37cb66e933d747a874':
  RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes
2022-09-15 13:08:42 +02:00
deiflaender 5503dbfffc RSS-105: Use string after match of first regEx for second regex in redactBetweenRegexes 2022-09-15 13:03:31 +02:00
Dominique Eiflaender 369710dbd4 Pull request #471: RSS-109: Added rule redactBetweenRegexes
Merge in RED/redaction-service from RSS-109 to master

* commit 'b9b3e8c8f5f11fbce6d0ae85f8e079e6285b70e0':
  RSS-109: Added rule redactBetweenRegexes
2022-09-15 11:47:22 +02:00
deiflaender b9b3e8c8f5 RSS-109: Added rule redactBetweenRegexes 2022-09-15 11:41:50 +02:00
Dominique Eiflaender c478935b86 Pull request #470: RSS-107: Added funktion to redact Headline
Merge in RED/redaction-service from RSS-107 to master

* commit 'd92757cda462d8af9ac852b88c867032c753a3b0':
  RSS-107: Added funktion to redact Headline
2022-09-13 12:12:35 +02:00
deiflaender d92757cda4 RSS-107: Added funktion to redact Headline 2022-09-13 12:09:09 +02:00
Dominique Eiflaender 5828e19422 Pull request #469: RSS-105: Added parameters to include/exclude start/stop in redactBetween rule
Merge in RED/redaction-service from RSS-105 to master

* commit 'b7bf84c323b1e730481dad16428db01227d3347e':
  RSS-105: Fixed checkstyle error
  RSS-105: Added parameters to include/exclude start/stop in redactBetween rule
2022-09-13 11:49:42 +02:00
deiflaender b7bf84c323 RSS-105: Fixed checkstyle error 2022-09-13 11:44:28 +02:00
deiflaender fe7e8a83df RSS-105: Added parameters to include/exclude start/stop in redactBetween rule 2022-09-13 11:35:36 +02:00
Dominique Eiflaender a71a7818c2 Pull request #468: RSS-106: Fixed not working redactions over multi pages
Merge in RED/redaction-service from RSS-106 to master

* commit 'ee5da81299425452171370ab609a0172858c4787':
  RSS-106: Fixed not working redactions over multi pages
2022-09-13 08:57:27 +02:00
deiflaender ee5da81299 RSS-106: Fixed not working redactions over multi pages 2022-09-12 17:40:16 +02:00
Dominique Eiflaender e486711755 Pull request #467: RSS-92: Fixed excludeHeadlines on tables
Merge in RED/redaction-service from RSS-92 to master

* commit '58c3d15b78e6ddf310b7288b4b703477c0a1218d':
  RSS-92: Fixed excludeHeadlines on tables
2022-09-12 13:45:19 +02:00
deiflaender 58c3d15b78 RSS-92: Fixed excludeHeadlines on tables 2022-09-12 12:58:15 +02:00
Christoph Schabert ce15faf072 Pull request #466: hotfix: remove push from nightly build
Merge in RED/redaction-service from fixNightly to master

* commit '430a08e611cb6ac99a31139f1a46bf36a66bb6ab':
  hotfix: remove push from nightly build
2022-09-12 11:29:26 +02:00
cschabert 430a08e611 hotfix: remove push from nightly build 2022-09-12 11:26:14 +02:00
Dominique Eiflaender 4cc40d8381 Pull request #465: RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions
Merge in RED/redaction-service from RSS-31 to master

* commit '317a8a9af9d9755befe66d82eef5f7b21fe49b33':
  RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions
2022-09-09 14:07:50 +02:00
deiflaender 317a8a9af9 RSS-31: Allow to skip removeEntitiesContainedInLarger in redactBetween rule and added param to sort result by positions 2022-09-09 13:25:53 +02:00
Dominique Eiflaender 28e437a037 Pull request #464: RSS-86: Added new rule function redactLineAfterAcrossColumns
Merge in RED/redaction-service from RSS-86 to master

* commit 'a5f27cfa4cf81bfe1cb29016d18d3fc40e77624d':
  RSS-86: Added new rule function redactLineAfterAcrossColumns
2022-09-08 13:47:45 +02:00
deiflaender a5f27cfa4c RSS-86: Added new rule function redactLineAfterAcrossColumns 2022-09-08 13:05:13 +02:00
Philipp Schramm 34be42cd45 Pull request #463: RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag
Merge in RED/redaction-service from RSS-2 to master

* commit 'f1b5d605ccbe4984943e7452dd701004f64f7525':
  RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag
2022-09-08 12:17:20 +02:00
Philipp Schramm f1b5d605cc RSS-2: Section: added method redactSectionText and extended redactBetween(Not) with excludeHeadline flag 2022-09-08 12:10:07 +02:00
Dominique Eiflaender c3dadd6906 Pull request #462: RSS-42: Made redactBetween rule more generic
Merge in RED/redaction-service from RSS-42 to master

* commit 'c29c5eef0d05a86a18fca8978a42ffff3c42ea16':
  RSS-42: Made redactBetween rule more generic
2022-09-07 14:37:26 +02:00
deiflaender c29c5eef0d RSS-42: Made redactBetween rule more generic 2022-09-07 14:31:08 +02:00
Philipp Schramm 9867cd6848 Pull request #461: RED-5002: Fixed return http error code for testing rules
Merge in RED/redaction-service from red-5002 to master

* commit '64bd25a9000286c2c30400f1da02cc2ff54ea946':
  RED-5002: Fixed return http error code for testing rules
2022-08-29 09:43:46 +02:00
Philipp Schramm 64bd25a900 RED-5002: Fixed return http error code for testing rules 2022-08-29 09:38:13 +02:00
Ali Oezyetimoglu 9b9b0ab271 Pull request #459: RED-4178: Remove special behavior for tables with 2 columns
Merge in RED/redaction-service from RED-4178-rs1 to master

* commit '29451db72d7cf0b22acb27fd86d3198dab2534ca':
  RED-4178: Remove special behavior for tables with 2 columns
2022-08-26 16:50:26 +02:00
Ali Oezyetimoglu da21d7da4b Pull request #460: RED-5032: Upgrade services to new base image
Merge in RED/redaction-service from RED-5032-rs1 to master

* commit 'b6471f904c14e02a7f42cab0aa1dc07bb1aa503a':
  RED-5032: Upgrade services to new base image
2022-08-26 16:20:58 +02:00
Ali Oezyetimoglu b6471f904c RED-5032: Upgrade services to new base image 2022-08-25 17:01:51 +02:00
Ali Oezyetimoglu 29451db72d RED-4178: Remove special behavior for tables with 2 columns 2022-08-25 11:13:14 +02:00
Dominique Eiflaender 87b76cdae0 Pull request #457: RED-5022: Add figure detection values the same way as normal images
Merge in RED/redaction-service from RED-5022 to master

* commit '85cad66ade28fd482beee57b4b01951ec2b93dfd':
  RED-5022: Add figure detection values the same way as normal images
2022-08-19 14:38:27 +02:00
deiflaender 85cad66ade RED-5022: Add figure detection values the same way as normal images 2022-08-19 13:50:57 +02:00
Viktor Seifert 1cc93d3a57 Pull request #456: RED-4824: Added annotations to correctly handle json deserialization
Merge in RED/redaction-service from RED-5010 to master

* commit 'a1ef711bc32c365b2f276df5de7f3578f82f567f':
  RED-4824: Corrected annotation for json deserialization so that is unambiguously detected as a delegating creator method
  RED-4824: Added annotations to correctly handle json deserialization
2022-08-18 09:32:27 +02:00
Viktor Seifert a1ef711bc3 RED-4824: Corrected annotation for json deserialization so that is unambiguously detected as a delegating creator method 2022-08-17 19:12:28 +02:00
Viktor Seifert b6a95244d8 RED-4824: Added annotations to correctly handle json deserialization 2022-08-17 18:46:13 +02:00
Ali Oezyetimoglu 4c06ab958c Pull request #455: RED-4890
Merge in RED/redaction-service from RED-4890 to master

* commit 'f5fb8f7c074b3e7b38c6cfdd28ba4c8cc1edfeff':
  RED-4890: Refactored redaction log tests to show failed files
  RED-4890: Refactored RulesTests, will fail individually
2022-08-16 11:16:09 +02:00
Ali Oezyetimoglu f5fb8f7c07 RED-4890: Refactored redaction log tests to show failed files 2022-08-16 10:56:00 +02:00
Ali Oezyetimoglu e962992b79 Pull request #454: RED-4891: Rectangle redactions not listed in DELTA view
Merge in RED/redaction-service from RED-4891-rs1 to master

* commit 'a9ae01ab32f08d0dadb1b12b0cc4db8070071b42':
  RED-4891: Rectangle redactions not listed in DELTA view
2022-08-16 09:32:29 +02:00
Ali Oezyetimoglu a9ae01ab32 RED-4891: Rectangle redactions not listed in DELTA view 2022-08-15 13:49:03 +02:00
Dominique Eiflaender e77180f6ac Pull request #453: RED-3974: Regenerated redactionLog files
Merge in RED/redaction-service from RED-3974 to master

* commit '269716125c19f8a4029a39566332d20d5cc53a91':
  RED-3974: Regenerated redactionLog files
2022-08-12 12:29:02 +02:00
deiflaender 269716125c RED-3974: Regenerated redactionLog files 2022-08-12 12:09:46 +02:00
Dominique Eiflaender dc21e14921 Pull request #452: RED-3974: Do not add header to header row
Merge in RED/redaction-service from RED-3974 to master

* commit '6bc5a6a1351f39bd1815b4fa288faf7bce82b1eb':
  RED-3974: Do not add header to header row
2022-08-12 11:38:58 +02:00
deiflaender 6bc5a6a135 RED-3974: Do not add header to header row 2022-08-12 11:29:12 +02:00
Dominique Eiflaender 4f66f5acf7 Pull request #451: RED-3974: Fixed special case table header with diffent size of columns in rows
Merge in RED/redaction-service from RED-3974 to master

* commit 'a9ff46a3fd3363b47799e8dfe5dc99575c36a33b':
  RED-3974: Fixed special case table header with diffent size of columns in rows
2022-08-11 17:12:07 +02:00
deiflaender a9ff46a3fd RED-3974: Fixed special case table header with diffent size of columns in rows 2022-08-11 17:09:25 +02:00
Dominique Eiflaender 07aaa9722a Pull request #450: RED-3974: Use first row as header if header detection does not find a header
Merge in RED/redaction-service from RED-3974 to master

* commit 'cba81ce061df867936314dcbdbc565248c8db006':
  RED-3974: Refactored processTablePerRow
  RED-3974: Use first row as header if header detection does not find a header
2022-08-11 14:28:33 +02:00
deiflaender cba81ce061 RED-3974: Refactored processTablePerRow 2022-08-11 14:18:36 +02:00
deiflaender f84a366328 RED-3974: Use first row as header if header detection does not find a header 2022-08-11 11:54:25 +02:00
Viktor Seifert 85c44374d9 Pull request #449: RED-4824
Merge in RED/redaction-service from RED-4824 to master

* commit '1d39c150c7e9084ae6a3833149ad66a4f31da774':
  RED-4824: Corrected code style by removing parameter assignment
  RED-4824: Replaced type of the text dir with an enum that only accepts values that can be handled by our code
  RED-4824: Recreated log files to include changes in rectangle calculation
  RED-4824: Removed obsolete logging statement
  RED-4824: Finalized affine transformation by correcting the center of the rotation according to the text direction
  RED-4824: WIP: Transformed rectangle calculation from a if-else chain to an affine transformation
  RED-4824: Added test with a simple pdf that contains all combinations for rotation and text direction
  RED-4824: Cleaned up code to make method more readable
  RED-4824: Created base test for text-position rectangle creation
2022-08-11 11:09:51 +02:00
Viktor Seifert 1d39c150c7 RED-4824: Corrected code style by removing parameter assignment 2022-08-10 15:49:27 +02:00
Viktor Seifert b2f1201d92 RED-4824: Replaced type of the text dir with an enum that only accepts values that can be handled by our code 2022-08-10 15:40:20 +02:00
Viktor Seifert b592fee500 RED-4824: Recreated log files to include changes in rectangle calculation 2022-08-09 18:58:44 +02:00
Viktor Seifert 4f36b8b43e RED-4824: Removed obsolete logging statement 2022-08-09 17:12:51 +02:00
Viktor Seifert c7a789ada6 Merge branch 'master' into RED-4824 2022-08-09 16:08:19 +02:00
Viktor Seifert 6de3a3b043 RED-4824: Finalized affine transformation by correcting the center of the rotation according to the text direction 2022-08-09 15:12:10 +02:00
Viktor Seifert 43ff331a42 RED-4824: WIP: Transformed rectangle calculation from a if-else chain to an affine transformation 2022-08-08 19:31:55 +02:00
Timo Bejan 3013dc95f6 Pull request #448: RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive
Merge in RED/redaction-service from RED-4835 to master

* commit '6e46cad2c5466136c7191042e6834eb074fc5911':
  RED-4835 - cleanup
  RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive
2022-08-08 16:39:49 +02:00
Viktor Seifert 435f75996f RED-4824: Added test with a simple pdf that contains all combinations for rotation and text direction 2022-08-08 15:16:38 +02:00
Viktor Seifert ef04b7168a RED-4824: Cleaned up code to make method more readable 2022-08-08 14:05:25 +02:00
Timo Bejan 6e46cad2c5 RED-4835 - cleanup 2022-08-08 13:31:22 +03:00
Timo Bejan c41e230f85 RED-4835 - entity position not calculated correctly for duplicates where 1 is marked as false positive 2022-08-08 13:30:18 +03:00
Viktor Seifert cadbc3c7a4 RED-4824: Created base test for text-position rectangle creation 2022-08-05 16:56:26 +02:00
Philipp Schramm f4655c4050 RED-4890: Refactored RulesTests, will fail individually 2022-08-04 13:18:05 +02:00
Ali Oezyetimoglu 16ea8364df Pull request #447: RED-4867: Sonar issues: RedactionService
Merge in RED/redaction-service from RED-4867-rs1 to master

* commit '3ec1e20e417e65907448930cf7f8eb9713874fb9':
  RED-4867: Sonar issues: RedactionService
2022-08-01 16:00:29 +02:00
Ali Oezyetimoglu 3ec1e20e41 RED-4867: Sonar issues: RedactionService 2022-08-01 15:55:22 +02:00
Dominique Eiflaender ce1f8e9117 Pull request #434: RED-4753: Reduced size of TEXT File
Merge in RED/redaction-service from RED-4753 to master

* commit '55bebe8541cfded8a456b63dbd8d8b7e79bab33f':
  RED-4753: Reduced size of TEXT File a bit more
  RED-4753: Reduced size of TEXT File
2022-08-01 11:04:25 +02:00
Kresnadi Budisantoso b7e97890be Pull request #441: RED-4829 Fix wrong annotation
Merge in RED/redaction-service from kbudisantoso/Sectionjava-1659012500262 to master

* commit '47de526b4bd451d302317647cb05ac685e069622':
  RED-4829 Fix wrong annotation
2022-08-01 09:07:11 +02:00
Dominique Eiflaender ec49945682 Pull request #446: RED-4843: Disable jsondsl per default
Merge in RED/redaction-service from RED-4843 to master

* commit 'b64222938c64eeb006d62807a705918ffd2c585b':
  RED-4843: Disable jsondsl per default
2022-07-29 12:47:38 +02:00
deiflaender b64222938c RED-4843: Disable jsondsl per default 2022-07-29 12:43:35 +02:00
Timo Bejan e4823e20d4 Pull request #444: RED-4799 - remove FP entities from redactionlog
Merge in RED/redaction-service from RED-4799-3 to master

* commit '53428ec65296b9fac2c206e7bd89916c933fb5c8':
  RED-4799 - remove FP entities from redactionlog
2022-07-29 10:53:51 +02:00
Timo Bejan 53428ec652 RED-4799 - remove FP entities from redactionlog 2022-07-29 11:50:38 +03:00
Dominique Eiflaender ca270fb7de Pull request #443: RED-4843: Upgraded storage commons
Merge in RED/redaction-service from RED-4843 to master

* commit '06fe0dbd97cdf2011be17df07f8c409a221530a4':
  RED-4843: Upgraded storage commons
2022-07-29 10:41:24 +02:00
deiflaender 06fe0dbd97 RED-4843: Upgraded storage commons 2022-07-29 10:27:18 +02:00
Kresnadi Budisantoso 47de526b4b RED-4829 Fix wrong annotation 2022-07-28 14:48:39 +02:00
Timo Bejan 1d8e86e4f6 Pull request #439: RED-4799
Merge in RED/redaction-service from RED-4799-fix to master

* commit '2b7972315f655bd296a03cae8cc5e4846d47bbce':
  RED-4799
2022-07-27 11:58:11 +02:00
Timo Bejan 2b7972315f RED-4799 2022-07-27 12:54:51 +03:00
Philipp Schramm 0af853dbfc Pull request #436: RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo
Merge in RED/redaction-service from RED-4510-Rework to master

* commit '0544a42117a742de70325cd1a84e57d6829aaa25':
  RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo
2022-07-26 17:08:59 +02:00
Philipp Schramm 0544a42117 RED-4510: Added generated RedactionLog Json files and optimizations for RulesTest.java for Bamboo 2022-07-26 17:06:04 +02:00
deiflaender 55bebe8541 RED-4753: Reduced size of TEXT File a bit more 2022-07-26 11:55:25 +02:00
Timo Bejan 8bdfd6747d Pull request #435: RED-4686
Merge in RED/redaction-service from tbejan/pomxml-1658825710723 to master

* commit '80c0c25c069995790d6bd043657f50dc8de0a451':
  RED-4686
2022-07-26 11:01:22 +02:00
Timo Bejan 80c0c25c06 RED-4686 2022-07-26 10:55:17 +02:00
deiflaender 1cc66a3092 RED-4753: Reduced size of TEXT File 2022-07-26 09:45:46 +02:00
Timo Bejan 17baf9c0eb Pull request #433: RED-4686 - updated cyclic deps
Merge in RED/redaction-service from RED-4686 to master

* commit '090a6196000dfcd7a447c54224fba8eda4605380':
  RED-4686 - updated cyclic deps
2022-07-26 09:25:58 +02:00
Timo Bejan 090a619600 RED-4686 - updated cyclic deps 2022-07-26 10:22:50 +03:00
574 changed files with 3939232 additions and 16932 deletions
+47 -1
View File
@@ -9,6 +9,49 @@
**/tmp/
**/.apt_generated/
HELP.md
target/
!.mvn/wrapper/maven-wrapper.jar
!**/src/main/**/target/
!**/src/test/**/target/
### maven build ###
*.class
/out/
/build/
/target/
**/out/
**/build/
**/target/
### STS ###
.apt_generated
.classpath
.factorypath
.project
.settings
.springBeans
.sts4-cache
.gradle
### IntelliJ IDEA ###
.idea
*.iws
*.iml
*.ipr
### NetBeans ###
/nbproject/private/
/nbbuild/
/dist/
/nbdist/
/.nb-gradle/
build/
!**/src/main/**/build/
!**/src/test/**/build/
### VS Code ###
.vscode/
.factorypath
.springBeans
@@ -26,4 +69,7 @@
**/.DS_Store
**/classpath-data.json
**/dependencies-and-licenses-overview.txt
/redaction-service-v1/redaction-service-server-v1/src/test/resources/RedactionLog/
gradle.properties
gradlew
gradlew.bat
gradle/
+6
View File
@@ -0,0 +1,6 @@
variables:
SONAR_PROJECT_KEY: 'RED_redaction-service'
include:
- project: 'gitlab/gitlab'
ref: 'main'
file: 'ci-templates/gradle_java.yml'
+1
View File
@@ -1,4 +1,5 @@
# Changelog
All notable changes to this project will be documented in this file.
## [Unreleased]
-37
View File
@@ -1,37 +0,0 @@
<project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/maven-v4_0_0.xsd">
<modelVersion>4.0.0</modelVersion>
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>8.1.3</version>
<relativePath/>
</parent>
<artifactId>bamboo-specs</artifactId>
<version>1.0.0-SNAPSHOT</version>
<packaging>jar</packaging>
<dependencies>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-api</artifactId>
</dependency>
<dependency>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs</artifactId>
</dependency>
<!-- Test dependencies -->
<dependency>
<groupId>junit</groupId>
<artifactId>junit</artifactId>
<scope>test</scope>
</dependency>
</dependencies>
<!-- run 'mvn test' to perform offline validation of the plan -->
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
</project>
@@ -1,157 +0,0 @@
package buildjob;
import static com.atlassian.bamboo.specs.builders.task.TestParserTask.createJUnitParserTask;
import java.time.LocalTime;
import com.atlassian.bamboo.specs.api.BambooSpec;
import com.atlassian.bamboo.specs.api.builders.BambooKey;
import com.atlassian.bamboo.specs.api.builders.Variable;
import com.atlassian.bamboo.specs.api.builders.docker.DockerConfiguration;
import com.atlassian.bamboo.specs.api.builders.permission.PermissionType;
import com.atlassian.bamboo.specs.api.builders.permission.Permissions;
import com.atlassian.bamboo.specs.api.builders.permission.PlanPermissions;
import com.atlassian.bamboo.specs.api.builders.plan.Job;
import com.atlassian.bamboo.specs.api.builders.plan.Plan;
import com.atlassian.bamboo.specs.api.builders.plan.PlanIdentifier;
import com.atlassian.bamboo.specs.api.builders.plan.Stage;
import com.atlassian.bamboo.specs.api.builders.plan.branches.BranchCleanup;
import com.atlassian.bamboo.specs.api.builders.plan.branches.PlanBranchManagement;
import com.atlassian.bamboo.specs.api.builders.project.Project;
import com.atlassian.bamboo.specs.builders.task.CheckoutItem;
import com.atlassian.bamboo.specs.builders.task.CleanWorkingDirectoryTask;
import com.atlassian.bamboo.specs.builders.task.InjectVariablesTask;
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
import com.atlassian.bamboo.specs.builders.task.VcsCheckoutTask;
import com.atlassian.bamboo.specs.builders.task.VcsTagTask;
import com.atlassian.bamboo.specs.builders.trigger.BitbucketServerTrigger;
import com.atlassian.bamboo.specs.builders.trigger.ScheduledTrigger;
import com.atlassian.bamboo.specs.model.task.InjectVariablesScope;
import com.atlassian.bamboo.specs.model.task.ScriptTaskProperties.Location;
import com.atlassian.bamboo.specs.util.BambooServer;
/**
* Plan configuration for Bamboo.
* Learn more on: <a href="https://confluence.atlassian.com/display/BAMBOO/Bamboo+Specs">https://confluence.atlassian.com/display/BAMBOO/Bamboo+Specs</a>
*/
@BambooSpec
public class PlanSpec {
private static final String SERVICE_NAME = "redaction-service";
private static final String JVM_ARGS = " -Xmx4g -XX:+ExitOnOutOfMemoryError -XX:SurvivorRatio=2 -XX:NewRatio=1 -XX:InitialTenuringThreshold=16 -XX:MaxTenuringThreshold=16 -XX:InitiatingHeapOccupancyPercent=35 ";
private static final String SERVICE_KEY = SERVICE_NAME.toUpperCase().replaceAll("-", "");
/**
* Run main to publish plan on Bamboo
*/
public static void main(final String[] args) {
//By default credentials are read from the '.credentials' file.
BambooServer bambooServer = new BambooServer("http://localhost:8085");
Plan plan = new PlanSpec().createPlan();
bambooServer.publish(plan);
PlanPermissions planPermission = new PlanSpec().createPlanPermission(plan.getIdentifier());
bambooServer.publish(planPermission);
Plan nightPlan = new PlanSpec().createNightPlan();
bambooServer.publish(nightPlan);
PlanPermissions nightPlanPermission = new PlanSpec().createPlanPermission(nightPlan.getIdentifier());
bambooServer.publish(nightPlanPermission);
Plan secPlan = new PlanSpec().createSecBuild();
bambooServer.publish(secPlan);
PlanPermissions secPlanPermission = new PlanSpec().createPlanPermission(secPlan.getIdentifier());
bambooServer.publish(secPlanPermission);
}
private PlanPermissions createPlanPermission(PlanIdentifier planIdentifier) {
Permissions permission = new Permissions().userPermissions("atlbamboo", PermissionType.EDIT, PermissionType.VIEW, PermissionType.ADMIN, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("development", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("devplant", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.loggedInUserPermissions(PermissionType.VIEW)
.anonymousUserPermissionView();
return new PlanPermissions(planIdentifier.getProjectKey(), planIdentifier.getPlanKey()).permissions(permission);
}
private Project project() {
return new Project().name("RED").key(new BambooKey("RED"));
}
public Plan createPlan() {
return new Plan(project(), SERVICE_NAME, new BambooKey(SERVICE_KEY)).description("Plan created from (enter repository url of your plan)")
.variables(new Variable("maven_add_param", ""))
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new ScriptTask().description("Clean")
.inlineBody("#!/bin/bash\n" + "set -e\n" + "rm -rf ./*"), new VcsCheckoutTask().description("Checkout Default Repository")
.cleanCheckout(true)
.checkoutItems(new CheckoutItem().defaultRepository()), new ScriptTask().description("Build")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
.argument(SERVICE_NAME), createJUnitParserTask().description("Resultparser")
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
.enabled(true), new InjectVariablesTask().description("Inject git Tag")
.path("git.tag")
.namespace("g")
.scope(InjectVariablesScope.LOCAL), new VcsTagTask().description("${bamboo.g.gitTag}").tagName("${bamboo.g.gitTag}").defaultRepository())
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(new BitbucketServerTrigger())
.planBranchManagement(new PlanBranchManagement().createForVcsBranch()
.delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14))
.notificationForCommitters());
}
public Plan createNightPlan() {
return new Plan(project(), SERVICE_NAME + "-Night", new BambooKey(SERVICE_KEY + "NIGHT")).description("Long running nightly Plan for tests")
.variables(new Variable("maven_add_param", "-Dtest-groups=rules-test"))
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new CleanWorkingDirectoryTask().description("Clean working directory.")
.enabled(true), new VcsCheckoutTask().description("Checkout Default Repository")
.cleanCheckout(true)
.checkoutItems(new CheckoutItem().defaultRepository()), new ScriptTask().description("Build")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
.argument(SERVICE_NAME), createJUnitParserTask().description("Resultparser")
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
.enabled(true))
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
.planBranchManagement(new PlanBranchManagement().delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14)).notificationForCommitters());
}
public Plan createSecBuild() {
return new Plan(project(), SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC")).description("Security Analysis Plan")
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new ScriptTask().description("Clean")
.inlineBody("#!/bin/bash\n" + "set -e\n" + "rm -rf ./*"), new VcsCheckoutTask().description("Checkout Default Repository")
.cleanCheckout(true)
.checkoutItems(new CheckoutItem().defaultRepository()), new ScriptTask().description("Sonar")
.location(Location.FILE)
.fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh")
.argument(SERVICE_NAME))
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
.dockerRunArguments("--net=host")
.volume("/etc/maven/settings.xml", "/usr/share/maven/conf/settings.xml")
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
.linkedRepositories("RED / " + SERVICE_NAME)
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
.planBranchManagement(new PlanBranchManagement().createForVcsBranchMatching("release.*").notificationForCommitters());
}
}
@@ -1,64 +0,0 @@
#!/bin/bash
set -e
SERVICE_NAME=$1
if [[ "$bamboo_planRepository_branchName" == "master" ]]
then
branchVersion=$(cat pom.xml | grep -Eo " <version>.*-SNAPSHOT</version>" | sed -s 's|<version>\(.*\)\..*\(-*.*\)</version>|\1|' | tr -d ' ')
latestVersion=$( semver $(git tag -l "${branchVersion}.*" ) | tail -n1 )
newVersion="$(semver $latestVersion -p -i minor)"
echo "new release on master with version $newVersion"
elif [[ "$bamboo_planRepository_branchName" == release* ]]
then
branchVersion=$(echo $bamboo_planRepository_branchName | sed -s 's|release\/\([0-9]\+\.[0-9]\+\)\.x|\1|')
latestVersion=$( semver $(git tag -l "${branchVersion}.*" ) | tail -n1 )
newVersion="$(semver $latestVersion -p -i patch)"
echo "new release on $bamboo_planRepository_branchName with version $newVersion"
elif [[ "${bamboo_version_tag}" != "dev" ]]
then
newVersion="${bamboo_version_tag}"
echo "new special version bild with $newVersion"
else
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
--no-transfer-progress \
${bamboo_maven_add_param} \
clean install \
-Djava.security.egd=file:/dev/./urandomelse
echo "dev build with tag ${bamboo_planRepository_1_branch}_${bamboo_buildNumber}"
echo "gitTag=${bamboo_planRepository_1_branch}_${bamboo_buildNumber}" > git.tag
exit 0
fi
echo "gitTag=${newVersion}" > git.tag
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
${bamboo_maven_add_param} \
versions:set \
-DnewVersion=${newVersion}
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
${bamboo_maven_add_param} \
versions:set \
-DnewVersion=${newVersion}
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
--no-transfer-progress \
clean deploy \
${bamboo_maven_add_param} \
-e \
-DdeployAtEnd=true \
-Dmaven.wagon.http.ssl.insecure=true \
-Dmaven.wagon.http.ssl.allowall=true \
-Dmaven.wagon.http.ssl.ignore.validity.dates=true \
-DaltDeploymentRepository=iqser_release::default::https://nexus.iqser.com/repository/red-platform-releases
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
package
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
docker:push
@@ -1,44 +0,0 @@
#!/bin/bash
set -e
SERVICE_NAME=$1
echo "build jar binaries"
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
--no-transfer-progress \
clean install \
-Djava.security.egd=file:/dev/./urandomelse
echo "dependency-check:aggregate"
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
org.owasp:dependency-check-maven:aggregate
if [[ -z "${bamboo_repository_pr_key}" ]]
then
echo "Sonar Scan for branch: ${bamboo_planRepository_1_branch}"
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
sonar:sonar \
-Dsonar.projectKey=RED_$SERVICE_NAME \
-Dsonar.host.url=https://sonarqube.iqser.com \
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
-Dsonar.branch.name=${bamboo_planRepository_1_branch} \
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
else
echo "Sonar Scan for PR with key1: ${bamboo_repository_pr_key}"
mvn --no-transfer-progress \
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
sonar:sonar \
-Dsonar.projectKey=RED_$SERVICE_NAME \
-Dsonar.host.url=https://sonarqube.iqser.com \
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
-Dsonar.pullrequest.key=${bamboo_repository_pr_key} \
-Dsonar.pullrequest.branch=${bamboo_repository_pr_sourceBranch} \
-Dsonar.pullrequest.base=${bamboo_repository_pr_targetBranch} \
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
fi
@@ -1,23 +0,0 @@
package buildjob;
import org.junit.Test;
import com.atlassian.bamboo.specs.api.builders.plan.Plan;
import com.atlassian.bamboo.specs.api.exceptions.PropertiesValidationException;
import com.atlassian.bamboo.specs.api.util.EntityPropertiesBuilders;
public class PlanSpecTest {
@Test
public void checkYourPlanOffline() throws PropertiesValidationException {
Plan plan = new PlanSpec().createPlan();
EntityPropertiesBuilders.build(plan);
Plan nightPlan = new PlanSpec().createNightPlan();
EntityPropertiesBuilders.build(nightPlan);
Plan secPlan = new PlanSpec().createSecBuild();
EntityPropertiesBuilders.build(secPlan);
}
}
+7
View File
@@ -0,0 +1,7 @@
plugins {
`kotlin-dsl`
}
repositories {
gradlePluginPortal()
}
@@ -0,0 +1,75 @@
plugins {
`java-library`
`maven-publish`
pmd
checkstyle
jacoco
}
group = "com.iqser.red"
java.sourceCompatibility = JavaVersion.VERSION_17
java.targetCompatibility = JavaVersion.VERSION_17
tasks.pmdMain {
pmd.ruleSetFiles = files("${rootDir}/config/pmd/pmd.xml")
}
tasks.pmdTest {
pmd.ruleSetFiles = files("${rootDir}/config/pmd/test_pmd.xml")
}
tasks.named<Test>("test") {
useJUnitPlatform()
reports {
junitXml.outputLocation.set(layout.buildDirectory.dir("reports/junit"))
}
}
tasks.test {
finalizedBy(tasks.jacocoTestReport) // report is always generated after tests run
}
tasks.jacocoTestReport {
dependsOn(tasks.test) // tests are required to run before generating the report
reports {
xml.required.set(true)
csv.required.set(false)
html.outputLocation.set(layout.buildDirectory.dir("jacocoHtml"))
}
}
allprojects {
publishing {
publications {
create<MavenPublication>(name) {
from(components["java"])
}
}
repositories {
maven {
url = uri("https://nexus.knecon.com/repository/red-platform-releases/")
credentials {
username = providers.gradleProperty("mavenUser").getOrNull();
password = providers.gradleProperty("mavenPassword").getOrNull();
}
}
}
}
}
java {
withJavadocJar()
}
repositories {
mavenLocal()
mavenCentral()
maven {
url = uri("https://nexus.knecon.com/repository/gindev/");
credentials {
username = providers.gradleProperty("mavenUser").getOrNull();
password = providers.gradleProperty("mavenPassword").getOrNull();
}
}
}
+39
View File
@@ -0,0 +1,39 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE module PUBLIC "-//Puppy Crawl//DTD Check Configuration 1.3//EN"
"http://www.puppycrawl.com/dtds/configuration_1_3.dtd">
<module name="Checker">
<property
name="severity"
value="error"/>
<module name="TreeWalker">
<module name="SuppressWarningsHolder"/>
<module name="MissingDeprecated"/>
<module name="MissingOverride"/>
<module name="AnnotationLocation"/>
<module name="JavadocStyle"/>
<module name="NonEmptyAtclauseDescription"/>
<module name="IllegalImport"/>
<module name="RedundantImport"/>
<module name="RedundantModifier"/>
<module name="EmptyBlock"/>
<module name="DefaultComesLast"/>
<module name="EmptyStatement"/>
<module name="EqualsHashCode"/>
<module name="ExplicitInitialization"/>
<module name="IllegalInstantiation"/>
<module name="ModifiedControlVariable"/>
<module name="MultipleVariableDeclarations"/>
<module name="PackageDeclaration"/>
<module name="ParameterAssignment"/>
<module name="SimplifyBooleanExpression"/>
<module name="SimplifyBooleanReturn"/>
<module name="StringLiteralEquality"/>
<module name="OneStatementPerLine"/>
<module name="FinalClass"/>
<module name="ArrayTypeStyle"/>
<module name="UpperEll"/>
<module name="OuterTypeFilename"/>
</module>
<module name="FileTabCharacter"/>
<module name="SuppressWarningsFilter"/>
</module>
+21
View File
@@ -0,0 +1,21 @@
<?xml version="1.0"?>
<ruleset name="Custom ruleset"
xmlns="http://pmd.sourceforge.net/ruleset/2.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://pmd.sourceforge.net/ruleset/2.0.0 http://pmd.sourceforge.net/ruleset_2_0_0.xsd">
<description>
Knecon ruleset checks the code for bad stuff
</description>
<rule ref="category/java/errorprone.xml">
<exclude name="DataflowAnomalyAnalysis"/>
<exclude name="MissingSerialVersionUID"/>
<exclude name="NullAssignment"/>
<exclude name="AvoidLiteralsInIfCondition"/>
<exclude name="AvoidDuplicateLiterals"/>
<exclude name="AvoidFieldNameMatchingMethodName"/>
<exclude name="AssignmentInOperand"/>
</rule>
</ruleset>
+24
View File
@@ -0,0 +1,24 @@
<?xml version="1.0"?>
<ruleset name="Custom ruleset"
xmlns="http://pmd.sourceforge.net/ruleset/2.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://pmd.sourceforge.net/ruleset/2.0.0 http://pmd.sourceforge.net/ruleset_2_0_0.xsd">
<description>
Knecon test ruleset checks the code for bad stuff
</description>
<rule ref="category/java/errorprone.xml">
<exclude name="DataflowAnomalyAnalysis"/>
<exclude name="MissingSerialVersionUID"/>
<exclude name="NullAssignment"/>
<exclude name="AvoidLiteralsInIfCondition"/>
<exclude name="AvoidDuplicateLiterals"/>
<exclude name="AvoidFieldNameMatchingMethodName"/>
<exclude name="AvoidFieldNameMatchingTypeName"/>
<exclude name="AssignmentInOperand"/>
<exclude name="TestClassWithoutTestCases"/>
</rule>
</ruleset>
+124
View File
@@ -0,0 +1,124 @@
/**
* Searches the provided SemanticNode for the keyword and creates an Entity for each occurrence.
* @param keyword the string to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> byString(String keyword, String type, EntityType entityType, SemanticNode node)
/**
* Same as byString, but case insensitive.
*/
public Stream<RedactionEntity> byStringIgnoreCase(String keyword, String type, EntityType entityType, SemanticNode node)
/**
* Searches the provided SemanticNode with the regexPattern and creates a new RedactionEntity with the provided group for each occurrence.
* @param regexPattern The regexPattern
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param group the regexPattern group, that should be the entity
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> byRegex(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegex, but case insensitive.
*/
public Stream<RedactionEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegex, but can handle patterns with linebreaks.
*/
public Stream<RedactionEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegexWithLineBreaks, but case insensitive.
*/
public Stream<RedactionEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Finds the provided string, and creates a new RedactionEntity from the text after until the end of the line it is found in.
* @param string The keyword to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> lineAfterString(String string, String type, EntityType entityType, SemanticNode node)
/**
* Same as lineAfterString, but with multiple keywords
*/
public Stream<RedactionEntity> lineAfterStrings(List<String> strings, String type, EntityType entityType, SemanticNode node)
/**
* Finds the provided string in a TableCell, and creates a new RedactionEntity in the same line but adjacent table cells to the right.
* @param string The keyword to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param table The TableNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> lineAfterStringAcrossColumns(String string, String type, EntityType entityType, TableNode table)
/**
* Creates a redaction entity based on the given boundary, type, entity type, and semantic node.
*
* @param boundary The boundary of the redaction entity.
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The semantic node where the boundary is.
* @return An Optional containing the new redaction entity.
*/
public Optional<RedactionEntity> byBoundary(Boundary boundary, String type, EntityType entityType, SemanticNode node)
/**
* Creates new RedactionEntities between the provided start and stop boundaries. The start and stop boundaries are excluded.
* If any boundaries of the new RedactionEntities overlap, only the shortest boundary will be used.
* @param startBoundaries List of start boundaries
* @param stopBoundaries List of stop boundaries
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The semantic node where the boundaries are.
* @return A Stream of new RedactionEntities between the start and stop boundaries
*/
public Stream<RedactionEntity> betweenBoundaries(List<Boundary> startBoundaries, List<Boundary> stopBoundaries, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenBoundaries, but it creates the start and stop boundaries by performing a text search on the provided SemanticNode.
*/
public Stream<RedactionEntity> betweenStrings(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenStrings, but case insensitive.
*/
public Stream<RedactionEntity> betweenStringsIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* These 6 functions work the same as betweenStrings, but they also include the start and/or stop strings or are case insensitive, depending on their name.
*/
public Stream<RedactionEntity> betweenStringsIncludeStart(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeEnd(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartAndEnd(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartAndEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenBoundaries, but it creates the start and stop boundaries by performing a regex search on the provided SemanticNode.
*/
public Stream<RedactionEntity> betweenRegexes(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenRegexes, but case insensitive.
*/
public Stream<RedactionEntity> betweenRegexesIgnoreCase(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node)
/**
* Creates a new RedactionEntity which has the same boundary as the provided SemanticNode.
* @param node The SemanticNode to create a new RedactionEntity from.
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @return An optional RedactionEntity. Is empty, if the provided SemanticNode is empty.
*/
public Optional<RedactionEntity> bySemanticNode(SemanticNode node, String type, EntityType entityType)
/**
* Same as bySemanticNode, but ignores the SemanticNode, if its not a Paragraph and all its child SemanticNodes, that are not Paragraphs.
*/
public Stream<RedactionEntity> bySemanticNodeParagraphsOnly(SemanticNode node, String type, EntityType entityType)
/**
* Searches the provided SemanticNode for the provided string, and creates a new RedactionEntity, from the end of the first occurrence of the string until the end of the SemanticNode.
* @param string The string to search for
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The SemanticNode to use and search in
* @return An optional RedactionEntity, is empty, if the SemanticNode is empty, or the string isn't found in the SemanticNode.
*/
public Optional<RedactionEntity> semanticNodeAfterString(String string, String type, EntityType entityType, SemanticNode node)
+18
View File
@@ -0,0 +1,18 @@
/**
* Retrieves the main body text block.
*
* @return The text block representing the main body of the document.
*/
public TextBlock getMainBodyTextBlock()
/**
* Gets all Entities located on the page
*
* @return Set of all Entities associated with this Page
*/
Set<RedactionEntity> getEntities();
/**
* Returns the Page Number
*
* @return The number of this page
*/
Integer getPageNumber();
+33
View File
@@ -0,0 +1,33 @@
/**
* Sets the Entity to applied, this is the default.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void apply(String ruleIdentifier, String reason, String legalBasis)
/**
* Same as apply, but legalBasis can be null.
*/
void force(String ruleIdentifier, String reason, String legalBasis)
/**
* Sets the Entity to not applied.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void skip(String ruleIdentifier, String reason, String legalBasis)
/**
* Sets the Entity to ignored, is preferred to remove in most cases.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void ignore(String ruleIdentifier, String reason, String legalBasis)
/**
* Removes the entity entirely, also removes it from all EntitySets and sets intersectingNodes and deepestFullyContainingNode to null.
* Should only be used in a few cases!
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void remove(String ruleIdentifier, String reason, String legalBasis)
@@ -0,0 +1,39 @@
/**
* A boundary has a start and end property accessible by start() and end() respectively. It marks the start and end offset in String offsets.
*/
Boundary boundary
/**
* The type of an Entity identifies groups of Entities. Examples for Entites we want to find are "CBI_author" for authors, "CBI_address" for addresses or "PII" for personally identifiable information.
* Other types include helper types, which interact with the main Entities, for example "published_information" or "vertebrate". These Entities are used to modify the main Entities, if they occur in the same Section.
* A typical example would be to ignore Entities of type "CBI_author", if they occur in the same Section as "published_information" entities.
*/
String type
/**
* The EntityType can be one of four different values: ENTITY, RECOMMENDATION, FALSE_POSITIVE, FALSE_RECOMMENDATION.
* If an ENTITY is overlapped by a FALSE_POSITIVE, the ENTITY is removed. If a RECOMMENDATION is overlapped by either an ENTITY or FALSE_RECOMMENDATION, it is removed.
*/
EntityType entityType
/**
* The text the Entity represents.
*/
String value
/**
* Up to three words after the Entity in the text.
*/
String textAfter
/**
* Up to three words before the Entity in the text.
*/
String textBefore
/**
* All pages whose TextBlock intersects the boundary of this entity. Is always equal to the Pages which have this RedactionEntity in their EntitySet.
*/
Set<Page> pages
/**
* All SemanticNodes whose TextBlock intersects the boundary of this entity. Is always equal to the SemanticNodes which have this RedactionEntity in their EntitySet.
*/
List<SemanticNode> intersectingNodes
/**
* The SemanticNode which is the deepest in the Tree structure and whose TextBlock fully contains the boundary of this Node.
*/
SemanticNode deepestFullyContainingNode
+6
View File
@@ -0,0 +1,6 @@
/**
* Determines whether this Section has any tables.
*
* @return {@code true} if there are tables, {@code false} otherwise
*/
public boolean hasTables()
+181
View File
@@ -0,0 +1,181 @@
/**
* Returns the type of this node, such as NodeType.SECTION, NodeType.PARAGRAPH, etc.
*
* @return NodeType of this node
*/
NodeType getType();
/**
* Any Node maintains its own Set of Entities.
* This Set contains all Entities whose boundary intersects the boundary of this node.
* The Entities might overlap with the Entities in other Sets
*
* @return Set of all Entities associated with this Node
*/
Set<RedactionEntity> getEntities();
/**
* Returns all Pages this SemanticNode is associated with.
*
* @return Set of Pages this node appears on.
*/
Set<Page> getPages()
/**
* Checks if this node appears on the specified page number.
*
* @param pageNumber The page number to check.
* @return True if this node is found on the specified page number, false otherwise.
*/
boolean isOnPage(int pageNumber)
/**
* Returns the closest Headline associated with this SemanticNode
*
* @return First Headline found.
*/
Headline getHeadline()
/**
* @return The SemanticNode representing the Parent in the DocumentTree
* throws NotFoundException, when no parent is present
*/
SemanticNode getParent()
/**
* Checks whether this SemanticNode has any Entity of the provided type.
* Ignores Entity with ignored == true or removed == true.
*
* @param type string representing the type of entity to check for
* @return true, if this SemanticNode has at least one Entity of the provided type
*/
boolean hasEntitiesOfType(String type)
/**
* Checks whether this SemanticNode has any Entity of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types an array of strings representing the types of entities to check for
* @return true, if this SemanticNode has at least one Entity of any of the provided types
*/
boolean hasEntitiesOfAnyType(String... types)
/**
* Checks whether this SemanticNode has at least one Entity of each of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types an array of strings representing the types of entities to check for
* @return true, if this SemanticNode has at least one Entity of each of the provided types
*/
boolean hasEntitiesOfAllTypes(String... types)
/**
* Returns a List of Entities in this SemanticNode which are of the provided type such as "CBI_author".
* Ignores Entity with ignored == true or removed == true.
*
* @param type string representing the type of entities to return
* @return List of RedactionEntities of any the type
*/
List<RedactionEntity> getEntitiesOfType(String type)
/**
* Returns a List of Entities in this SemanticNode which have any of the provided types such as "CBI_author".
* Ignores Entity with ignored == true or removed == true.
*
* @param types A list of strings representing the types of entities to return
* @return List of RedactionEntities of any provided type
*/
List<RedactionEntity> getEntitiesOfType(List<String> types)
/**
* Returns a List of Entities in this SemanticNode which have any of the provided types.
* Ignores Entity with the ignored flag set to true or the removed flag set to true.
*
* @param types A list of strings representing the types of entities to return
* @return List of RedactionEntities that match any of the provided types
*/
List<RedactionEntity> getEntitiesOfType(String... types)
/**
* Checks whether this SemanticNode contains the provided String.
*
* @param string A String which the TextBlock might contain
* @return true, if this node's TextBlock contains the string
*/
boolean containsString(String string)
/**
* Checks whether this SemanticNode contains all the provided Strings.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains all strings
*/
boolean containsAllStrings(String... strings)
/**
* Checks whether this SemanticNode contains any of the provided Strings.
*
* @param strings A List of Strings to check if they are contained in the TextBlock
* @return true, if this node's TextBlock contains any of the provided strings
*/
boolean containsAnyString(String... strings)
/**
* Checks whether this SemanticNode contains all the provided Strings ignoring case.
*
* @param string A String which the TextBlock might contain
* @return true, if this node's TextBlock contains the string ignoring case
*/
boolean containsStringIgnoreCase(String string)
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
boolean containsAnyStringIgnoreCase(String... strings)
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
boolean containsAllStringsIgnoreCase(String... strings)
/**
* Checks whether this SemanticNode matches the provided regex pattern.
*
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
* @return true, if this node's TextBlock contains the regex pattern
*/
boolean matchesRegex(String regexPattern)
/**
* Checks whether this SemanticNode matches the provided regex pattern ignoring case.
*
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
* @return true, if this node's TextBlock contains the regex pattern ignoring case
*/
boolean matchesRegexIgnoreCase(String regexPattern)
/**
* Streams all children located directly underneath this node in the DocumentTree.
*
* @return Stream of all children
*/
Stream<SemanticNode> streamChildren()
/**
* Streams all children located directly underneath this node in the DocumentTree of the provided type.
*
* @param nodeType the type of nodes to stream
* @return Stream of all children of the provided type
*/
Stream<SemanticNode> streamChildrenOfType(NodeType nodeType)
/**
* Recursively streams all SemanticNodes located underneath this node in the DocumentTree in order.
*
* @return Stream of all SubNodes
*/
Stream<SemanticNode> streamAllSubNodes()
/**
* Recursively streams all SemanticNodes of a specified type located underneath this node in the DocumentTree in order.
*
* @param nodeType the type of nodes to be streamed
* @return a Stream of all SubNodes of the specified type
*/
Stream<SemanticNode> streamAllSubNodesOfType(NodeType nodeType)
/**
* The Boundary is the start and end string offsets in the reading order of the document.
*
* @return Boundary of this Node's TextBlock
*/
Boundary getBoundary()
/**
* The SectionIdentifier uses the numeric identifiers of Headlines to infer a tree structure.
* It implements functions such as sectionIdentifier.isChildOf(otherSectionIdentifier) and sectionIdentifier.isParentOf(otherSectionIdentifier)
*
* @return The SectionIdentifier from the first Headline.
*/
SectionIdentifier getSectionIdentifier()
+77
View File
@@ -0,0 +1,77 @@
/**
* Streams all entities in this table, that appear in a row, which contains any of the provided strings.
*
* @param strings Strings to check whether a row contains them
* @return Stream of all entities in this table, that appear in a row, which contains any of the provided strings
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsStringsIgnoreCase(List<String> strings)
/**
* Checks whether the specified row contains all the provided strings.
*
* @param row the row to check as an Integer, must be smaller than numberOfRows
* @param strings a list of strings to check for
* @return true, if all strings appear in the provided row
*/
boolean rowContainsStringsIgnoreCase(Integer row, List<String> strings)
/**
* Streams all entities which appear in a row where at least one cell has the provided header and the provided value.
*
* @param header the header value to search for
* @param value the string which the table cell should contain
* @return a stream of all entities, which appear in a row where at least one cell has the provided header and the provided value.
*/
Stream<RedactionEntity> streamEntitiesWhereRowHasHeaderAndValue(String header, String value)
/**
* Streams all entities which appear in a row where at least one cell has the provided header and any provided value.
*
* @param header the header value to search for
* @param values the strings which the table cell should contain
* @return a stream of all entities, which appear in a row where at least one cell has the provided header and any provided value.
*/
Stream<RedactionEntity> streamEntitiesWhereRowHasHeaderAndAnyValue(String header, List<String> values)
/**
* Streams all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types type strings to check whether a row contains an entity like them
* @return Stream of all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsEntitiesOfType(List<String> types)
/**
* Streams all entities in this table, that appear in a row, which contains no entity of any of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types type strings to check whether a row contains an entity like them
* @return Stream of all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsNoEntitiesOfType(List<String> types)
/**
* Streams all TableCells in this Table which have the provided header row-wise.
*
* @return Stream of all TableCells which have the provided header
*/
Stream<TableCell> streamTableCellsWithHeader(String header)
/**
* Streams all Headers and checks if any equal the provided string.
*
* @param header string to check the headers for
* @return true, if at least one header equals the provided string
*/
boolean hasHeader(String header)
/**
* Checks if this table has a column with the provided header and any of the table cells in that column contain the provided value.
*
* @param header string to find header cells
* @param value string to check cells with provided header
* @return true, if this table has a column with the provided header and any of the table cells in that column contain the provided value
*/
boolean hasRowWithHeaderAndValue(String header, String value)
/**
* Finds all entities of the provided type, which appear in the same row that the provided entity appears in.
* Ignores Entity with ignored == true or removed == true.
*
* @param type the type of entities to search for
* @param redactionEntity the entity, which appears in the row to search
* @return List of all entities of the provided type, which appear in the same row that the provided entity appears in.
*/
List<RedactionEntity> getEntitiesOfTypeInSameRow(String type, RedactionEntity redactionEntity)
+591
View File
@@ -0,0 +1,591 @@
From now on, you are a Drools rule generator.
You have a Document data structure written in Java with the following objects:
- Section
- Table
- TableCell
- Paragraph
- Headline
- Page
- RedactionEntity
- EntityCreationService
The Section, Table, TableCell, Paragraph, and Headline implement a common interface called SemanticNode. SemanticNodes are arranged in a tree-like fashion, where any SemanticNode can have multiple SemanticNodes as children. The arrangement is as follows:
- Tables only have TableCells as children.
- TableCells may have any child, except TableCells.
- Paragraphs and Headlines have no children.
- Sections may have any child except TableCells, but if it contains Paragraphs as well as Tables, it is split into a Section with multiple Sections as children, where any child Section only contains either Tables or Paragraphs.
Further, if the first SemanticNode is a Headline it remains the first child in the Parent Section, before any subsections.
The goal of the Software is to find pieces of Text that are relevant. Each piece of text is represented by a RedactionEntity.
The main pieces of relevant text are Text we want to redact. For example, we want to redact all Authors of a dossier. Or all personally identifiable information, such as E-Mails and Telephone Numbers.
RedactionEntities may also represent other pieces of text, such as published information, or certain species of vertebrates.
The RedactionEntities are part of the document structure, such that they are referenced in each SemanticNode and Page, which contains it, and further, the RedactionEntity references each Page and SemanticNode it occurs in.
So the same RedactionEntity occurs in the paragraph it is located, as well as all its parent Sections.
Previous to the execution of rules, the Document structure is assembled and a text search is performed to create initial Entities of different types. They are then inserted into the document structure.
Then the KieSession is created and each SemanticNode and Entity is inserted into its working memory.
----------------------------------------------------------------
The relevant functions for SemanticNode:
/**
* Returns the type of this node, such as NodeType.SECTION, NodeType.PARAGRAPH, etc.
*
* @return NodeType of this node
*/
NodeType getType();
/**
* Any Node maintains its own Set of Entities.
* This Set contains all Entities whose boundary intersects the boundary of this node.
* The Entities might overlap with the Entities in other Sets
*
* @return Set of all Entities associated with this Node
*/
Set<RedactionEntity> getEntities();
/**
* Returns all Pages this SemanticNode is associated with.
*
* @return Set of Pages this node appears on.
*/
Set<Page> getPages()
/**
* Checks if this node appears on the specified page number.
*
* @param pageNumber The page number to check.
* @return True if this node is found on the specified page number, false otherwise.
*/
boolean onPage(int pageNumber)
/**
* For Sections it searches its children and returns the first Headline.
* For Paragraphs, Tables, and TableCells it returns getHeadline() of getParent()
* For Headline it returns itself and for Headers or Footers it returns an empty dummy Headline.
*
* @return First Headline found.
*/
Headline getHeadline()
/**
* @return The SemanticNode representing the Parent in the DocumentTree
* When no parent is present, the Document is returned. And for the Document itself it throws an UnsupportedOperationException.
*/
SemanticNode getParent()
/**
* Checks whether this SemanticNode has any Entity of the provided type.
* Ignores Entity with ignored == true or removed == true.
*
* @param type string representing the type of entity to check for
* @return true, if this SemanticNode has at least one Entity of the provided type
*/
boolean hasEntitiesOfType(String type)
/**
* Checks whether this SemanticNode has any Entity of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types an array of strings representing the types of entities to check for
* @return true, if this SemanticNode has at least one Entity of any of the provided types
*/
boolean hasEntitiesOfAnyType(String... types)
/**
* Checks whether this SemanticNode has at least one Entity of each of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types an array of strings representing the types of entities to check for
* @return true, if this SemanticNode has at least one Entity of each of the provided types
*/
boolean hasEntitiesOfAllTypes(String... types)
/**
* Returns a List of Entities in this SemanticNode which are of the provided type such as "CBI_author".
* Ignores Entity with ignored == true or removed == true.
*
* @param type string representing the type of entities to return
* @return List of RedactionEntities of any the type
*/
List<RedactionEntity> getEntitiesOfType(String type)
/**
* Returns a List of Entities in this SemanticNode which have any of the provided types such as "CBI_author".
* Ignores Entity with ignored == true or removed == true.
*
* @param types A list of strings representing the types of entities to return
* @return List of RedactionEntities of any provided type
*/
List<RedactionEntity> getEntitiesOfType(List<String> types)
/**
* Returns a List of Entities in this SemanticNode which have any of the provided types.
* Ignores Entity with the ignored flag set to true or the removed flag set to true.
*
* @param types A list of strings representing the types of entities to return
* @return List of RedactionEntities that match any of the provided types
*/
List<RedactionEntity> getEntitiesOfType(String... types)
/**
* Checks whether this SemanticNode contains the provided String.
*
* @param string A String which the TextBlock might contain
* @return true, if this node's TextBlock contains the string
*/
boolean containsString(String string)
/**
* Checks whether this SemanticNode contains all the provided Strings.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains all strings
*/
boolean containsAllStrings(String... strings)
/**
* Checks whether this SemanticNode contains any of the provided Strings.
*
* @param strings A List of Strings to check if they are contained in the TextBlock
* @return true, if this node's TextBlock contains any of the provided strings
*/
boolean containsAnyString(String... strings)
/**
* Checks whether this SemanticNode contains all the provided Strings ignoring case.
*
* @param string A String which the TextBlock might contain
* @return true, if this node's TextBlock contains the string ignoring case
*/
boolean containsStringIgnoreCase(String string)
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
boolean containsAnyStringIgnoreCase(String... strings)
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
boolean containsAllStringsIgnoreCase(String... strings)
/**
* Checks whether this SemanticNode matches the provided regex pattern.
*
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
* @return true, if this node's TextBlock contains the regex pattern
*/
boolean matchesRegex(String regexPattern)
/**
* Checks whether this SemanticNode matches the provided regex pattern ignoring case.
*
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
* @return true, if this node's TextBlock contains the regex pattern ignoring case
*/
boolean matchesRegexIgnoreCase(String regexPattern)
/**
* Streams all children located directly underneath this node in the DocumentTree.
*
* @return Stream of all children
*/
Stream<SemanticNode> streamChildren()
/**
* Streams all children located directly underneath this node in the DocumentTree of the provided type.
*
* @param nodeType the type of nodes to stream
* @return Stream of all children of the provided type
*/
Stream<SemanticNode> streamChildrenOfType(NodeType nodeType)
/**
* Recursively streams all SemanticNodes located underneath this node in the DocumentTree in order.
*
* @return Stream of all SubNodes
*/
Stream<SemanticNode> streamAllSubNodes()
/**
* Recursively streams all SemanticNodes of a specified type located underneath this node in the DocumentTree in order.
*
* @param nodeType the type of nodes to be streamed
* @return a Stream of all SubNodes of the specified type
*/
Stream<SemanticNode> streamAllSubNodesOfType(NodeType nodeType)
/**
* The Boundary is the start and end string offsets in the reading order of the document.
*
* @return Boundary of this Node's TextBlock
*/
Boundary getBoundary()
/**
* The SectionIdentifier uses the numeric identifiers of Headlines to infer a tree structure.
* It implements functions such as sectionIdentifier.isChildOf(otherSectionIdentifier) and sectionIdentifier.isParentOf(otherSectionIdentifier)
*
* @return The SectionIdentifier from the first Headline.
*/
SectionIdentifier getSectionIdentifier()
----------------------------------------------------------------
The Table has the additional functions:
/**
* Streams all entities in this table, that appear in a row, which contains any of the provided strings.
*
* @param strings Strings to check whether a row contains them
* @return Stream of all entities in this table, that appear in a row, which contains any of the provided strings
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsStringsIgnoreCase(List<String> strings)
/**
* Checks whether the specified row contains all the provided strings.
*
* @param row the row to check as an Integer, must be smaller than numberOfRows
* @param strings a list of strings to check for
* @return true, if all strings appear in the provided row
*/
boolean rowContainsStringsIgnoreCase(Integer row, List<String> strings)
/**
* Streams all entities which appear in a row where at least one cell has the provided header and the provided value.
*
* @param header the header value to search for
* @param value the string which the table cell should contain
* @return a stream of all entities, which appear in a row where at least one cell has the provided header and the provided value.
*/
Stream<RedactionEntity> streamEntitiesWhereRowHasHeaderAndValue(String header, String value)
/**
* Streams all entities which appear in a row where at least one cell has the provided header and any provided value.
*
* @param header the header value to search for
* @param values the strings which the table cell should contain
* @return a stream of all entities, which appear in a row where at least one cell has the provided header and any provided value.
*/
Stream<RedactionEntity> streamEntitiesWhereRowHasHeaderAndAnyValue(String header, List<String> values)
/**
* Streams all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types type strings to check whether a row contains an entity like them
* @return Stream of all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsEntitiesOfType(List<String> types)
/**
* Streams all entities in this table, that appear in a row, which contains no entity of any of the provided types.
* Ignores Entity with ignored == true or removed == true.
*
* @param types type strings to check whether a row contains an entity like them
* @return Stream of all entities in this table, that appear in a row, which contains at least one entity with any of the provided types.
*/
Stream<RedactionEntity> streamEntitiesWhereRowContainsNoEntitiesOfType(List<String> types)
/**
* Streams all TableCells in this Table which have the provided header row-wise.
*
* @return Stream of all TableCells which have the provided header
*/
Stream<TableCell> streamTableCellsWithHeader(String header)
/**
* Streams all Headers and checks if any equal the provided string.
*
* @param header string to check the headers for
* @return true, if at least one header equals the provided string
*/
boolean hasHeader(String header)
/**
* Checks if this table has a column with the provided header and any of the table cells in that column contain the provided value.
*
* @param header string to find header cells
* @param value string to check cells with provided header
* @return true, if this table has a column with the provided header and any of the table cells in that column contain the provided value
*/
boolean hasRowWithHeaderAndValue(String header, String value)
/**
* Finds all entities of the provided type, which appear in the same row that the provided entity appears in.
* Ignores Entity with ignored == true or removed == true.
*
* @param type the type of entities to search for
* @param redactionEntity the entity, which appears in the row to search
* @return List of all entities of the provided type, which appear in the same row that the provided entity appears in.
*/
List<RedactionEntity> getEntitiesOfTypeInSameRow(String type, RedactionEntity redactionEntity)
----------------------------------------------------------------
The Section has these additional Rules:
/**
* Determines whether this Section has any tables.
*
* @return {@code true} if there are tables, {@code false} otherwise
*/
boolean hasTables()
----------------------------------------------------------------
The Page Object has the following functions:
/**
* Retrieves the main body text block.
* @return The text block representing the main body of the document.
*/
public TextBlock getMainBodyTextBlock()
/**
* @return All SemanticNodes that occur on the page, except Header and Footer
*/
public List<SemanticNode> getMainBody()
/**
* Gets all Entities located on the page
* @return Set of all Entities associated with this Page
*/
Set<RedactionEntity> getEntities();
/**
* Returns the Page Number
*
* @return The number of this page
*/
Integer getPageNumber();
----------------------------------------------------------------
----------------------------------------------------------------
The RedactionEntity has the following properties:
/**
* A boundary has a start and end property accessible by start() and end() respectively. It marks the start and end offset in String offsets.
*/
Boundary boundary
/**
* The type of an Entity identifies groups of Entities. Examples for Entites we want to find are "CBI_author" for authors, "CBI_address" for addresses or "PII" for personally identifiable information.
* Other types include helper types, which interact with the main Entities, for example "published_information" or "vertebrate". These Entities are used to modify the main Entities, if they occur in the same Section.
* A typical example would be to ignore Entities of type "CBI_author", if they occur in the same Section as "published_information" entities.
*/
String type
/**
* The EntityType can be one of four different values: ENTITY, RECOMMENDATION, FALSE_POSITIVE, FALSE_RECOMMENDATION.
* If an ENTITY is overlapped by a FALSE_POSITIVE, the ENTITY is removed. If a RECOMMENDATION is overlapped by either an ENTITY or FALSE_RECOMMENDATION, it is removed.
*/
EntityType entityType
/**
* The text the Entity represents.
*/
String value
/**
* Up to three words after the Entity in the text.
*/
String textAfter
/**
* Up to three words before the Entity in the text.
*/
String textBefore
/**
* All pages whose TextBlock intersects the boundary of this entity. Is always equal to the Pages which have this RedactionEntity in their EntitySet.
*/
Set<Page> pages
/**
* All SemanticNodes whose TextBlock intersects the boundary of this entity. Is always equal to the SemanticNodes which have this RedactionEntity in their EntitySet.
*/
List<SemanticNode> intersectingNodes
/**
* The SemanticNode which is the deepest in the Tree structure and whose TextBlock fully contains the boundary of this Node.
*/
SemanticNode deepestFullyContainingNode
The RedactionEntity also has the following methods:
/**
* Sets the Entity to applied, this is the default.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void apply(String ruleIdentifier, String reason, String legalBasis)
/**
* Same as apply, but legalBasis can be null.
*/
void force(String ruleIdentifier, String reason, String legalBasis)
/**
* Sets the Entity to not applied.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void skip(String ruleIdentifier, String reason, String legalBasis)
/**
* Sets the Entity to ignored, is preferred to remove in most cases.
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void ignore(String ruleIdentifier, String reason, String legalBasis)
/**
* Removes the entity entirely, also removes it from all EntitySets and sets intersectingNodes and deepestFullyContainingNode to null.
* Should only be used in a few cases!
* @param ruleIdentifier Should always be equal to the ruleIdentifier in the Rule name
* @param reason Should describe the intention of the rule in a few words
* @param legalBasis Is dependent on the rule, if none is known the default is "n-a", can't be null
*/
void remove(String ruleIdentifier, String reason, String legalBasis)
----------------------------------------------------------------
The EntityCreationService offers the following functions:
/**
* Searches the provided SemanticNode for the keyword and creates an Entity for each occurrence.
* @param keyword the string to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> byString(String keyword, String type, EntityType entityType, SemanticNode node)
/**
* Same as byString, but case insensitive.
*/
public Stream<RedactionEntity> byStringIgnoreCase(String keyword, String type, EntityType entityType, SemanticNode node)
/**
* Searches the provided SemanticNode with the regexPattern and creates a new RedactionEntity with the provided group for each occurrence.
* @param regexPattern The regexPattern
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param group the regexPattern group, that should be the entity
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> byRegex(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegex, but case insensitive.
*/
public Stream<RedactionEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegex, but can handle patterns with linebreaks.
*/
public Stream<RedactionEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Same as byRegexWithLineBreaks, but case insensitive.
*/
public Stream<RedactionEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node)
/**
* Finds the provided string, and creates a new RedactionEntity from the text after until the end of the line it is found in.
* @param string The keyword to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param node The SemanticNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> lineAfterString(String string, String type, EntityType entityType, SemanticNode node)
/**
* Same as lineAfterString, but with multiple keywords
*/
public Stream<RedactionEntity> lineAfterStrings(List<String> strings, String type, EntityType entityType, SemanticNode node)
/**
* Finds the provided string in a TableCell, and creates a new RedactionEntity in the same line but adjacent table cells to the right.
* @param string The keyword to search for
* @param type The type of the RedactionEntity to be created
* @param entityType The EntityType of the RedactionEntity to be created
* @param table The TableNode to search in
* @return A Stream of RedactionEntities with the keyword as value, the type as type and the provided EntityType
*/
public Stream<RedactionEntity> lineAfterStringAcrossColumns(String string, String type, EntityType entityType, TableNode table)
/**
* Creates a redaction entity based on the given boundary, type, entity type, and semantic node.
*
* @param boundary The boundary of the redaction entity.
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The semantic node where the boundary is.
* @return An Optional containing the new redaction entity.
*/
public Optional<RedactionEntity> byBoundary(Boundary boundary, String type, EntityType entityType, SemanticNode node)
/**
* Creates new RedactionEntities between the provided start and stop boundaries. The start and stop boundaries are excluded.
* If any boundaries of the new RedactionEntities overlap, only the shortest boundary will be used.
* @param startBoundaries List of start boundaries
* @param stopBoundaries List of stop boundaries
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The semantic node where the boundaries are.
* @return A Stream of new RedactionEntities between the start and stop boundaries
*/
public Stream<RedactionEntity> betweenBoundaries(List<Boundary> startBoundaries, List<Boundary> stopBoundaries, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenBoundaries, but it creates the start and stop boundaries by performing a text search on the provided SemanticNode.
*/
public Stream<RedactionEntity> betweenStrings(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenStrings, but case insensitive.
*/
public Stream<RedactionEntity> betweenStringsIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* These 6 functions work the same as betweenStrings, but they also include the start and/or stop strings or are case insensitive, depending on their name.
*/
public Stream<RedactionEntity> betweenStringsIncludeStart(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeEnd(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartAndEnd(String start, String stop, String type, EntityType entityType, SemanticNode node)
public Stream<RedactionEntity> betweenStringsIncludeStartAndEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenBoundaries, but it creates the start and stop boundaries by performing a regex search on the provided SemanticNode.
*/
public Stream<RedactionEntity> betweenRegexes(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node)
/**
* Same as betweenRegexes, but case insensitive.
*/
public Stream<RedactionEntity> betweenRegexesIgnoreCase(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node)
/**
* Creates a new RedactionEntity which has the same boundary as the provided SemanticNode.
* @param node The SemanticNode to create a new RedactionEntity from.
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @return An optional RedactionEntity. Is empty, if the provided SemanticNode is empty.
*/
public Optional<RedactionEntity> bySemanticNode(SemanticNode node, String type, EntityType entityType)
/**
* Same as bySemanticNode, but ignores the SemanticNode, if its not a Paragraph and all its child SemanticNodes, that are not Paragraphs.
*/
public Stream<RedactionEntity> bySemanticNodeParagraphsOnly(SemanticNode node, String type, EntityType entityType)
/**
* Searches the provided SemanticNode for the provided string, and creates a new RedactionEntity, from the end of the first occurrence of the string until the end of the SemanticNode.
* @param string The string to search for
* @param type The type of the redaction entity.
* @param entityType The entity type of the redaction entity.
* @param node The SemanticNode to use and search in
* @return An optional RedactionEntity, is empty, if the SemanticNode is empty, or the string isn't found in the SemanticNode.
*/
public Optional<RedactionEntity> semanticNodeAfterString(String string, String type, EntityType entityType, SemanticNode node)
----------------------------------------------------------------
Rules may be grouped into two categories.
The first category changes existing RedactionEntities, and the second creates new RedactionEntities.
There are two different types of rules, one you create new Entities and in the other you change or remove existing Entities.
An Entity is any piece of text, uniquely identified in the Document by its Boundary, its Type and its EntityType. The Boundary consists of a start and stop index in the text of the document.
The Type is a String like "PII", which stands for
The goal is to find entities that fulfill certain conditions. Each SemanticNode has its own set of entities, but these sets may have intersections.
For example, a Section contains all the entities in any of its children. Additionally, if an entity overlaps two SemanticNodes, both paragraphs have this entity in their sets.
To generate Drools rules for the scenario of changing or updating Entities, consider the following information:
1. Conditions: Specify the conditions that must be met for an entity to be selected. For example:
- The entity has a specific attribute value.
- The entity is within a certain range of values.
- The entity satisfies a complex combination of conditions.
2. Actions: Define the actions to be performed when an entity fulfills the conditions. This could include:
- Adding the entity to a result set.
- Modifying the entity's attributes.
- Triggering some other behavior or logic.
3. Rule Structure: Determine the structure of the Drools rules. This typically consists of:
- Rule names: Choose meaningful names for your rules.
- Rule attributes: Set the salience (priority) of rules if necessary.
- Conditions: Define the conditions based on the requirements.
- Actions: Specify the actions to be performed when the conditions are met.
Remember to provide specific examples, use case scenarios, and any additional requirements you have for the Drools rules.
Please provide any specific conditions, actions, or examples that you would like to be incorporated into the Drools rules.
+1
View File
@@ -0,0 +1 @@
version = 4.0-SNAPSHOT
-21
View File
@@ -1,21 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<modelVersion>4.0.0</modelVersion>
<artifactId>redaction-service</artifactId>
<groupId>com.iqser.red.service</groupId>
<version>3.0-SNAPSHOT</version>
<packaging>pom</packaging>
<modules>
<module>bamboo-specs</module>
<module>redaction-service-v1</module>
<module>redaction-service-image-v1</module>
</modules>
</project>
-98
View File
@@ -1,98 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<parent>
<groupId>com.iqser.red</groupId>
<artifactId>platform-docker-dependency</artifactId>
<version>1.2.0</version>
<relativePath />
</parent>
<modelVersion>4.0.0</modelVersion>
<artifactId>redaction-service-image-v1</artifactId>
<groupId>com.iqser.red.service</groupId>
<version>1.0-SNAPSHOT</version>
<packaging>pom</packaging>
<properties>
<service.server>redaction-service-server-v1</service.server>
<platform.jar>${service.server}.jar</platform.jar>
<docker.skip.push>false</docker.skip.push>
<docker.image.name>${docker.image.prefix}/${service.server}</docker.image.name>
</properties>
<build>
<plugins>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-dependency-plugin</artifactId>
</plugin>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-resources-plugin</artifactId>
</plugin>
<plugin>
<groupId>org.codehaus.mojo</groupId>
<artifactId>exec-maven-plugin</artifactId>
</plugin>
<plugin>
<groupId>io.fabric8</groupId>
<artifactId>docker-maven-plugin</artifactId>
</plugin>
</plugins>
<pluginManagement>
<plugins>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-dependency-plugin</artifactId>
<executions>
<execution>
<id>download-platform-jar</id>
<phase>prepare-package</phase>
<goals>
<goal>copy</goal>
</goals>
<configuration>
<artifactItems>
<dependency>
<groupId>${project.groupId}</groupId>
<artifactId>${service.server}</artifactId>
<version>${project.version}</version>
<type>jar</type>
<overWrite>true</overWrite>
<destFileName>${platform.jar}</destFileName>
</dependency>
</artifactItems>
<outputDirectory>${docker.build.directory}</outputDirectory>
</configuration>
</execution>
</executions>
</plugin>
<plugin>
<groupId>io.fabric8</groupId>
<artifactId>docker-maven-plugin</artifactId>
<configuration>
<images>
<image>
<name>${docker.image.name}</name>
<build>
<dockerFileDir>${docker.build.directory}</dockerFileDir>
<args>
<PLATFORM_JAR>${platform.jar}</PLATFORM_JAR>
</args>
<tags>
<tag>${docker.image.version}</tag>
<tag>latest</tag>
</tags>
</build>
</image>
</images>
</configuration>
</plugin>
</plugins>
</pluginManagement>
</build>
</project>
@@ -1,9 +0,0 @@
FROM red/redaction-service-base-v1:1.0.0
ARG PLATFORM_JAR
ENV PLATFORM_JAR ${PLATFORM_JAR}
ENV USES_ELASTICSEARCH false
COPY ["${PLATFORM_JAR}", "/"]
-87
View File
@@ -1,87 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<parent>
<artifactId>platform-dependency</artifactId>
<groupId>com.iqser.red</groupId>
<version>1.10.0</version>
<relativePath />
</parent>
<modelVersion>4.0.0</modelVersion>
<artifactId>redaction-service-v1</artifactId>
<groupId>com.iqser.red.service</groupId>
<version>1.0-SNAPSHOT</version>
<packaging>pom</packaging>
<modules>
<module>redaction-service-api-v1</module>
<module>redaction-service-server-v1</module>
</modules>
<properties>
<pdfbox.version>2.0.24</pdfbox.version>
<dsljson.version>1.9.9</dsljson.version>
</properties>
<dependencyManagement>
<dependencies>
<dependency>
<groupId>com.iqser.red</groupId>
<artifactId>platform-commons-dependency</artifactId>
<version>1.17.0</version>
<scope>import</scope>
<type>pom</type>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
<version>${pdfbox.version}</version>
</dependency>
</dependencies>
</dependencyManagement>
<build>
<pluginManagement>
<plugins>
<plugin>
<groupId>org.sonarsource.scanner.maven</groupId>
<artifactId>sonar-maven-plugin</artifactId>
</plugin>
<plugin>
<groupId>org.owasp</groupId>
<artifactId>dependency-check-maven</artifactId>
<configuration>
<format>ALL</format>
</configuration>
</plugin>
<plugin>
<groupId>org.jacoco</groupId>
<artifactId>jacoco-maven-plugin</artifactId>
<executions>
<execution>
<id>prepare-agent</id>
<goals>
<goal>prepare-agent</goal>
</goals>
</execution>
<execution>
<id>report</id>
<goals>
<goal>report</goal>
</goals>
</execution>
</executions>
</plugin>
</plugins>
</pluginManagement>
</build>
</project>
@@ -0,0 +1,14 @@
plugins {
id("com.iqser.red.service.java-conventions")
id("io.freefair.lombok") version "8.1.0"
}
description = "redaction-service-api-v1"
dependencies {
implementation("org.springframework:spring-web:6.0.11")
implementation("com.iqser.red.service:persistence-service-internal-api-v1:RED-6725")
}
@@ -1,61 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<modelVersion>4.0.0</modelVersion>
<parent>
<artifactId>redaction-service-v1</artifactId>
<groupId>com.iqser.red.service</groupId>
<version>1.0-SNAPSHOT</version>
</parent>
<artifactId>redaction-service-api-v1</artifactId>
<properties>
<persistence-service.version>1.240.0</persistence-service.version>
</properties>
<dependencies>
<!-- https://mvnrepository.com/artifact/com.dslplatform/dsl-json-java8 -->
<dependency>
<groupId>com.dslplatform</groupId>
<artifactId>dsl-json-java8</artifactId>
<version>${dsljson.version}</version>
</dependency>
<dependency>
<groupId>org.springframework</groupId>
<artifactId>spring-web</artifactId>
<optional>true</optional>
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>persistence-service-api-v1</artifactId>
<version>${persistence-service.version}</version>
<exclusions>
<exclusion>
<groupId>com.iqser.red.service</groupId>
<artifactId>redaction-service-api-v1</artifactId>
</exclusion>
</exclusions>
</dependency>
</dependencies>
<build>
<plugins>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-compiler-plugin</artifactId>
<configuration>
<annotationProcessors>
<annotationProcessor>lombok.launch.AnnotationProcessorHider$AnnotationProcessor</annotationProcessor>
<annotationProcessor>com.dslplatform.json.processor.CompiledJsonAnnotationProcessor</annotationProcessor>
</annotationProcessors>
</configuration>
</plugin>
</plugins>
</build>
</project>
@@ -1,39 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnalyzeRequest {
private MessageType messageType;
private String dossierId;
private String fileId;
private String dossierTemplateId;
private ManualRedactions manualRedactions;
private OffsetDateTime lastProcessed;
private int analysisNumber;
@Builder.Default
private Set<Integer> excludedPages = new HashSet<>();
@Builder.Default
private Set<Integer> sectionsToReanalyse = new HashSet<>();
@Builder.Default
private List<FileAttribute> fileAttributes = new ArrayList<>();
}
@@ -1,36 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnalyzeResult {
private MessageType messageType;
private String dossierId;
private String fileId;
private long duration;
private int numberOfPages;
private boolean hasUpdates;
private long dictionaryVersion;
private long dossierDictionaryVersion;
private long rulesVersion;
private long legalBasisVersion;
private boolean wasReanalyzed;
private int analysisVersion;
private int analysisNumber;
private ManualRedactions manualRedactions;
}
@@ -2,6 +2,14 @@ package com.iqser.red.service.redaction.v1.model;
public enum ArgumentType {
INTEGER, BOOLEAN, STRING, FILE_ATTRIBUTE, REGEX, TYPE, RULE_NUMBER, LEGAL_BASIS, REFERENCE_TYPE
INTEGER,
BOOLEAN,
STRING,
FILE_ATTRIBUTE,
REGEX,
TYPE,
RULE_NUMBER,
LEGAL_BASIS,
REFERENCE_TYPE
}
@@ -1,16 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class CellRectangle {
private Point topLeft;
private float width;
private float height;
}
@@ -1,19 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class Change {
private int analysisNumber;
private ChangeType type;
private OffsetDateTime dateTime;
}
@@ -1,5 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum ChangeType {
ADDED, REMOVED, CHANGED
}
@@ -1,5 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum Engine {
DICTIONARY, NER, RULE
}
@@ -1,19 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class FileAttribute {
private String id;
private String label;
private String placeholder;
private String value;
}
@@ -1,21 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ImportedRedaction {
private String id;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
}
@@ -1,22 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class ImportedRedactions {
@Builder.Default
private Map<Integer, List<ImportedRedaction>> importedRedactions = new HashMap<>();
}
@@ -1,50 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.BaseAnnotation;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
import java.util.HashMap;
import java.util.Map;
@Data
@AllArgsConstructor
@NoArgsConstructor
@Builder
public class ManualChange {
private AnnotationStatus annotationStatus;
private ManualRedactionType manualRedactionType;
private OffsetDateTime processedDate;
private OffsetDateTime requestedDate;
private String userId;
private Map<String, String> propertyChanges = new HashMap<>();
public static ManualChange from(BaseAnnotation baseAnnotation) {
ManualChange manualChange = new ManualChange();
manualChange.annotationStatus = baseAnnotation.getStatus();
manualChange.processedDate = baseAnnotation.getProcessedDate();
manualChange.requestedDate = baseAnnotation.getRequestDate();
manualChange.userId = baseAnnotation.getUser();
return manualChange;
}
public boolean isProcessed() {
return processedDate != null;
}
public ManualChange withManualRedactionType(ManualRedactionType manualRedactionType) {
this.manualRedactionType = manualRedactionType;
return this;
}
public ManualChange withChange(String property, String value) {
this.propertyChanges.put(property, value);
return this;
}
}
@@ -1,13 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum ManualRedactionType {
ADD_LOCALLY,
ADD_TO_DICTIONARY,
REMOVE_LOCALLY,
REMOVE_FROM_DICTIONARY,
FORCE_REDACT,
FORCE_HINT,
RECATEGORIZE,
LEGAL_BASIS_CHANGE,
RESIZE
}
@@ -1,7 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum MessageType {
ANALYSE, REANALYSE, STRUCTURE_ANALYSE, SURROUNDING_TEXT
}
@@ -1,15 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class Point {
private float x;
private float y;
}
@@ -1,15 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ReanalyzeResult {
private RedactionLog redactionLog;
}
@@ -1,17 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class Rectangle {
private Point topLeft;
private float width;
private float height;
private int page;
}
@@ -1,39 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@CompiledJson
@AllArgsConstructor
@NoArgsConstructor
public class RedactionLog {
/**
* Version 0 Redaction Logs have manual redactions merged inside them
* Version 1 Redaction Logs only contain system ( rule/dictionary ) redactions. Manual Redactions are merged in at runtime.
*/
private long analysisVersion;
/**
* Which analysis created this redactionLog.
*/
private int analysisNumber;
private List<RedactionLogEntry> redactionLogEntry = new ArrayList<>();
private List<RedactionLogLegalBasis> legalBasis = new ArrayList<>();
private long dictionaryVersion = -1;
private long dossierDictionaryVersion = -1;
private long rulesVersion = -1;
private long legalBasisVersion = -1;
}
@@ -1,17 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class RedactionLogChanges {
private RedactionLog redactionLog;
private boolean hasChanges;
}
@@ -1,22 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.time.OffsetDateTime;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class RedactionLogComment {
private long id;
private String user;
private String text;
private String annotationId;
private String fileId;
private OffsetDateTime date;
private OffsetDateTime softDeletedTime;
}
@@ -1,91 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
import lombok.*;
import java.util.*;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@EqualsAndHashCode
public class RedactionLogEntry {
private String id;
private String type;
private String value;
private String reason;
private int matchedRule;
private boolean rectangle;
private String legalBasis;
private boolean imported;
private boolean redacted;
private boolean isHint;
private boolean isRecommendation;
private boolean isFalsePositive;
private String section;
private float[] color;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private String textBefore;
private String textAfter;
@Builder.Default
private List<RedactionLogComment> comments = new ArrayList<>();
private int startOffset;
private int endOffset;
private boolean isImage;
private boolean imageHasTransparency;
private boolean isDictionaryEntry;
private boolean isDossierDictionaryEntry;
private boolean excluded;
private String sourceId;
@EqualsAndHashCode.Exclude
@Builder.Default
private List<Change> changes = new ArrayList<>();
@EqualsAndHashCode.Exclude
@Builder.Default
private List<ManualChange> manualChanges = new ArrayList<>();
private Set<Engine> engines = new HashSet<>();
private Set<String> reference = new HashSet<>();
@Builder.Default
private Set<String> importedRedactionIntersections = new HashSet<>();
public boolean lastChangeIsRemoved() {
return last(changes).map(c -> c.getType() == ChangeType.REMOVED).orElse(false);
}
public boolean isLocalManualRedaction() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
public boolean isManuallyRemoved() {
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY &&
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
}
private <T> Optional<T> last(List<T> list) {
return list.isEmpty() ? Optional.empty() : Optional.of(list.get(list.size() - 1));
}
}
@@ -1,16 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class RedactionLogLegalBasis {
private String name;
private String description;
private String reason;
}
@@ -1,32 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.Type;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class RedactionRequest {
private String dossierId;
private String fileId;
private String dossierTemplateId;
private ManualRedactions manualRedactions;
@Builder.Default
private Set<Integer> excludedPages = new HashSet<>();
private Colors colors;
private List<Type> types;
private boolean includeFalsePositives;
}
@@ -1,17 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class RedactionResult {
private byte[] document;
private int numberOfPages;
}
@@ -1,27 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class SectionArea {
private Point topLeft;
private float width;
private float height;
private int page;
private String header;
public boolean contains(Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft().getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft().getY() + other.getHeight();
}
// TODO we should only use one rectangle class.
public boolean contains(com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
}
}
@@ -1,31 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.*;
@Data
@CompiledJson
@AllArgsConstructor
@NoArgsConstructor
public class SectionGrid {
private Map<Integer, List<SectionRectangle>> rectanglesPerPage = new HashMap<>();
private List<SectionGridSection> sections = new ArrayList<>();
@Data
@NoArgsConstructor
@AllArgsConstructor
public static class SectionGridSection {
private int sectionNumber;
private String headline;
private Set<Integer> pages;
private List<SectionArea> sectionAreas;
}
}
@@ -1,22 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.List;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class SectionRectangle {
private Point topLeft;
private float width;
private float height;
private int part;
private int numberOfParts;
private List<CellRectangle> tableCells;
}
@@ -1,39 +1,12 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
public interface RedactionResource {
@PostMapping(value = "/debug/classifications", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult classify(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/debug/sections", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult sections(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/debug/htmlTables", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
void testRules(@RequestBody String rules);
@PostMapping(value = "/redaction-log/preview", consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionLog getRedactionLog(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/manual/surrounding-text/{dossierId}/{fileId}", consumes = MediaType.APPLICATION_JSON_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions);
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
@@ -0,0 +1,90 @@
import org.springframework.boot.gradle.tasks.bundling.BootBuildImage
plugins {
application
id("com.iqser.red.service.java-conventions")
id("org.springframework.boot") version "3.1.2"
id("io.spring.dependency-management") version "1.1.2"
id("org.sonarqube") version "4.3.0.3225"
id("io.freefair.lombok") version "8.1.0"
}
description = "redaction-service-server-v1"
val layoutParserVersion = "0.23.0"
val jacksonVersion = "2.15.2"
val droolsVersion = "8.42.0.Final"
val pdfBoxVersion = "3.0.0-alpha2"
configurations {
all {
exclude(group = "org.springframework.boot", module = "spring-boot-starter-logging")
}
}
dependencies {
implementation(project(":redaction-service-api-v1")) { exclude(group = "com.iqser.red.service", module = "persistence-service-internal-api-v1") }
implementation("com.iqser.red.service:persistence-service-internal-api-v1:2.119.0") { exclude(group = "org.springframework.boot") }
implementation("com.knecon.fforesight:layoutparser-service-internal-api:${layoutParserVersion}")
implementation("com.iqser.red.commons:spring-commons:2.6.0")
implementation("com.iqser.red.commons:metric-commons:2.3.0")
implementation("com.iqser.red.commons:dictionary-merge-commons:1.5.0")
implementation("com.iqser.red.commons:storage-commons:2.24.0")
implementation("com.knecon.fforesight:tenant-commons:0.10.0")
implementation("com.fasterxml.jackson.module:jackson-module-afterburner:${jacksonVersion}")
implementation("com.fasterxml.jackson.datatype:jackson-datatype-jsr310:${jacksonVersion}")
implementation("org.ahocorasick:ahocorasick:0.6.3")
implementation("org.javassist:javassist:3.29.2-GA")
implementation("org.drools:drools-engine:${droolsVersion}")
implementation("org.drools:drools-mvel:${droolsVersion}")
implementation("org.kie:kie-spring:7.74.1.Final")
implementation("org.locationtech.jts:jts-core:1.19.0")
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:4.0.4")
implementation("org.springframework.boot:spring-boot-starter-amqp:3.1.2")
testImplementation("org.apache.pdfbox:pdfbox:${pdfBoxVersion}")
testImplementation("org.apache.pdfbox:pdfbox-tools:${pdfBoxVersion}")
testImplementation("org.springframework.boot:spring-boot-starter-test:3.1.2")
testImplementation("com.knecon.fforesight:layoutparser-service-processor:${layoutParserVersion}") {
exclude(
group = "com.iqser.red.service",
module = "persistence-service-shared-api-v1"
)
}
}
tasks.test {
configure<JacocoTaskExtension> {
excludes = listOf("org/drools/**/*")
}
}
tasks.named<BootBuildImage>("bootBuildImage") {
imageName.set("nexus.knecon.com:5001/red/${project.name}:${project.version}")
if (project.hasProperty("buildbootDockerHostNetwork")) {
network.set("host")
}
docker {
if (project.hasProperty("buildbootDockerHostNetwork")) {
bindHostToBuilder.set(true)
}
verboseLogging.set(true)
publishRegistry {
username.set(providers.gradleProperty("mavenUser").getOrNull())
password.set(providers.gradleProperty("mavenPassword").getOrNull())
email.set(providers.gradleProperty("mavenEmail").getOrNull())
url.set("https://nexus.knecon.com:5001/")
}
}
}
@@ -1,198 +0,0 @@
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xmlns="http://maven.apache.org/POM/4.0.0"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<modelVersion>4.0.0</modelVersion>
<parent>
<artifactId>redaction-service-v1</artifactId>
<groupId>com.iqser.red.service</groupId>
<version>1.0-SNAPSHOT</version>
</parent>
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<drools.version>7.68.0.Final</drools.version>
<kie.version>7.68.0.Final</kie.version>
<locationtech.version>1.18.2</locationtech.version>
<javaassist.version>3.28.0-GA</javaassist.version>
<ahocorasick.version>0.6.3</ahocorasick.version>
<jackson.version>2.13.2</jackson.version>
</properties>
<dependencies>
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-aop</artifactId>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<artifactId>storage-commons</artifactId>
</dependency>
<dependency>
<groupId>com.fasterxml.jackson.module</groupId>
<artifactId>jackson-module-afterburner</artifactId>
<version>${jackson.version}</version>
</dependency>
<dependency>
<groupId>com.fasterxml.jackson.datatype</groupId>
<artifactId>jackson-datatype-jsr310</artifactId>
<version>${jackson.version}</version>
</dependency>
<dependency>
<groupId>org.ahocorasick</groupId>
<artifactId>ahocorasick</artifactId>
<version>${ahocorasick.version}</version>
</dependency>
<dependency>
<groupId>org.javassist</groupId>
<artifactId>javassist</artifactId>
<version>${javaassist.version}</version>
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>redaction-service-api-v1</artifactId>
<version>${project.version}</version>
</dependency>
<dependency>
<groupId>org.drools</groupId>
<artifactId>drools-core</artifactId>
<version>${drools.version}</version>
</dependency>
<dependency>
<groupId>org.kie</groupId>
<artifactId>kie-spring</artifactId>
<version>${kie.version}</version>
</dependency>
<dependency>
<groupId>org.locationtech.jts</groupId>
<artifactId>jts-core</artifactId>
<version>${locationtech.version}</version>
</dependency>
<dependency>
<groupId>com.google.guava</groupId>
<artifactId>guava</artifactId>
</dependency>
<!-- commons -->
<dependency>
<groupId>com.iqser.red.commons</groupId>
<artifactId>spring-commons</artifactId>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<artifactId>logging-commons</artifactId>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<artifactId>metric-commons</artifactId>
</dependency>
<!-- other external -->
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
</dependency>
<!-- spring -->
<dependency>
<groupId>org.springframework.cloud</groupId>
<artifactId>spring-cloud-starter-openfeign</artifactId>
</dependency>
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-amqp</artifactId>
</dependency>
<!-- test dependencies -->
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-test</artifactId>
<scope>test</scope>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<artifactId>test-commons</artifactId>
<scope>test</scope>
</dependency>
</dependencies>
<build>
<plugins>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-compiler-plugin</artifactId>
<configuration>
<annotationProcessors>
<annotationProcessor>lombok.launch.AnnotationProcessorHider$AnnotationProcessor</annotationProcessor>
<annotationProcessor>com.dslplatform.json.processor.CompiledJsonAnnotationProcessor</annotationProcessor>
</annotationProcessors>
</configuration>
</plugin>
<plugin>
<!-- generate git.properties for exposure in /info -->
<groupId>pl.project13.maven</groupId>
<artifactId>git-commit-id-plugin</artifactId>
<executions>
<execution>
<goals>
<goal>revision</goal>
</goals>
<configuration>
<generateGitPropertiesFile>true</generateGitPropertiesFile>
<gitDescribe>
<tags>true</tags>
</gitDescribe>
</configuration>
</execution>
</executions>
</plugin>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
<artifactId>maven-jar-plugin</artifactId>
<executions>
<execution>
<id>original-jar</id>
<goals>
<goal>jar</goal>
</goals>
<configuration>
<classifier>original</classifier>
</configuration>
</execution>
</executions>
</plugin>
<plugin>
<!-- repackages the generated jar into a runnable fat-jar and makes it
executable -->
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-maven-plugin</artifactId>
<executions>
<execution>
<goals>
<goal>repackage</goal>
</goals>
<configuration>
<executable>true</executable>
</configuration>
</execution>
</executions>
</plugin>
</plugins>
</build>
</project>
@@ -1,10 +1,8 @@
package com.iqser.red.service.redaction.v1.server;
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.ImportAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
import org.springframework.boot.context.properties.EnableConfigurationProperties;
@@ -12,24 +10,39 @@ import org.springframework.cloud.openfeign.EnableFeignClients;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Import;
import com.iqser.red.service.dictionarymerge.commons.DictionaryMergeService;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import com.iqser.red.storage.commons.StorageAutoConfiguration;
import com.knecon.fforesight.tenantcommons.MultiTenancyAutoConfiguration;
import io.micrometer.core.aop.TimedAspect;
import io.micrometer.core.instrument.MeterRegistry;
@Import({DefaultWebMvcConfiguration.class})
@ImportAutoConfiguration({MultiTenancyAutoConfiguration.class})
@Import({MetricsConfiguration.class, StorageAutoConfiguration.class})
@EnableFeignClients(basePackageClasses = RulesClient.class)
@EnableConfigurationProperties(RedactionServiceSettings.class)
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
public class Application {
public static void main(String[] args) {
System.setProperty("org.apache.pdfbox.rendering.UsePureJavaCMYKConversion", "true");
SpringApplication.run(Application.class, args);
}
@Bean
public TimedAspect timedAspect(MeterRegistry registry) {
return new TimedAspect(registry);
}
@Bean
public DictionaryMergeService dictionaryMergeService() {
return new DictionaryMergeService();
}
}
@@ -0,0 +1,26 @@
package com.iqser.red.service.redaction.v1.server;
import org.springframework.beans.factory.annotation.Qualifier;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.context.annotation.Import;
import com.iqser.gin4.commons.metrics.MetricCommonsConfiguration;
import com.iqser.gin4.commons.metrics.meters.FunctionTimerFactory;
import com.iqser.gin4.commons.metrics.meters.FunctionTimerValues;
@Configuration
@Import({MetricCommonsConfiguration.class})
public class MetricsConfiguration {
private static final String REDACTMANAGER_ANALYZE_PAGEWISE_METRIC_NAME = "redactmanager_analyze.pagewise";
@Bean
@Qualifier(REDACTMANAGER_ANALYZE_PAGEWISE_METRIC_NAME)
public FunctionTimerValues redactmanagerAnalyzePagewise(FunctionTimerFactory factory) {
return factory.create(REDACTMANAGER_ANALYZE_PAGEWISE_METRIC_NAME);
}
}
@@ -1,31 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryVersion;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
public class Document {
private List<Page> pages = new ArrayList<>();
private List<Paragraph> paragraphs = new ArrayList<>();
private List<Header> headers = new ArrayList<>();
private List<Footer> footers = new ArrayList<>();
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
private FloatFrequencyCounter fontSizeCounter = new FloatFrequencyCounter();
private StringFrequencyCounter fontCounter = new StringFrequencyCounter();
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private boolean headlines;
private SectionGrid sectionGrid = new SectionGrid();
private DictionaryVersion dictionaryVersion;
private long rulesVersion;
}
@@ -1,73 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import lombok.Getter;
import java.util.ArrayList;
import java.util.Collections;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
public class FloatFrequencyCounter {
@Getter
Map<Float, Integer> countPerValue = new HashMap<>();
public void add(float value) {
if (!countPerValue.containsKey(value)) {
countPerValue.put(value, 1);
} else {
countPerValue.put(value, countPerValue.get(value) + 1);
}
}
public void addAll(Map<Float, Integer> otherCounter) {
for (Map.Entry<Float, Integer> entry : otherCounter.entrySet()) {
if (countPerValue.containsKey(entry.getKey())) {
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
} else {
countPerValue.put(entry.getKey(), entry.getValue());
}
}
}
public Float getMostPopular() {
Map.Entry<Float, Integer> mostPopular = null;
for (Map.Entry<Float, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
mostPopular = entry;
} else if (entry.getValue() >= mostPopular.getValue()) {
mostPopular = entry;
}
}
return mostPopular != null ? mostPopular.getKey() : null;
}
public List<Float> getHighterThanMostPopular() {
Float mostPopular = getMostPopular();
List<Float> higher = new ArrayList<>();
for (Float value : countPerValue.keySet()) {
if (value > mostPopular) {
higher.add(value);
}
}
return higher.stream().sorted(Collections.reverseOrder()).collect(Collectors.toList());
}
public Float getHighest() {
Float highest = null;
for (Float value : countPerValue.keySet()) {
if (highest == null) {
highest = value;
} else if (value > highest) {
highest = value;
}
}
return highest;
}
}
@@ -1,26 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
import java.util.List;
@Data
@AllArgsConstructor
public class Footer {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -1,26 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
import java.util.List;
@Data
@AllArgsConstructor
public class Header {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -1,6 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
public enum Orientation {
NONE, LEFT, RIGHT
}
@@ -1,42 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.Data;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@RequiredArgsConstructor
public class Page {
@NonNull
private List<AbstractTextContainer> textBlocks;
private List<PdfImage> images = new ArrayList<>();
private Rectangle bodyTextFrame;
private boolean landscape;
private int rotation;
private int pageNumber;
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
private FloatFrequencyCounter fontSizeCounter = new FloatFrequencyCounter();
private StringFrequencyCounter fontCounter = new StringFrequencyCounter();
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private double cropBoxArea;
public boolean isRotated() {
return rotation != 0;
}
}
@@ -1,64 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@NoArgsConstructor
public class Paragraph implements Comparable {
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
private List<PdfImage> images = new ArrayList<>();
private String headline;
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
pageBlocks.forEach(block -> {
if (block instanceof TextBlock) {
searchableText.addAll(((TextBlock) block).getSequences());
}
});
return searchableText;
}
public List<Table> getTables() {
List<Table> tables = new ArrayList<>();
pageBlocks.forEach(block -> {
if (block instanceof Table) {
tables.add((Table) block);
}
});
return tables;
}
public List<TextBlock> getTextBlocks() {
List<TextBlock> textBlocks = new ArrayList<>();
pageBlocks.forEach(block -> {
if (block instanceof TextBlock) {
textBlocks.add((TextBlock) block);
}
});
return textBlocks;
}
@Override
public int compareTo(Object o) {
return 0;
}
}
@@ -1,56 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.*;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class SectionText {
private int sectionNumber;
private String text;
private boolean isTable;
private String headline;
private List<SectionArea> sectionAreas = new ArrayList<>();
private Set<Image> images = new HashSet<>();
private List<TextBlock> textBlocks = new ArrayList<>();
private Map<String, CellValue> tabularData = new HashMap<>();
private List<Integer> cellStarts = new ArrayList<>();
public void setTabularData(Map<String, CellValue> tabularData) {
tabularData.remove(null);
this.tabularData = tabularData;
}
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> {
if (block != null) {
searchableText.addAll(block.getSequences());
}
});
return searchableText;
}
}
@@ -1,20 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedSectionText {
private int sectionNumber;
private String text;
}
@@ -1,23 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.ArrayList;
import java.util.List;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class SimplifiedText {
private int numberOfPages;
private List<SimplifiedSectionText> sectionTexts = new ArrayList<>();
}
@@ -1,49 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import lombok.Getter;
import java.util.HashMap;
import java.util.Map;
public class StringFrequencyCounter {
@Getter
private final Map<String, Integer> countPerValue = new HashMap<>();
public void add(String value) {
if (!countPerValue.containsKey(value)) {
countPerValue.put(value, 1);
} else {
countPerValue.put(value, countPerValue.get(value) + 1);
}
}
public void addAll(Map<String, Integer> otherCounter) {
for (Map.Entry<String, Integer> entry : otherCounter.entrySet()) {
if (countPerValue.containsKey(entry.getKey())) {
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
} else {
countPerValue.put(entry.getKey(), entry.getValue());
}
}
}
public String getMostPopular() {
Map.Entry<String, Integer> mostPopular = null;
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
mostPopular = entry;
} else if (entry.getValue() > mostPopular.getValue()) {
mostPopular = entry;
}
}
return mostPopular != null ? mostPopular.getKey() : null;
}
}
@@ -1,20 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@Data
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class Text {
private int numberOfPages;
private List<SectionText> sectionTexts = new ArrayList<>();
}
@@ -1,149 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.CompiledJson;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.ArrayList;
import java.util.List;
@AllArgsConstructor
@Builder
@Data
@CompiledJson
@NoArgsConstructor
public class TextBlock extends AbstractTextContainer {
@Builder.Default
private List<TextPositionSequence> sequences = new ArrayList<>();
private int rotation;
private String mostPopularWordFont;
private String mostPopularWordStyle;
private float mostPopularWordFontSize;
private float mostPopularWordHeight;
private float mostPopularWordSpaceWidth;
private float highestFontSize;
private String classification;
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation) {
this.minX = minX;
this.maxX = maxX;
this.minY = minY;
this.maxY = maxY;
this.sequences = sequences;
this.rotation = rotation;
}
public TextBlock union(TextPositionSequence r) {
TextBlock union = this.copy();
union.add(r);
return union;
}
public TextBlock union(TextBlock r) {
TextBlock union = this.copy();
union.add(r);
return union;
}
public void add(TextBlock r) {
if (r.getMinX() < minX) {
minX = r.getMinX();
}
if (r.getMaxX() > maxX) {
maxX = r.getMaxX();
}
if (r.getMinY() < minY) {
minY = r.getMinY();
}
if (r.getMaxY() > maxY) {
maxY = r.getMaxY();
}
sequences.addAll(r.getSequences());
}
public void add(TextPositionSequence r) {
if (r.getX1() < minX) {
minX = r.getX1();
}
if (r.getX2() > maxX) {
maxX = r.getX2();
}
if (r.getY1() < minY) {
minY = r.getY1();
}
if (r.getY2() > maxY) {
maxY = r.getY2();
}
}
public TextBlock copy() {
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation);
}
public void resize(float x1, float y1, float width, float height) {
set(x1, y1, x1 + width, y1 + height);
}
public void set(float x1, float y1, float x2, float y2) {
this.minX = Math.min(x1, x2);
this.maxX = Math.max(x1, x2);
this.minY = Math.min(y1, y2);
this.maxY = Math.max(y1, y2);
}
@Override
public String toString() {
StringBuilder builder = new StringBuilder();
for (int i = 0; i < sequences.size(); i++) {
String sequenceAsString = sequences.get(i).toString();
// Fix for missing Whitespace. This is recognized in getSequences method. See PDFTextStripper Line 1730.
if (i != 0 && sequences.get(i - 1).charAt(sequences.get(i - 1).length() - 1) != ' ' && sequenceAsString.charAt(0) != ' ') {
builder.append(' ');
}
builder.append(sequenceAsString);
}
return builder.toString();
}
@Override
@JsonIgnore
@JsonAttribute(ignore = true)
public String getText() {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
}
}
sb.append(word.toString());
previous = word;
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString());
}
}
@@ -1,26 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import com.dslplatform.json.JsonAttribute;
import com.fasterxml.jackson.annotation.JsonIgnore;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
import java.util.List;
@Data
@AllArgsConstructor
public class UnclassifiedText {
private List<TextBlock> textBlocks;
@JsonIgnore
@JsonAttribute(ignore = true)
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -1,332 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.service;
import static java.util.stream.Collectors.toSet;
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.Orientation;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.StringFrequencyCounter;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import org.springframework.stereotype.Service;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.Iterator;
import java.util.List;
@Service
@SuppressWarnings("all")
public class BlockificationService {
static final float THRESHOLD = 1f;
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines,
List<Ruling> verticalRulingLines) {
sortRotatedSequences(textPositions);
List<TextPositionSequence> chunkWords = new ArrayList<>();
List<AbstractTextContainer> chunkBlockList1 = new ArrayList<>();
float minX = 1000, maxX = 0, minY = 1000, maxY = 0;
TextPositionSequence prev = null;
boolean wasSplitted = false;
Float splitX1 = null;
for (TextPositionSequence word : textPositions) {
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
boolean startFromTop = word.getY1() > maxY + word.getHeight();
boolean splitByX = prev != null && maxX + 50 < word.getX1() && prev.getY1() == word.getY1();
boolean newLineAfterSplit = prev != null && word.getY1() != prev.getY1() && wasSplitted && splitX1 != word.getX1();
boolean splittedByRuling = word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines) || word
.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines) || word
.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines) || word
.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines);
if (prev != null && (lineSeparation || startFromTop || splitByX || newLineAfterSplit || splittedByRuling)) {
Orientation prevOrientation = null;
if (!chunkBlockList1.isEmpty()) {
prevOrientation = chunkBlockList1.get(chunkBlockList1.size() - 1).getOrientation();
}
TextBlock cb1 = buildTextBlock(chunkWords);
chunkBlockList1.add(cb1);
chunkWords = new ArrayList<>();
if (splitByX && !splittedByRuling) {
wasSplitted = true;
cb1.setOrientation(Orientation.LEFT);
splitX1 = word.getX1();
} else if (newLineAfterSplit && !splittedByRuling) {
wasSplitted = false;
cb1.setOrientation(Orientation.RIGHT);
splitX1 = null;
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !splittedByRuling)) {
cb1.setOrientation(Orientation.LEFT);
}
minX = 1000;
maxX = 0;
minY = 1000;
maxY = 0;
prev = null;
}
chunkWords.add(word);
prev = word;
if (word.getX1() < minX) {
minX = word.getX1();
}
if (word.getX2() > maxX) {
maxX = word.getX2();
}
if (word.getY1() < minY) {
minY = word.getY1();
}
if (word.getY2() > maxY) {
maxY = word.getY2();
}
}
TextBlock cb1 = buildTextBlock(chunkWords);
if (cb1 != null) {
chunkBlockList1.add(cb1);
}
Iterator<AbstractTextContainer> itty = chunkBlockList1.iterator();
TextBlock previousLeft = null;
TextBlock previousRight = null;
while (itty.hasNext()) {
TextBlock block = (TextBlock) itty.next();
if (previousLeft != null && block.getOrientation().equals(Orientation.LEFT)) {
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft
.getMinY()) {
previousLeft.add(block);
itty.remove();
continue;
}
}
if (previousRight != null && block.getOrientation().equals(Orientation.RIGHT)) {
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight
.getMinY()) {
previousRight.add(block);
itty.remove();
continue;
}
}
if (block.getOrientation().equals(Orientation.LEFT)) {
previousLeft = block;
} else if (block.getOrientation().equals(Orientation.RIGHT)) {
previousRight = block;
}
}
itty = chunkBlockList1.iterator();
TextBlock previous = null;
while (itty.hasNext()) {
TextBlock block = (TextBlock) itty.next();
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY()) || previous != null && previous
.getOrientation()
.equals(Orientation.LEFT) && block.getOrientation()
.equals(Orientation.RIGHT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY())) {
previous.add(block);
itty.remove();
continue;
}
previous = block;
}
return new Page(chunkBlockList1);
}
private boolean equalsWithThreshold(float f1, float f2) {
return Math.abs(f1 - f2) < THRESHOLD;
}
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList) {
TextBlock textBlock = null;
FloatFrequencyCounter lineHeightFrequencyCounter = new FloatFrequencyCounter();
FloatFrequencyCounter fontSizeFrequencyCounter = new FloatFrequencyCounter();
FloatFrequencyCounter spaceFrequencyCounter = new FloatFrequencyCounter();
StringFrequencyCounter fontFrequencyCounter = new StringFrequencyCounter();
StringFrequencyCounter styleFrequencyCounter = new StringFrequencyCounter();
for (TextPositionSequence wordBlock : wordBlockList) {
lineHeightFrequencyCounter.add(wordBlock.getTextHeight());
fontSizeFrequencyCounter.add(wordBlock.getFontSize());
spaceFrequencyCounter.add(wordBlock.getSpaceWidth());
fontFrequencyCounter.add(wordBlock.getFont());
styleFrequencyCounter.add(wordBlock.getFontStyle());
if (textBlock == null) {
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
.getRotation());
} else {
TextBlock spatialEntity = textBlock.union(wordBlock);
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
.getHeight());
}
}
if (textBlock != null) {
textBlock.setMostPopularWordFont(fontFrequencyCounter.getMostPopular());
textBlock.setMostPopularWordStyle(styleFrequencyCounter.getMostPopular());
textBlock.setMostPopularWordFontSize(fontSizeFrequencyCounter.getMostPopular());
textBlock.setMostPopularWordHeight(lineHeightFrequencyCounter.getMostPopular());
textBlock.setMostPopularWordSpaceWidth(spaceFrequencyCounter.getMostPopular());
textBlock.setHighestFontSize(fontSizeFrequencyCounter.getHighest());
}
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences()
.stream()
.map(t -> round(t.getY1(), 3))
.collect(toSet())
.size() == 1) {
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getX1));
}
return textBlock;
}
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1,
List<Ruling> rulingLines) {
for (Ruling ruling : rulingLines) {
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
return true;
}
}
return false;
}
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter,
boolean landscape) {
float minX = 10000;
float maxX = -100;
float minY = 10000;
float maxY = -100;
for (Page page : pages) {
if (page.getTextBlocks().isEmpty() || landscape != page.isLandscape()) {
continue;
}
for (AbstractTextContainer container : page.getTextBlocks()) {
if (container instanceof TextBlock) {
TextBlock textBlock = (TextBlock) container;
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
continue;
}
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
if (approxLineCount < 2.9f) {
continue;
}
if (documentFontSizeCounter.getMostPopular() != null) {
if (textBlock.getMostPopularWordFontSize() >= documentFontSizeCounter.getMostPopular()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
if (textBlock.getMaxX() > maxX) {
maxX = textBlock.getMaxX();
}
if (textBlock.getMinY() < minY) {
minY = textBlock.getMinY();
}
if (textBlock.getMaxY() > maxY) {
maxY = textBlock.getMaxY();
}
}
}
}
if (container instanceof Table) {
Table table = (Table) container;
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (cell == null || cell.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : cell.getTextBlocks()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
if (textBlock.getMaxX() > maxX) {
maxX = textBlock.getMaxX();
}
if (textBlock.getMinY() < minY) {
minY = textBlock.getMinY();
}
if (textBlock.getMaxY() > maxY) {
maxY = textBlock.getMaxY();
}
}
}
}
}
}
}
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
}
private void sortRotatedSequences(List<TextPositionSequence> sequences) {
List<TextPositionSequence> rotatedWords = new ArrayList<>();
Iterator<TextPositionSequence> itty = sequences.iterator();
while (itty.hasNext()) {
var pos = itty.next();
if (pos.getTextPositions().get(0).getDir() == 270) {
rotatedWords.add(pos);
itty.remove();
}
}
if (!rotatedWords.isEmpty() && !sequences.isEmpty()) {
rotatedWords.sort(Comparator.comparing(TextPositionSequence::getX1));
}
sequences.addAll(rotatedWords);
}
private double round(float value, int decimalPoints) {
var d = Math.pow(10, decimalPoints);
return Math.round(value * d) / d;
}
}
@@ -1,119 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import java.util.List;
import java.util.regex.Pattern;
@Slf4j
@Service
@RequiredArgsConstructor
public class ClassificationService {
private final BlockificationService blockificationService;
public void classifyDocument(Document document) {
Rectangle bodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false);
Rectangle landscapeBodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true);
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
for (Page page : document.getPages()) {
Rectangle btf = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
page.setBodyTextFrame(btf);
classifyPage(btf, page, document, headlineFontSizes);
}
}
public void classifyPage(Rectangle bodyTextFrame, Page page, Document document, List<Float> headlineFontSizes) {
for (AbstractTextContainer textBlock : page.getTextBlocks()) {
if (textBlock instanceof TextBlock) {
classifyBlock((TextBlock) textBlock, bodyTextFrame, page, document, headlineFontSizes);
}
}
}
public void classifyBlock(TextBlock textBlock, Rectangle bodyTextFrame, Page page, Document document,
List<Float> headlineFontSizes) {
if (document.getFontSizeCounter().getMostPopular() == null) {
textBlock.setClassification("Other");
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
textBlock.setClassification("Header");
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
textBlock.setClassification("Footer");
} else if (page.getPageNumber() == 1 && (!PositionUtils.isTouchingUnderBodyTextFrame(bodyTextFrame, textBlock) && PositionUtils
.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock, document.getTextHeightCounter()
.getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter()
.getMostPopular() || page.getTextBlocks().size() == 1)) {
if (!Pattern.matches("[0-9]+", textBlock.toString())) {
textBlock.setClassification("Title");
}
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle()
.equals("bold") || !document.getFontStyleCounter().getCountPerValue().containsKey("bold") && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() + 1) && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
for (int i = 1; i <= headlineFontSizes.size(); i++) {
if (textBlock.getMostPopularWordFontSize() == headlineFontSizes.get(i - 1)) {
textBlock.setClassification("H " + i);
document.setHeadlines(true);
}
}
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText()
.startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter()
.getMostPopular()
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
textBlock.setClassification("H " + (headlineFontSizes.size() + 1));
document.setHeadlines(true);
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
textBlock.setClassification("TextBlock Bold");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFont()
.equals(document.getFontCounter().getMostPopular()) && textBlock.getMostPopularWordStyle()
.equals(document.getFontStyleCounter()
.getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
.getMostPopular()) {
textBlock.setClassification("TextBlock");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("italic") && !document.getFontStyleCounter()
.getMostPopular()
.equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
textBlock.setClassification("TextBlock Italic");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock)) {
textBlock.setClassification("TextBlock Unknown");
} else {
textBlock.setClassification("Other");
}
}
}
@@ -1,94 +0,0 @@
package com.iqser.red.service.redaction.v1.server.classification.utils;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.experimental.UtilityClass;
@UtilityClass
@SuppressWarnings("all")
public class PositionUtils {
public boolean isWithinBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
if (btf == null || textBlock == null) {
return false;
}
double threshold = textBlock.getMostPopularWordHeight() * 3;
if (textBlock.getMinX() + threshold > btf.getX() &&
textBlock.getMaxX() - threshold < btf.getX() + btf.getWidth() &&
textBlock.getMinY() + threshold > btf.getY() &&
textBlock.getMaxY() - threshold < btf.getY() + btf.getHeight()) {
return true;
} else {
return false;
}
}
public boolean isOverBodyTextFrame(Rectangle btf, TextBlock textBlock, boolean rotated) {
if (btf == null || textBlock == null) {
return false;
}
if (rotated && textBlock.getMinX() < btf.getX()) {
// Its very strange, P{0,0} is on top left in this case, instead of lower left.
return true;
} else if (!rotated && textBlock.getMinY() > btf.getY() + btf.getHeight()) {
return true;
} else {
return false;
}
}
public boolean isUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
if (btf == null || textBlock == null) {
return false;
}
if (textBlock.getMaxY() < btf.getY()) {
return true;
} else {
return false;
}
}
public boolean isTouchingUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
//TODO Currently this is not working for rotated pages.
if (btf == null || textBlock == null) {
return false;
}
if (textBlock.getMinY() < btf.getY()) {
return true;
} else {
return false;
}
}
public float getHeightDifferenceBetweenChunkWordAndDocumentWord(TextBlock textBlock, Float documentMostPopularWordHeight) {
return textBlock.getMostPopularWordHeight() - documentMostPopularWordHeight;
}
public Float getApproxLineCount(TextBlock textBlock) {
return textBlock.getHeight() / textBlock.getMostPopularWordHeight();
}
}
@@ -2,8 +2,9 @@ package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.persistence.service.v1.api.resources.DictionaryResource;
import com.iqser.red.service.persistence.service.v1.api.internal.resources.DictionaryResource;
@FeignClient(name = "DictionaryResource", url = "${persistence-service.url}")
public interface DictionaryClient extends DictionaryResource {
}
}
@@ -1,16 +0,0 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
import com.iqser.red.service.redaction.v1.server.client.model.EntityRecognitionRequest;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
@FeignClient(name = "EntityRecognitionClient", url = "${entity-recognition-service.url}")
public interface EntityRecognitionClient {
@PostMapping(value = "/find_authors", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
NerEntities findAuthors(EntityRecognitionRequest entityRecognitionRequest);
}
@@ -1,10 +1,10 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.persistence.service.v1.api.resources.FileStatusProcessingUpdateResource;
import com.iqser.red.service.persistence.service.v1.api.internal.resources.FileStatusProcessingUpdateResource;
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${persistence-service.url}")
public interface FileStatusProcessingUpdateClient extends FileStatusProcessingUpdateResource {
}
@@ -2,8 +2,9 @@ package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.persistence.service.v1.api.resources.LegalBasisMappingResource;
import com.iqser.red.service.persistence.service.v1.api.internal.resources.LegalBasisMappingResource;
@FeignClient(name = "LegalBasisMappingResource", url = "${persistence-service.url}")
public interface LegalBasisClient extends LegalBasisMappingResource {
}
@@ -32,8 +32,7 @@ public class MockMultipartFile implements MultipartFile {
}
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
@Nullable byte[] content) {
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, @Nullable byte[] content) {
Assert.hasLength(name, "Name must not be empty");
this.name = name;
@@ -43,8 +42,7 @@ public class MockMultipartFile implements MultipartFile {
}
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
InputStream contentStream) throws IOException {
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, InputStream contentStream) throws IOException {
this(name, originalFilename, contentType, FileCopyUtils.copyToByteArray(contentStream));
}
@@ -2,8 +2,9 @@ package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.persistence.service.v1.api.resources.RulesResource;
import com.iqser.red.service.persistence.service.v1.api.internal.resources.RulesResource;
@FeignClient(name = "RulesResource", url = "${persistence-service.url}")
public interface RulesClient extends RulesResource {
}
@@ -1,21 +0,0 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@CompiledJson
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecogintionEntity {
private String value;
private int startOffset;
private int endOffset;
private String type;
}
@@ -1,6 +1,5 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
@@ -11,8 +10,11 @@ import lombok.NoArgsConstructor;
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecognitionRequest {
public class EntityRecognitionEntity {
private List<EntityRecognitionSection> data;
private String value;
private int startOffset;
private int endOffset;
private String type;
}
@@ -1,20 +0,0 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class EntityRecognitionResult {
@Builder.Default
private Map<Integer, List<EntityRecogintionEntity>> entities = new HashMap<>();
}
@@ -13,4 +13,5 @@ public class EntityRecognitionSection {
private int sectionNumber;
private String text;
}
@@ -4,18 +4,15 @@ import java.util.HashMap;
import java.util.List;
import java.util.Map;
import com.dslplatform.json.CompiledJson;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@CompiledJson
@NoArgsConstructor
@AllArgsConstructor
public class NerEntities {
public class NerEntitiesModel {
private Map<Integer, List<EntityRecogintionEntity>> data = new HashMap<>();
private Map<Integer, List<EntityRecognitionEntity>> data = new HashMap<>();
}
@@ -3,7 +3,9 @@ package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.commons.spring.ErrorMessage;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import lombok.extern.slf4j.Slf4j;
import org.springframework.http.HttpStatus;
import org.springframework.web.bind.annotation.ExceptionHandler;
import org.springframework.web.bind.annotation.ResponseBody;
@@ -18,10 +20,12 @@ public class ControllerAdvice {
/* error handling */
@ResponseBody
@ResponseStatus(value = HttpStatus.INTERNAL_SERVER_ERROR)
@ExceptionHandler(value = NullPointerException.class)
public ErrorMessage handleContentNotFoundException(NullPointerException e) {
if (e != null) {
log.error(e.getMessage(), e);
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
@@ -30,17 +34,21 @@ public class ControllerAdvice {
return new ErrorMessage(OffsetDateTime.now(), "Nullpointer exception");
}
@ResponseBody
@ResponseStatus(value = HttpStatus.BAD_REQUEST)
@ExceptionHandler(value = RulesValidationException.class)
public ErrorMessage handleRulesValidationException(RulesValidationException e) {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
@ResponseBody
@ResponseStatus(value = HttpStatus.NOT_FOUND)
@ExceptionHandler(value = NotFoundException.class)
public ErrorMessage handleFileNotFoundException(NotFoundException e) {
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
}
@@ -1,156 +1,31 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.redaction.v1.model.*;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.*;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.stream.Collectors;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RestController
@RequiredArgsConstructor
public class RedactionController implements RedactionResource {
private final PdfVisualisationService pdfVisualisationService;
private final DroolsExecutionService droolsExecutionService;
private final PdfSegmentationService pdfSegmentationService;
private final RedactionStorageService redactionStorageService;
private final RedactionLogMergeService redactionLogMergeService;
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
@Override
public RedactionResult classify(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
pdDocument.setAllSecurityToBeRemoved(true);
pdfVisualisationService.visualizeClassifications(classifiedDoc, pdDocument);
return convert(pdDocument, classifiedDoc.getPages().size());
} catch (IOException e) {
throw new RedactionException(e);
}
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try {
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
pdDocument.setAllSecurityToBeRemoved(true);
pdfVisualisationService.visualizeParagraphs(classifiedDoc, pdDocument);
return convert(pdDocument, classifiedDoc.getPages().size());
} catch (IOException e) {
throw new RedactionException(e);
}
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest) {
Document classifiedDoc;
try {
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
} catch (Exception e) {
throw new RedactionException(e);
}
StringBuilder sb = new StringBuilder();
for (Page page : classifiedDoc.getPages()) {
for (AbstractTextContainer textContainer : page.getTextBlocks()) {
if (textContainer instanceof Table) {
Table table = (Table) textContainer;
sb.append(table.getTextAsHtml()).append("<br />").append("<br />");
}
}
}
return RedactionResult.builder().document(sb.toString().getBytes()).build();
}
@Override
public void testRules(@RequestBody String rules) {
droolsExecutionService.testRules(rules);
}
@Override
public RedactionLog getRedactionLog(RedactionRequest redactionRequest) {
return redactionLogMergeService.provideRedactionLog(redactionRequest);
}
private RedactionResult convert(PDDocument document, int numberOfPages) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
return RedactionResult.builder()
.document(byteArrayOutputStream.toByteArray())
.numberOfPages(numberOfPages)
.build();
try {
droolsExecutionService.testRules(rules);
} catch (Exception e) {
throw new RulesValidationException("Could not test rules: " + e.getMessage(), e);
}
}
@Override
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
@PathVariable("fileId") String fileId,
@RequestBody ManualRedactions manualRedactions) {
var result = manualRedactionSurroundingTextService.addSurroundingText(dossierId, fileId, manualRedactions);
log.info("Added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", dossierId, fileId, result.getDuration());
return result.getManualRedactions();
}
}
@@ -1,11 +1,12 @@
package com.iqser.red.service.redaction.v1.server.controller;
import org.springframework.web.bind.annotation.RestController;
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
import com.iqser.red.service.redaction.v1.resources.RuleBuilderResource;
import com.iqser.red.service.redaction.v1.server.redaction.rulebuilder.RuleBuilderModelService;
import com.iqser.red.service.redaction.v1.server.redaction.service.RuleBuilderModelService;
import lombok.RequiredArgsConstructor;
import org.springframework.web.bind.annotation.RestController;
@RestController
@RequiredArgsConstructor
@@ -13,8 +14,10 @@ public class RuleBuilderController implements RuleBuilderResource {
private final RuleBuilderModelService ruleBuilderModelService;
@Override
public RuleBuilderModel getRuleBuilderModel() {
return ruleBuilderModelService.getRuleBuilderModel();
}
@@ -0,0 +1,25 @@
package com.iqser.red.service.redaction.v1.server.document.data;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPage;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPositionData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentStructure;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentTextData;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class DocumentData {
DocumentPage[] documentPages;
DocumentTextData[] documentTextData;
DocumentPositionData[] documentPositionData;
DocumentStructure documentStructure;
}
@@ -0,0 +1,198 @@
package com.iqser.red.service.redaction.v1.server.document.data.mapper;
import java.util.Arrays;
import java.util.HashSet;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.NoSuchElementException;
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentTree;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Footer;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Header;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Image;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Page;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Paragraph;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Section;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SemanticNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableCell;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Document;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Headline;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Table;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPage;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPositionData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentStructure;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentTextData;
import lombok.experimental.UtilityClass;
@UtilityClass
public class DocumentGraphMapper {
public Document toDocumentGraph(DocumentData documentData) {
Document document = new Document();
DocumentTree documentTree = new DocumentTree(document);
Context context = new Context(documentData, documentTree);
context.pageData.addAll(Arrays.stream(documentData.getDocumentPages()).map(DocumentGraphMapper::buildPage).toList());
context.documentTree.getRoot().getChildren().addAll(buildEntries(documentData.getDocumentStructure().getRoot().getChildren(), context));
document.setDocumentTree(context.documentTree);
document.setPages(new HashSet<>(context.pageData));
document.setNumberOfPages(documentData.getDocumentPages().length);
document.setTextBlock(document.getTextBlock());
return document;
}
private List<DocumentTree.Entry> buildEntries(List<DocumentStructure.EntryData> entries, Context context) {
List<DocumentTree.Entry> newEntries = new LinkedList<>();
for (DocumentStructure.EntryData entryData : entries) {
List<Page> pages = Arrays.stream(entryData.getPageNumbers()).map(pageNumber -> getPage(pageNumber, context)).toList();
SemanticNode node = switch (entryData.getType()) {
case SECTION -> buildSection(context);
case PARAGRAPH -> buildParagraph(context);
case HEADLINE -> buildHeadline(context);
case HEADER -> buildHeader(context);
case FOOTER -> buildFooter(context);
case TABLE -> buildTable(context, entryData.getProperties());
case TABLE_CELL -> buildTableCell(context, entryData.getProperties());
case IMAGE -> buildImage(context, entryData.getProperties(), entryData.getPageNumbers());
default -> throw new UnsupportedOperationException("Not yet implemented for type " + entryData.getType());
};
if (entryData.getAtomicBlockIds().length > 0) {
TextBlock textBlock = toTextBlock(entryData.getAtomicBlockIds(), context, node);
node.setLeafTextBlock(textBlock);
}
List<Integer> treeId = Arrays.stream(entryData.getTreeId()).boxed().toList();
node.setTreeId(treeId);
switch (entryData.getType()) {
case HEADER -> pages.forEach(page -> page.setHeader((Header) node));
case FOOTER -> pages.forEach(page -> page.setFooter((Footer) node));
default -> pages.forEach(page -> page.getMainBody().add(node));
}
newEntries.add(DocumentTree.Entry.builder().treeId(treeId).children(buildEntries(entryData.getChildren(), context)).node(node).build());
}
return newEntries;
}
private Headline buildHeadline(Context context) {
return Headline.builder().documentTree(context.documentTree).build();
}
private Image buildImage(Context context, Map<String, String> properties, Long[] pageNumbers) {
assert pageNumbers.length == 1;
Page page = getPage(pageNumbers[0], context);
var builder = Image.builder();
PropertiesMapper.parseImageProperties(properties, builder);
return builder.documentTree(context.documentTree).page(page).build();
}
private TableCell buildTableCell(Context context, Map<String, String> properties) {
TableCell.TableCellBuilder builder = TableCell.builder();
PropertiesMapper.parseTableCellProperties(properties, builder);
return builder.documentTree(context.documentTree).build();
}
private Table buildTable(Context context, Map<String, String> properties) {
Table.TableBuilder builder = Table.builder();
PropertiesMapper.parseTableProperties(properties, builder);
return builder.documentTree(context.documentTree).build();
}
private Footer buildFooter(Context context) {
return Footer.builder().documentTree(context.documentTree).build();
}
private Header buildHeader(Context context) {
return Header.builder().documentTree(context.documentTree).build();
}
private Section buildSection(Context context) {
return Section.builder().documentTree(context.documentTree).build();
}
private Paragraph buildParagraph(Context context) {
return Paragraph.builder().documentTree(context.documentTree).build();
}
private TextBlock toTextBlock(Long[] atomicTextBlockIds, Context context, SemanticNode parent) {
return Arrays.stream(atomicTextBlockIds).map(atomicTextBlockId -> getAtomicTextBlock(context, parent, atomicTextBlockId)).collect(new TextBlockCollector());
}
private AtomicTextBlock getAtomicTextBlock(Context context, SemanticNode parent, Long atomicTextBlockId) {
return AtomicTextBlock.fromAtomicTextBlockData(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)),
context.documentPositionData.get(Math.toIntExact(atomicTextBlockId)),
parent,
getPage(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)).getPage(), context));
}
private Page buildPage(DocumentPage p) {
return Page.builder().rotation(p.getRotation()).height(p.getHeight()).width(p.getWidth()).number(p.getNumber()).mainBody(new LinkedList<>()).build();
}
private Page getPage(Long pageIndex, Context context) {
return context.pageData.stream()
.filter(page -> page.getNumber() == Math.toIntExact(pageIndex))
.findFirst()
.orElseThrow(() -> new NoSuchElementException(String.format("ClassificationPage with number %d not found", pageIndex)));
}
static final class Context {
private final DocumentTree documentTree;
private final List<Page> pageData;
private final List<DocumentTextData> documentTextData;
private final List<DocumentPositionData> documentPositionData;
Context(DocumentData documentData, DocumentTree documentTree) {
this.documentTree = documentTree;
this.pageData = new LinkedList<>();
this.documentTextData = Arrays.stream(documentData.getDocumentTextData()).toList();
this.documentPositionData = Arrays.stream(documentData.getDocumentPositionData()).toList();
}
}
}
@@ -0,0 +1,52 @@
package com.iqser.red.service.redaction.v1.server.document.data.mapper;
import java.awt.geom.Rectangle2D;
import java.util.Arrays;
import java.util.List;
import java.util.Map;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Image;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ImageType;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Table;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableCell;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentStructure;
import lombok.AccessLevel;
import lombok.experimental.FieldDefaults;
import lombok.experimental.UtilityClass;
@UtilityClass
public class PropertiesMapper {
public void parseImageProperties(Map<String, String> properties, Image.ImageBuilder builder) {
builder.imageType(ImageType.fromString(properties.get(DocumentStructure.ImageProperties.IMAGE_TYPE)));
builder.transparent(Boolean.parseBoolean(properties.get(DocumentStructure.ImageProperties.TRANSPARENT)));
builder.position(parseRectangle2D(properties.get(DocumentStructure.ImageProperties.POSITION)));
builder.id(properties.get(DocumentStructure.ImageProperties.ID));
}
public void parseTableCellProperties(Map<String, String> properties, TableCell.TableCellBuilder builder) {
builder.row(Integer.parseInt(properties.get(DocumentStructure.TableCellProperties.ROW)));
builder.col(Integer.parseInt(properties.get(DocumentStructure.TableCellProperties.COL)));
builder.header(Boolean.parseBoolean(properties.get(DocumentStructure.TableCellProperties.HEADER)));
builder.bBox(parseRectangle2D(properties.get(DocumentStructure.TableCellProperties.B_BOX)));
}
public void parseTableProperties(Map<String, String> properties, Table.TableBuilder builder) {
builder.numberOfRows(Integer.parseInt(properties.get(DocumentStructure.TableProperties.NUMBER_OF_ROWS)));
builder.numberOfCols(Integer.parseInt(properties.get(DocumentStructure.TableProperties.NUMBER_OF_COLS)));
}
private Rectangle2D parseRectangle2D(String bBox) {
List<Float> floats = Arrays.stream(bBox.split(DocumentStructure.RECTANGLE_DELIMITER)).map(Float::parseFloat).toList();
return new Rectangle2D.Float(floats.get(0), floats.get(1), floats.get(2), floats.get(3));
}
}
@@ -0,0 +1,166 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import static java.lang.String.format;
import java.util.Collection;
import java.util.LinkedList;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
import lombok.EqualsAndHashCode;
import lombok.Setter;
@Setter
@EqualsAndHashCode
public class Boundary implements Comparable<Boundary> {
private int start;
private int end;
public Boundary(int start, int end) {
if (start > end) {
throw new IllegalArgumentException(format("start: %d > end: %d", start, end));
}
this.start = start;
this.end = end;
}
public int length() {
return end - start;
}
public int start() {
return start;
}
public int end() {
return end;
}
public boolean contains(Boundary boundary) {
return start <= boundary.start() && boundary.end() <= end;
}
public boolean containedBy(Boundary boundary) {
return boundary.contains(this);
}
public boolean contains(int start, int end) {
if (start > end) {
throw new IllegalArgumentException(format("start: %d > end: %d", start, end));
}
return this.start <= start && end <= this.end;
}
public boolean containedBy(int start, int end) {
if (start > end) {
throw new IllegalArgumentException(format("start: %d > end: %d", start, end));
}
return start <= this.start && this.end <= end;
}
public boolean contains(int index) {
return start <= index && index < end;
}
public boolean intersects(Boundary boundary) {
return boundary.start() < this.end && this.start < boundary.end();
}
public List<Boundary> split(List<Integer> splitIndices) {
if (splitIndices.stream().anyMatch(idx -> !this.contains(idx))) {
throw new IndexOutOfBoundsException(format("%s splitting indices are out of range for %s", splitIndices.stream().filter(idx -> !this.contains(idx)).toList(), this));
}
List<Boundary> splitBoundaries = new LinkedList<>();
int previousIndex = start;
for (int splitIndex : splitIndices) {
// skip split if it would produce a boundary of length 0
if (splitIndex == previousIndex) {
continue;
}
splitBoundaries.add(new Boundary(previousIndex, splitIndex));
previousIndex = splitIndex;
}
splitBoundaries.add(new Boundary(previousIndex, end));
return splitBoundaries;
}
public static Boundary merge(Collection<Boundary> boundaries) {
int minStart = boundaries.stream().mapToInt(Boundary::start).min().orElseThrow(IllegalArgumentException::new);
int maxEnd = boundaries.stream().mapToInt(Boundary::end).max().orElseThrow(IllegalArgumentException::new);
return new Boundary(minStart, maxEnd);
}
@Override
public String toString() {
return format("Boundary [%d|%d)", start, end);
}
@Override
public int compareTo(Boundary boundary) {
if (end < boundary.end() && start < boundary.start()) {
return -1;
}
if (start > boundary.start() && end > boundary.end()) {
return 1;
}
return 0;
}
/**
* shrinks the boundary, such that textBlock.subSequence(boundary) returns a string without trailing or preceding whitespaces.
*
* @param textBlock TextBlock to check whitespaces against
* @return trimmed boundary
*/
public Boundary trim(TextBlock textBlock) {
if (this.length() == 0) {
return this;
}
int trimmedStart = this.start;
while (textBlock.containsIndex(trimmedStart) && trimmedStart < end && Character.isWhitespace(textBlock.charAt(trimmedStart))) {
trimmedStart++;
}
int trimmedEnd = this.end;
while (textBlock.containsIndex(trimmedEnd - 1) && trimmedStart < trimmedEnd && Character.isWhitespace(textBlock.charAt(trimmedEnd - 1))) {
trimmedEnd--;
}
return new Boundary(trimmedStart, Math.max(trimmedEnd, trimmedStart));
}
}
@@ -0,0 +1,70 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import java.util.LinkedList;
import java.util.List;
import java.util.Set;
import java.util.function.BiConsumer;
import java.util.function.BinaryOperator;
import java.util.function.Function;
import java.util.function.Supplier;
import java.util.stream.Collector;
import com.google.common.base.Functions;
public class ConsecutiveBoundaryCollector implements Collector<Boundary, List<Boundary>, List<Boundary>> {
@Override
public Supplier<List<Boundary>> supplier() {
return LinkedList::new;
}
@Override
public BiConsumer<List<Boundary>, Boundary> accumulator() {
return (existingList, boundary) -> {
if (existingList.isEmpty()) {
existingList.add(boundary);
return;
}
Boundary prevBoundary = existingList.get(existingList.size() - 1);
if (prevBoundary.end() > boundary.start()) {
throw new IllegalArgumentException(String.format("Can't concatenate %s and %s. Boundaries must be ordered!", prevBoundary, boundary));
}
if (prevBoundary.end() == boundary.start()) {
existingList.remove(existingList.size() - 1);
existingList.add(Boundary.merge(List.of(prevBoundary, boundary)));
} else {
existingList.add(boundary);
}
};
}
@Override
public BinaryOperator<List<Boundary>> combiner() {
return (list1, list2) -> {
list1.addAll(list2);
return list1;
};
}
@Override
public Function<List<Boundary>, List<Boundary>> finisher() {
return Functions.identity();
}
@Override
public Set<Characteristics> characteristics() {
return Set.of(Characteristics.IDENTITY_FINISH);
}
}
@@ -0,0 +1,217 @@
package com.iqser.red.service.redaction.v1.server.document.graph;
import static java.lang.String.format;
import java.util.Collections;
import java.util.LinkedList;
import java.util.List;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.GenericSemanticNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SemanticNode;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableCell;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Document;
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.Table;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.Getter;
import lombok.experimental.FieldDefaults;
@Data
@EqualsAndHashCode
public class DocumentTree {
private final Entry root;
public DocumentTree(Document document) {
root = Entry.builder().treeId(Collections.emptyList()).children(new LinkedList<>()).node(document).build();
}
public TextBlock buildTextBlock() {
return allEntriesInOrder().map(Entry::getNode).filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
}
public List<Integer> createNewMainEntryAndReturnId(GenericSemanticNode node) {
return createNewChildEntryAndReturnIdImpl(Collections.emptyList(), node);
}
public List<Integer> createNewChildEntryAndReturnId(GenericSemanticNode parentNode, GenericSemanticNode node) {
return createNewChildEntryAndReturnIdImpl(parentNode.getTreeId(), node);
}
public List<Integer> createNewChildEntryAndReturnId(GenericSemanticNode parentNode, Table node) {
return createNewChildEntryAndReturnIdImpl(parentNode.getTreeId(), node);
}
public List<Integer> createNewTableChildEntryAndReturnId(Table parentTable, TableCell tableCell) {
return createNewChildEntryAndReturnIdImpl(parentTable.getTreeId(), tableCell);
}
@SuppressWarnings("PMD.UnusedPrivateMethod") // PMD actually flags this wrong
private List<Integer> createNewChildEntryAndReturnIdImpl(List<Integer> parentId, SemanticNode node) {
if (!entryExists(parentId)) {
throw new IllegalArgumentException(format("parentId %s does not exist!", parentId));
}
Entry parent = getEntryById(parentId);
List<Integer> newId = new LinkedList<>(parentId);
newId.add(parent.children.size());
parent.children.add(Entry.builder().treeId(newId).node(node).build());
return newId;
}
private boolean entryExists(List<Integer> treeId) {
if (treeId.isEmpty()) {
return root != null;
}
Entry entry = root.children.get(treeId.get(0));
for (int id : treeId.subList(1, treeId.size())) {
if (id >= entry.children.size() || 0 > id) {
return false;
}
entry = entry.children.get(id);
}
return true;
}
public Entry getParentEntryById(List<Integer> treeId) {
return getEntryById(getParentId(treeId));
}
public boolean hasParentById(List<Integer> treeId) {
return !treeId.isEmpty();
}
public Stream<SemanticNode> childNodes(List<Integer> treeId) {
return getEntryById(treeId).children.stream().map(Entry::getNode);
}
public Stream<SemanticNode> childNodesOfType(List<Integer> treeId, NodeType nodeType) {
return getEntryById(treeId).children.stream().filter(entry -> entry.node.getType().equals(nodeType)).map(Entry::getNode);
}
private static List<Integer> getParentId(List<Integer> treeId) {
if (treeId.isEmpty()) {
throw new UnsupportedOperationException("Root has no parent!");
}
if (treeId.size() < 2) {
return Collections.emptyList();
}
return treeId.subList(0, treeId.size() - 1);
}
public Entry getEntryById(List<Integer> treeId) {
if (treeId.isEmpty()) {
return root;
}
Entry entry = root.children.get(treeId.get(0));
for (int id : treeId.subList(1, treeId.size())) {
entry = entry.children.get(id);
}
return entry;
}
public Stream<Entry> mainEntries() {
return root.children.stream();
}
public Stream<Entry> allEntriesInOrder() {
return Stream.of(root).flatMap(DocumentTree::flatten);
}
public Stream<Entry> allSubEntriesInOrder(List<Integer> parentId) {
return getEntryById(parentId).children.stream().flatMap(DocumentTree::flatten);
}
@Override
public String toString() {
return String.join("\n", allEntriesInOrder().map(Entry::toString).toList());
}
private static Stream<Entry> flatten(Entry entry) {
return Stream.concat(Stream.of(entry), entry.children.stream().flatMap(DocumentTree::flatten));
}
public SemanticNode getHighestParentById(List<Integer> treeId) {
if (treeId.isEmpty()) {
return root.node;
}
return root.children.get(treeId.get(0)).node;
}
@Builder
@Getter
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE, makeFinal = true)
public static class Entry {
List<Integer> treeId;
SemanticNode node;
@Builder.Default
List<Entry> children = new LinkedList<>();
@Override
public String toString() {
return node.toString();
}
public NodeType getType() {
return node.getType();
}
}
}
@@ -0,0 +1,8 @@
package com.iqser.red.service.redaction.v1.server.document.graph.entity;
public enum EntityType {
ENTITY,
RECOMMENDATION,
FALSE_POSITIVE,
FALSE_RECOMMENDATION
}
@@ -0,0 +1,70 @@
package com.iqser.red.service.redaction.v1.server.document.graph.entity;
import java.util.Collections;
import java.util.Objects;
import java.util.Set;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.EqualsAndHashCode;
import lombok.Getter;
import lombok.experimental.FieldDefaults;
@Getter
@Builder
@AllArgsConstructor
@EqualsAndHashCode
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public final class MatchedRule implements Comparable<MatchedRule> {
@Builder.Default
RuleIdentifier ruleIdentifier = RuleIdentifier.empty();
@Builder.Default
String reason = "";
@Builder.Default
String legalBasis = "";
boolean applied;
boolean writeValueWithLineBreaks;
@Builder.Default
Set<RedactionEntity> references = Collections.emptySet();
public static MatchedRule empty() {
return MatchedRule.builder().build();
}
@Override
public int compareTo(MatchedRule matchedRule) {
RuleIdentifier otherRuleIdentifier = matchedRule.getRuleIdentifier();
if (!Objects.equals(ruleIdentifier.type(), otherRuleIdentifier.type())) {
if (Objects.equals(otherRuleIdentifier.type(), "MAN")) {
return 1;
}
if (Objects.equals(ruleIdentifier.type(), "MAN")) {
return -1;
}
if (Objects.equals(otherRuleIdentifier.type(), "X")) {
return 1;
}
if (Objects.equals(ruleIdentifier.type(), "X")) {
return -1;
}
}
if (!Objects.equals(otherRuleIdentifier.unit(), getRuleIdentifier().unit())) {
return otherRuleIdentifier.unit() - ruleIdentifier.unit();
}
return otherRuleIdentifier.id() - ruleIdentifier.id();
}
@Override
public String toString() {
return "MatchedRule[" + "ruleIdentifier=" + ruleIdentifier + ", " + "reason=" + reason + ", " + "legalBasis=" + legalBasis + ", " + "applied=" + applied + ", " + "writeValueWithLineBreaks=" + writeValueWithLineBreaks + ", " + "references=" + references + ']';
}
}
@@ -0,0 +1,159 @@
package com.iqser.red.service.redaction.v1.server.document.graph.entity;
import java.util.Collection;
import java.util.HashSet;
import java.util.PriorityQueue;
import java.util.Set;
import lombok.NonNull;
public interface MatchedRuleHolder {
PriorityQueue<MatchedRule> getMatchedRuleList();
boolean isIgnored();
boolean isRemoved();
void setIgnored(boolean ignored);
void setRemoved(boolean ignored);
default boolean isApplied() {
return getMatchedRule().isApplied();
}
default Set<RedactionEntity> getReferences() {
return getMatchedRule().getReferences();
}
default boolean isActive() {
return !(isRemoved() || isIgnored());
}
default void apply(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
}
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).legalBasis(legalBasis).applied(true).build());
}
default void force(@NonNull String ruleIdentifier, String reason, String legalBasis) {
addMatchedRule(MatchedRule.builder()
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(getLegalBasisOrPreviousLegalBasisOrPlaceHolder(legalBasis))
.applied(true)
.build());
}
default void skip(@NonNull String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).build());
}
default void remove(String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).build());
setRemoved(true);
}
default void ignore(String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).build());
setIgnored(true);
}
private String getLegalBasisOrPreviousLegalBasisOrPlaceHolder(String legalBasis) {
if (legalBasis == null || legalBasis.isBlank() || legalBasis.isEmpty()) {
if (getMatchedRule() == null || !getMatchedRule().isApplied()) {
return "n-a";
}
return getMatchedRule().getLegalBasis();
}
return legalBasis;
}
default void applyWithLineBreaks(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
}
getMatchedRuleList().add(MatchedRule.builder()
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.writeValueWithLineBreaks(true)
.build());
}
default void applyWithReferences(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis, Collection<RedactionEntity> references) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
}
getMatchedRuleList().add(MatchedRule.builder()
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.references(new HashSet<>(references))
.build());
}
default void skipWithReferences(@NonNull String ruleIdentifier, String reason, Collection<RedactionEntity> references) {
getMatchedRuleList().add(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).references(new HashSet<>(references)).build());
}
default void addMatchedRule(MatchedRule matchedRule) {
getMatchedRuleList().add(matchedRule);
}
default void addMatchedRules(Collection<MatchedRule> matchedRules) {
getMatchedRuleList().addAll(matchedRules);
}
default int getMatchedRuleUnit() {
return getMatchedRule().getRuleIdentifier().unit();
}
default MatchedRule getMatchedRule() {
if (getMatchedRuleList().isEmpty()) {
return MatchedRule.empty();
}
return getMatchedRuleList().peek();
}
}

Some files were not shown because too many files have changed in this diff Show More