Compare commits

...
Author SHA1 Message Date
Maverick Studer af8a7dac42 Merge branch 'RED-9104' into 'master'
RED-9104: Rectangle redaction cannot be removed

Closes RED-9104

See merge request redactmanager/redaction-service!390
2024-05-03 13:36:32 +02:00
Maverick Studer 7746c63063 RED-9104: Rectangle redaction cannot be removed 2024-05-03 13:36:32 +02:00
Maverick Studer b871b5bb08 Merge branch 'RED-9091-fp' into 'master'
RED-9091: Cannot re-add manual redaction on same position with same reason

Closes RED-9091

See merge request redactmanager/redaction-service!389
2024-05-03 10:03:11 +02:00
Maverick Studer 2ae9498954 RED-9091: Cannot re-add manual redaction on same position with same reason 2024-05-03 10:03:11 +02:00
Kilian Schüttler 9565104585 Merge branch 'RED-9042-fp' into 'master'
RED-9042: merge recategorize and legal basis

Closes RED-9042

See merge request redactmanager/redaction-service!387
2024-04-26 12:38:54 +02:00
Kilian Schüttler 4450d0738f RED-9042: merge recategorize and legal basis 2024-04-26 12:38:54 +02:00
Yannik Hampe d1e962861b Merge branch 'RED-8701' into 'master'
RED-8701 - Move files to customer data repositories

Closes RED-8701

See merge request redactmanager/redaction-service!383
2024-04-25 09:07:55 +02:00
Corina Olariu 5f637dcbca RED-8701 - Move files to customer data repositories 2024-04-25 09:07:55 +02:00
Dominique Eifländer dce2d1b898 Merge branch 'RED-7384-fp' into 'master'
RED-7384: improve performance significantly

Closes RED-7384

See merge request redactmanager/redaction-service!385
2024-04-24 11:57:04 +02:00
Kilian Schüttler 13b734de37 RED-7384: improve performance significantly 2024-04-24 11:57:03 +02:00
Dominique Eifländer ac4f51652f Merge branch 'RED-8826' into 'master'
RED-8826: Added new ImageType.GRAPHIC

Closes RED-8826

See merge request redactmanager/redaction-service!382
2024-04-23 13:13:04 +02:00
Dominique Eifländer 30c53d56c3 RED-8826: Added new ImageType.GRAPHIC 2024-04-23 09:45:09 +02:00
Kevin Tumma 47d9ec8a29 Update .gitlab-ci.yml file 2024-04-18 15:42:30 +02:00
Kevin Tumma 7b06cab73b Update .gitlab-ci.yml file 2024-04-18 15:38:01 +02:00
Andrei Isvoran c3cd6f9c1b Merge branch 'RED-8958' into 'master'
RED-8959 - UT for local resize on dictionary entity

Closes RED-8958

See merge request redactmanager/redaction-service!380
2024-04-18 10:14:20 +02:00
Andrei Isvoran 4e6769a7f9 RED-8959 - UT for local resize on dictionary entity 2024-04-18 10:14:19 +02:00
Andrei Isvoran 0e73512d9a Merge branch 'RED-8650' into 'master'
RED-8650 - Support more date formats

Closes RED-8650

See merge request redactmanager/redaction-service!379
2024-04-18 09:45:01 +02:00
Andrei Isvoran dda8c98b87 RED-8650 - Support more date formats 2024-04-18 09:55:42 +03:00
Yannik Hampe 1af74c3f2e Merge branch 'RED-8402' into 'master'
RED-8402: hotfix for fixing exception caused by layout parser changes

Closes RED-8402

See merge request redactmanager/redaction-service!376
2024-04-17 12:50:27 +02:00
Yannik Hampe 231f0bccd8 RED-8402: hotfix for fixing exception caused by layout parser changes 2024-04-17 12:50:27 +02:00
Maverick Studer 5a91b01f92 Merge branch 'RED-8971' into 'master'
RED-8971: Changing Vertebrate Study to 'Yes' has no effect

Closes RED-8971

See merge request redactmanager/redaction-service!375
2024-04-17 09:09:53 +02:00
Maverick Studer bda756c7fe RED-8971: Changing Vertebrate Study to 'Yes' has no effect 2024-04-17 09:09:53 +02:00
Andrei Isvoran 65fe05be79 Merge branch 'RED-8650-fix' into 'master'
RED-8650 - Fix date_formats path

Closes RED-8650

See merge request redactmanager/redaction-service!374
2024-04-16 16:19:32 +02:00
Andrei Isvoran 7dc56551e3 RED-8650 - Fix date_formats path 2024-04-16 16:19:32 +02:00
Andrei Isvoran 167f27138d Merge branch 'RED-8650' into 'master'
RED-8650 - Add support for more date formats

Closes RED-8650

See merge request redactmanager/redaction-service!373
2024-04-15 14:55:12 +02:00
Andrei Isvoran 7ec734e160 RED-8650 - Add support for more date formats 2024-04-15 14:55:12 +02:00
Kilian Schüttler c8b1eb31b7 Merge branch 'RED-8610' into 'master'
RED-8610: dossier-dictionary entities should only have the engine DOSSIER_DICTIONARY

Closes RED-8610

See merge request redactmanager/redaction-service!372
2024-04-15 14:31:50 +02:00
Kilian Schuettler 70d568f866 RED-8610: dossier-dictionary entities should only have the engine DOSSIER_DICTIONARY 2024-04-15 10:52:19 +02:00
Maverick Studer 5df3859cd3 Merge branch 'RED-8702-fix' into 'master'
RED-8702: Explore document databases to store entityLog

Closes RED-8702

See merge request redactmanager/redaction-service!371
2024-04-12 13:23:31 +02:00
Maverick Studer 352e8a7659 RED-8702: Explore document databases to store entityLog 2024-04-12 13:23:31 +02:00
Maverick Studer 1ac31e083d Merge branch 'RED-8702-fix' into 'master'
RED-8702: Explore document databases to store entityLog

Closes RED-8702

See merge request redactmanager/redaction-service!370
2024-04-11 16:42:38 +02:00
Maverick Studer 033fe17482 RED-8702: Explore document databases to store entityLog 2024-04-11 16:42:36 +02:00
Andrei Isvoran e9043c930a Merge branch 'RED-8694' into 'master'
RED-8694 - Add Javadoc to classes/methods used in rules

Closes RED-8694

See merge request redactmanager/redaction-service!369
2024-04-11 14:52:33 +02:00
Andrei Isvoran 93250d5463 RED-8694 - Add Javadoc to classes/methods used in rules 2024-04-11 14:52:33 +02:00
Kilian Schüttler 5ca3ad521e Merge branch 'RED-7384-fp' into 'master'
RED-7384: fix imported stuff

Closes RED-7384

See merge request redactmanager/redaction-service!368
2024-04-11 14:13:49 +02:00
Kilian Schüttler 7deb0b924d RED-7384: fix imported stuff 2024-04-11 14:13:48 +02:00
Kilian Schüttler 8775545703 Merge branch 'RED-8690' into 'master'
RED-8690: Overlapping SKIPPED and APPLIED of same type

Closes RED-8690

See merge request redactmanager/redaction-service!366
2024-04-10 14:52:04 +02:00
Kilian Schüttler ce596adc7e RED-8690: Overlapping SKIPPED and APPLIED of same type 2024-04-10 14:52:04 +02:00
Kilian Schüttler a16ffb3b95 Merge branch 'RED-8905' into 'master'
RED-8905: DM: File in error state when using getPreviousSibling()

Closes RED-8905

See merge request redactmanager/redaction-service!362
2024-04-04 15:41:07 +02:00
Kilian Schuettler 5e0060864f RED-8905: DM: File in error state when using getPreviousSibling() 2024-04-04 15:23:54 +02:00
Maverick Studer 33862237b1 Merge branch 'RED-8702-backup2' into 'master'
RED-8702: Explore document databases to store entityLog

Closes RED-8702

See merge request redactmanager/redaction-service!359
2024-04-03 17:25:56 +02:00
Maverick Studer dbc5d9ab43 RED-8702: Explore document databases to store entityLog 2024-04-03 17:25:56 +02:00
Kilian Schüttler b80914288b Merge branch 'RED-7384-fp' into 'master'
RED-7384: handle pending dict application in redaction-service instead of persistence

Closes RED-7384

See merge request redactmanager/redaction-service!357
2024-04-03 16:32:08 +02:00
Kilian Schüttler df694a90a8 RED-7384: handle pending dict application in redaction-service instead of persistence 2024-04-03 16:32:08 +02:00
Andrei Isvoran 49b91955b8 Merge branch 'RED-8877' into 'master'
RED-8877 - Remove X.1.0 rule

Closes RED-8877

See merge request redactmanager/redaction-service!361
2024-04-03 15:33:10 +02:00
Andrei Isvoran 04ec3cb7b9 RED-8877 - Remove X.1.0 rule 2024-04-03 15:33:10 +02:00
Kilian Schüttler fcae77dcd4 Merge branch 'RED-8773' into 'master'
RED-8773 - Wrong value for recategorized and forced logo

Closes RED-8773

See merge request redactmanager/redaction-service!329
2024-04-03 15:18:06 +02:00
Corina Olariu 13e60e05b6 RED-8773 - Wrong value for recategorized and forced logo 2024-04-03 15:18:06 +02:00
Andrei Isvoran 90ba3239fc Merge branch 'RED-8775' into 'master'
RED-8775 - Add X.11 remove rule

Closes RED-8775

See merge request redactmanager/redaction-service!340
2024-04-02 08:33:54 +02:00
Andrei Isvoran 6de0346052 RED-8775 - Add X.11 remove rule 2024-04-02 08:33:54 +02:00
Andrei Isvoran 7b520ee0ab Merge branch 'RED-8828-fix-npe' into 'master'
RED-8828 - Fix error when resizing dict based redaction

Closes RED-8828

See merge request redactmanager/redaction-service!353
2024-04-02 08:33:43 +02:00
Andrei Isvoran bf638dac01 RED-8828 - Fix error when resizing dict based redaction 2024-03-29 14:30:55 +02:00
Kilian Schüttler 9354cbb6bc Merge branch 'hotfix_removechain' into 'master'
Hotfix removechain

See merge request redactmanager/redaction-service!349
2024-03-27 16:27:09 +01:00
Kilian Schüttler 406f5a48cf Hotfix removechain 2024-03-27 16:27:09 +01:00
Christoph Schabert a71ee61e1b Update .gitlab-ci.yml 2024-03-27 15:41:37 +01:00
Andrei Isvoran 096544d8e2 Merge branch 'RED-8840-rules' into 'master'
RED-8840 - Adjust rules

Closes RED-8840

See merge request redactmanager/redaction-service!351
2024-03-27 15:33:49 +01:00
Andrei Isvoran 646f2020a8 RED-8840 - Adjust rules 2024-03-27 16:14:47 +02:00
Ali Oezyetimoglu 72a7761dd8 Merge branch 'RED-8480-B' into 'master'
RED-8480: addded property "value" to places with recategorizations

Closes RED-8480

See merge request redactmanager/redaction-service!348
2024-03-27 11:24:32 +01:00
Ali Oezyetimoglu c869a624e6 RED-8480: addded property "value" to places with recategorizations 2024-03-27 11:10:59 +01:00
Dominique Eifländer 3ca37133c6 Merge branch 'RED-8834' into 'master'
RED-8834: Fixed text entities with empty text range

Closes RED-8834

See merge request redactmanager/redaction-service!345
2024-03-27 09:59:33 +01:00
Dominique Eifländer 93baf97ce3 RED-8834: Fixed text entities with empty text range 2024-03-27 09:47:56 +01:00
Andrei Isvoran a0773d16dc Merge branch 'hotfix-get-dictionary-npe' into 'master'
Hotfix NPE on getDeepCopyDictionary

See merge request redactmanager/redaction-service!341
2024-03-26 10:36:02 +01:00
Andrei Isvoran c4816579c9 Hotfix NPE on getDeepCopyDictionary 2024-03-26 11:23:14 +02:00
Ali Oezyetimoglu c9e80e3fa1 Merge branch 'RED-8840' into 'master'
RED-8840 - Add PII.4 rules for sanitisation with correct legal basis

Closes RED-8840

See merge request redactmanager/redaction-service!337
2024-03-25 16:33:07 +01:00
Andrei Isvoran 677d62deca RED-8840 - Add PII.4 rules for sanitisation with correct legal basis 2024-03-25 16:33:07 +01:00
Kilian Schüttler 5229cd87f6 Merge branch 'RED-8610' into 'master'
RED-8610: Dictionary remove on dossier level should be displayed as skipped

Closes RED-8610

See merge request redactmanager/redaction-service!338
2024-03-25 14:13:44 +01:00
Kilian Schüttler 50e137575d RED-8610: Dictionary remove on dossier level should be displayed as skipped 2024-03-25 14:13:44 +01:00
Kilian Schüttler 7303a7579b Merge branch 'RED-8610' into 'master'
RED-8610: Dictionary remove on dossier level should be displayed as skipped

Closes RED-8610

See merge request redactmanager/redaction-service!335
2024-03-22 10:19:59 +01:00
Kilian Schüttler 2b4f174bdf RED-8610: Dictionary remove on dossier level should be displayed as skipped 2024-03-22 10:19:59 +01:00
Ali Oezyetimoglu d36d2ea2fa Merge branch 'RED-8480-B' into 'master'
RED-8480: updated code according to changes from ManualRecategorization

Closes RED-8480

See merge request redactmanager/redaction-service!332
2024-03-21 11:43:44 +01:00
Ali Oezyetimoglu b1c618505e RED-8480: updated code according to changes from ManualRecategorization 2024-03-20 18:44:20 +01:00
Andrei Isvoran 983f728248 Merge branch 'RED-8784' into 'master'
RED-8784 - Change PII.9.1/PII.9.2 to CBI.23.0/CBI.23.1

Closes RED-8784

See merge request redactmanager/redaction-service!326
2024-03-19 12:32:28 +01:00
Andrei Isvoran fe07e500b9 RED-8784 - Change PII.9.1/PII.9.2 to CBI.23.0/CBI.23.1 2024-03-19 12:32:28 +01:00
Dominique Eifländer 25015f633e Merge branch 'RED-7141' into 'master'
RED-7141: Fixed overlaps in duplicated and non duplicated blocks

Closes RED-7141

See merge request redactmanager/redaction-service!327
2024-03-18 14:37:52 +01:00
Dominique Eifländer 9b29d6f8f1 RED-7141: Fixed overlaps in duplicated and non duplicated blocks 2024-03-18 14:06:08 +01:00
Dominique Eifländer 16963d64ee Merge branch 'RED-7384' into 'master'
RED-7384: add useful fields to ManualRedactionEntry

Closes RED-7384

See merge request redactmanager/redaction-service!314
2024-03-18 13:17:15 +01:00
Kilian Schüttler a7674f406d RED-7384: add useful fields to ManualRedactionEntry 2024-03-18 13:17:15 +01:00
Andrei Isvoran 415daee3b6 Merge branch 'RED-8680-seeds' into 'master'
RED-8680 - Add specific CBI rules for seeds

Closes RED-8680

See merge request redactmanager/redaction-service!324
2024-03-18 10:58:18 +01:00
Andrei Isvoran d0a0bbc627 RED-8680 - Add specific CBI rules for seeds 2024-03-15 16:06:42 +02:00
Andrei Isvoran 93299c4e5f Merge branch 'RED-8705' into 'master'
RED-8705 - Fix image type

Closes RED-8705

See merge request redactmanager/redaction-service!321
2024-03-15 09:06:55 +01:00
Andrei Isvoran 72f8885ff0 RED-8705 - Fix image type 2024-03-14 17:15:14 +02:00
Andrei Isvoran cd6724b952 Merge branch 'RED-8680-seeds' into 'master'
RED-8680 - Add rules for syngenta sanitisation seeds

Closes RED-8680

See merge request redactmanager/redaction-service!320
2024-03-14 12:36:35 +01:00
Andrei Isvoran 3405cc6c1b RED-8680 - Add rules for syngenta sanitisation seeds 2024-03-14 10:59:37 +02:00
Andrei Isvoran 53ac78f788 Merge branch 'RED-8645-more-fixes-4.1' into 'master'
RED-8645 - Fix some more rules

Closes RED-8645

See merge request redactmanager/redaction-service!318
2024-03-13 08:48:01 +01:00
Andrei Isvoran 41821ff5ca RED-8645 - Fix some more rules 2024-03-12 15:45:38 +02:00
Dominique Eifländer f59dd7ff65 Merge branch 'RED-7384-4.1-2' into 'master'
RED-7384: Fixed missing requestDate in new created manualRedactions for resizes

Closes RED-7384

See merge request redactmanager/redaction-service!316
2024-03-12 12:12:51 +01:00
Dominique Eifländer 4e5718ba23 RED-7384: Fixed missing requestDate in new created manualRedactions for resizes 2024-03-12 12:01:13 +01:00
Dominique Eifländer e79bcbbc10 Merge branch 'RED-7384-4.1' into 'master'
RED-7384: Fixed migration problem for a specific file

Closes RED-7384

See merge request redactmanager/redaction-service!315
2024-03-12 09:48:40 +01:00
Dominique Eifländer d531f1be7f RED-7384: Fixed migration problem for a specific file 2024-03-12 09:35:53 +01:00
Dominique Eifländer 812946c81d Merge branch 'RED-7141' into 'master'
RED-7141: Adapted to layout parser using docstrum

Closes RED-7141

See merge request redactmanager/redaction-service!312
2024-03-08 15:08:23 +01:00
Dominique Eifländer b93f9a2c20 RED-7141: Adapted to layout parser using docstrum 2024-03-08 14:54:48 +01:00
Andrei Isvoran fa55917a89 Merge branch 'RED-8645-rules' into 'master'
RED-8645 - Update RM rules

Closes RED-8645

See merge request redactmanager/redaction-service!309
2024-03-08 14:34:07 +01:00
Andrei Isvoran d00847e955 RED-8645 - Update RM rules 2024-03-08 14:34:07 +01:00
Yannik Hampe 80cd2f2eda Merge branch 'RED-8515' into 'master'
proof of concept: use git lfs to store customer files

Closes RED-8515

See merge request redactmanager/redaction-service!304
2024-03-08 12:43:29 +01:00
Yannik Hampe d7f3f351ec proof of concept: use git lfs to store customer files 2024-03-08 12:43:29 +01:00
Maverick Studer b5d993561a Merge branch 'RED-7700' into 'master'
RED-7700: Safe rule execution

Closes RED-7700

See merge request redactmanager/redaction-service!311
2024-03-07 14:38:14 +01:00
Maverick Studer aad890b893 RED-7700: Safe rule execution 2024-03-07 14:38:13 +01:00
Kilian Schüttler c39c98b955 Merge branch 'image-name' into 'master'
revert image name change

See merge request redactmanager/redaction-service!306
2024-03-05 18:32:19 +01:00
Kilian Schuettler 620c315867 revert image name change 2024-03-05 16:30:58 +01:00
Andrei Isvoran 58fb062a79 Merge branch 'RED-8632' into 'master'
RED-8632 - Generate javadoc automatically

Closes RED-8632

See merge request redactmanager/redaction-service!305
2024-03-05 13:49:08 +01:00
Andrei Isvoran 9dfb0f49c6 RED-8632 - Generate javadoc automatically 2024-03-05 13:49:08 +01:00
Kilian Schüttler 7e144e30bf Merge branch 'reformat' into 'master'
reformat

See merge request redactmanager/redaction-service!303
2024-03-01 15:49:37 +01:00
Kilian Schuettler 9f8b1134b9 reformat 2024-03-01 15:41:24 +01:00
Andrei Isvoran 313df1e118 Merge branch 'RED-8586-dossier-redactions' into 'master'
RED-8586 - Don't treat dossier redactions differently

Closes RED-8586

See merge request redactmanager/redaction-service!302
2024-03-01 12:41:55 +01:00
Andrei Isvoran 24c1be66fd RED-8586 - Don't treat dossier redactions differently 2024-03-01 12:41:55 +01:00
Andrei Isvoran 0825686741 Merge branch 'RED-8586-fix' into 'master'
RED-8586 - Add higher salience to rule ETC.5.1

Closes RED-8586

See merge request redactmanager/redaction-service!299
2024-02-29 15:42:39 +01:00
Andrei Isvoran fd10d6e325 RED-8586 - Add higher salience to rule ETC.5.1 2024-02-29 16:28:33 +02:00
391 changed files with 11683 additions and 1928827 deletions
+32 -7
View File
@@ -1,23 +1,48 @@
variables:
SONAR_PROJECT_KEY: 'RED_redaction-service'
GIT_SUBMODULE_STRATEGY: recursive
GIT_SUBMODULE_FORCE_HTTPS: "true"
include:
- project: 'gitlab/gitlab'
ref: 'main'
file: 'ci-templates/gradle_java.yml'
deploy:
deploy JavaDoc:
stage: deploy
tags:
- dind
script:
- echo "Building with gradle version ${BUILDVERSION}"
- echo "Building JavaDoc with gradle version ${BUILDVERSION}"
- gradle -Pversion=${BUILDVERSION} publish
- gradle bootBuildImage --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
- echo "BUILDVERSION=$BUILDVERSION" >> version.env
artifacts:
reports:
dotenv: version.env
rules:
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
- if: $CI_COMMIT_BRANCH =~ /^release/
- if: $CI_COMMIT_TAG
generateJavaDoc:
stage: build
tags:
- dind
script:
- echo "Generating Javadoc..."
- gradle generateJavaDoc -PjavadocDestinationDir="javadoc"
artifacts:
paths:
- redaction-service-v1/redaction-service-server-v1/javadoc/*
rules:
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
- if: $CI_COMMIT_BRANCH =~ /^release/
- if: $CI_COMMIT_TAG
pages:
stage: build
needs:
- generateJavaDoc
script:
- mkdir public
- mv redaction-service-v1/redaction-service-server-v1/javadoc/* public/
rules:
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
artifacts:
paths:
- public
+8
View File
@@ -0,0 +1,8 @@
[submodule "redaction-service-v1/redaction-service-server-v1/src/test/resources/files/syngenta"]
path = redaction-service-v1/redaction-service-server-v1/src/test/resources/files/syngenta
url = ssh://git@git.knecon.com:22222/fforesight/documents/syngenta.git
update = merge
[submodule "redaction-service-v1/redaction-service-server-v1/src/test/resources/files/basf"]
path = redaction-service-v1/redaction-service-server-v1/src/test/resources/files/basf
url = ssh://git@git.knecon.com:22222/fforesight/documents/basf.git
update = merge
@@ -7,7 +7,7 @@ description = "redaction-service-api-v1"
dependencies {
implementation("org.springframework:spring-web:6.0.12")
implementation("com.iqser.red.service:persistence-service-internal-api-v1:2.351.0")
implementation("com.iqser.red.service:persistence-service-internal-api-v1:2.410.0")
}
publishing {
@@ -15,4 +15,5 @@ public class AnalyzeResponse {
private String fileId;
private List<UnprocessedManualEntity> unprocessedManualEntities;
}
@@ -0,0 +1,27 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.List;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.experimental.SuperBuilder;
@Data
@SuperBuilder
@AllArgsConstructor
@NoArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
@EqualsAndHashCode(callSuper = true)
public class DroolsBlacklistErrorMessage extends DroolsValidationMessage {
List<String> blacklistedKeywords;
public String getMessage() {
return String.format("Blacklisted keywords found in this rule: %s", String.join(", ", blacklistedKeywords));
}
}
@@ -4,17 +4,19 @@ import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.experimental.SuperBuilder;
@Data
@Builder
@SuperBuilder
@AllArgsConstructor
@NoArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DroolsSyntaxDeprecatedWarnings {
@EqualsAndHashCode(callSuper = true)
public class DroolsSyntaxDeprecatedWarnings extends DroolsValidationMessage {
Integer line;
Integer column;
String message;
}
@@ -4,17 +4,19 @@ import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.experimental.SuperBuilder;
@Data
@Builder
@SuperBuilder
@AllArgsConstructor
@NoArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DroolsSyntaxErrorMessage {
@EqualsAndHashCode(callSuper = true)
public class DroolsSyntaxErrorMessage extends DroolsValidationMessage {
Integer line;
Integer column;
String message;
}
@@ -1,36 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.LinkedList;
import java.util.List;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DroolsSyntaxValidation {
@Builder.Default
List<DroolsSyntaxErrorMessage> droolsSyntaxErrorMessages = new LinkedList<>();
@Builder.Default
List<DroolsSyntaxDeprecatedWarnings> droolsSyntaxDeprecatedWarnings = new LinkedList<>();
public void addErrorMessage(int line, int column, String message) {
getDroolsSyntaxErrorMessages().add(DroolsSyntaxErrorMessage.builder().line(line).column(column).message(message).build());
}
public boolean isCompiled() {
return droolsSyntaxErrorMessages.isEmpty();
}
}
@@ -0,0 +1,39 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
public class DroolsValidation {
@Builder.Default
List<DroolsSyntaxErrorMessage> syntaxErrorMessages = new ArrayList<>();
@Builder.Default
List<DroolsSyntaxDeprecatedWarnings> deprecatedWarnings = new ArrayList<>();
@Builder.Default
List<DroolsBlacklistErrorMessage> blacklistErrorMessages = new ArrayList<>();
public void addErrorMessage(int line, int column, String message) {
getSyntaxErrorMessages().add(DroolsSyntaxErrorMessage.builder().line(line).column(column).message(message).build());
}
public boolean isCompiled() {
return syntaxErrorMessages.isEmpty() && blacklistErrorMessages.isEmpty();
}
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.experimental.SuperBuilder;
@Data
@SuperBuilder
@AllArgsConstructor
@NoArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
@EqualsAndHashCode
public class DroolsValidationMessage {
Integer line;
Integer column;
}
@@ -1,11 +1,15 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.Collections;
import java.util.Set;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.NonNull;
@Data
@Builder
@@ -13,9 +17,18 @@ import lombok.NoArgsConstructor;
@AllArgsConstructor
public class MigrationRequest {
@NonNull
String dossierTemplateId;
@NonNull
String dossierId;
@NonNull
String fileId;
boolean fileIsApproved;
@NonNull
ManualRedactions manualRedactions;
@NonNull
@Builder.Default
Set<String> entitiesWithComments = Collections.emptySet();
}
@@ -24,4 +24,5 @@ public class UnprocessedManualEntity {
private String section;
@Builder.Default
private List<Position> positions = new ArrayList<>();
}
@@ -4,12 +4,12 @@ import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
public interface RedactionResource {
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
DroolsSyntaxValidation testRules(@RequestBody RuleValidationModel rulesValidationModel);
DroolsValidation testRules(@RequestBody RuleValidationModel rulesValidationModel);
}
@@ -12,12 +12,14 @@ plugins {
description = "redaction-service-server-v1"
val layoutParserVersion = "0.94.0"
val layoutParserVersion = "0.116.0"
val jacksonVersion = "2.15.2"
val droolsVersion = "9.44.0.Final"
val pdfBoxVersion = "3.0.0"
val persistenceServiceVersion = "2.360.0"
val persistenceServiceVersion = "2.410.0"
val springBootStarterVersion = "3.1.5"
val springCloudVersion = "4.0.4"
val testContainersVersion = "1.19.7"
configurations {
all {
@@ -31,6 +33,7 @@ dependencies {
implementation(project(":redaction-service-api-v1")) { exclude(group = "com.iqser.red.service", module = "persistence-service-internal-api-v1") }
implementation("com.iqser.red.service:persistence-service-internal-api-v1:${persistenceServiceVersion}") { exclude(group = "org.springframework.boot") }
implementation("com.iqser.red.service:persistence-service-shared-mongo-v1:${persistenceServiceVersion}")
implementation("com.knecon.fforesight:layoutparser-service-internal-api:${layoutParserVersion}")
implementation("com.iqser.red.commons:spring-commons:6.2.0")
@@ -38,7 +41,7 @@ dependencies {
implementation("com.iqser.red.commons:dictionary-merge-commons:1.5.0")
implementation("com.iqser.red.commons:storage-commons:2.45.0")
implementation("com.knecon.fforesight:tenant-commons:0.21.0")
implementation("com.knecon.fforesight:tenant-commons:0.24.0")
implementation("com.knecon.fforesight:tracing-commons:0.5.0")
implementation("com.fasterxml.jackson.module:jackson-module-afterburner:${jacksonVersion}")
@@ -52,7 +55,7 @@ dependencies {
implementation("org.locationtech.jts:jts-core:1.19.0")
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:4.0.4")
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:${springCloudVersion}")
implementation("org.springframework.boot:spring-boot-starter-amqp:${springBootStarterVersion}")
implementation("org.springframework.boot:spring-boot-starter-cache:${springBootStarterVersion}")
implementation("org.springframework.boot:spring-boot-starter-data-redis:${springBootStarterVersion}")
@@ -66,6 +69,9 @@ dependencies {
testImplementation("org.apache.pdfbox:pdfbox:${pdfBoxVersion}")
testImplementation("org.apache.pdfbox:pdfbox-tools:${pdfBoxVersion}")
testImplementation("org.testcontainers:testcontainers:${testContainersVersion}")
testImplementation("org.testcontainers:junit-jupiter:${testContainersVersion}")
testImplementation("org.springframework.boot:spring-boot-starter-test:${springBootStarterVersion}")
testImplementation("com.knecon.fforesight:viewer-doc-processor:${layoutParserVersion}")
testImplementation("com.knecon.fforesight:layoutparser-service-processor:${layoutParserVersion}") {
@@ -76,6 +82,12 @@ dependencies {
}
}
dependencyManagement {
imports {
mavenBom("org.testcontainers:testcontainers-bom:${testContainersVersion}")
}
}
tasks.test {
configure<JacocoTaskExtension> {
excludes = listOf("org/drools/**/*")
@@ -113,3 +125,42 @@ tasks.named<BootBuildImage>("bootBuildImage") {
tags.set(listOf(dockerTag))
}
}
fun parseDroolsImports(droolsFilePath: String): List<String> {
val imports = mutableListOf<String>()
val importPattern = Regex("^import\\s+(com\\.iqser\\.red\\.service\\.redaction\\.v1\\.[\\w.]+);")
val desiredPrefix = "com.iqser.red.service.redaction.v1"
File(droolsFilePath).forEachLine { line ->
importPattern.find(line)?.let { matchResult ->
val importPath = matchResult.groupValues[1].trim()
if (importPath.startsWith(desiredPrefix)) {
val formattedPath = importPath.replace('.', '/')
imports.add("$formattedPath.java")
}
}
}
return imports
}
val droolsImports = parseDroolsImports("redaction-service-v1/redaction-service-server-v1/src/main/resources/drools/all_rules_documine.drl")
tasks.register("generateJavaDoc", Javadoc::class) {
dependsOn("compileJava")
dependsOn("delombok")
classpath = project.sourceSets["main"].runtimeClasspath
source = fileTree("${buildDir}/generated/sources/delombok/java/main") {
include(droolsImports)
}
destinationDir = file(project.findProperty("javadocDestinationDir")?.toString() ?: "")
options.memberLevel = JavadocMemberLevel.PUBLIC
(options as StandardJavadocDocletOptions).apply {
header = "Redaction Service ${project.version}"
footer = "Redaction Service ${project.version}"
title = "API Documentation for Redaction Service ${project.version}"
}
}
@@ -4,16 +4,24 @@ import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.ImportAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.boot.autoconfigure.data.mongo.MongoDataAutoConfiguration;
import org.springframework.boot.autoconfigure.jdbc.DataSourceAutoConfiguration;
import org.springframework.boot.autoconfigure.liquibase.LiquibaseAutoConfiguration;
import org.springframework.boot.autoconfigure.mongo.MongoAutoConfiguration;
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
import org.springframework.boot.context.properties.EnableConfigurationProperties;
import org.springframework.cache.annotation.EnableCaching;
import org.springframework.cloud.openfeign.EnableFeignClients;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Import;
import org.springframework.data.mongodb.repository.config.EnableMongoRepositories;
import com.iqser.red.service.dictionarymerge.commons.DictionaryMergeService;
import com.iqser.red.service.persistence.service.v1.api.shared.mongo.SharedMongoAutoConfiguration;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.storage.commons.StorageAutoConfiguration;
import com.knecon.fforesight.mongo.database.commons.MongoDatabaseCommonsAutoConfiguration;
import com.knecon.fforesight.mongo.database.commons.liquibase.EnableMongoLiquibase;
import com.knecon.fforesight.tenantcommons.MultiTenancyAutoConfiguration;
import io.micrometer.core.aop.TimedAspect;
@@ -22,11 +30,13 @@ import io.micrometer.observation.ObservationRegistry;
import io.micrometer.observation.aop.ObservedAspect;
@EnableCaching
@ImportAutoConfiguration({MultiTenancyAutoConfiguration.class})
@Import({MetricsConfiguration.class, StorageAutoConfiguration.class})
@ImportAutoConfiguration({MultiTenancyAutoConfiguration.class, SharedMongoAutoConfiguration.class})
@Import({MetricsConfiguration.class, StorageAutoConfiguration.class, MongoDatabaseCommonsAutoConfiguration.class})
@EnableFeignClients(basePackageClasses = RulesClient.class)
@EnableConfigurationProperties(RedactionServiceSettings.class)
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
@EnableMongoRepositories(basePackages = "com.iqser.red.service.persistence")
@EnableMongoLiquibase
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class, DataSourceAutoConfiguration.class, LiquibaseAutoConfiguration.class, MongoAutoConfiguration.class, MongoDataAutoConfiguration.class})
public class Application {
public static void main(String[] args) {
@@ -35,11 +45,14 @@ public class Application {
SpringApplication.run(Application.class, args);
}
@Bean
public ObservedAspect observedAspect(ObservationRegistry observationRegistry) {
return new ObservedAspect(observationRegistry);
}
@Bean
public TimedAspect timedAspect(MeterRegistry registry) {
@@ -95,6 +95,7 @@ public class DeprecatedElementsFinder {
return this.deprecatedClasses;
}
private String getMethodSignature(Method method) {
String methodName = method.getName();
@@ -30,4 +30,6 @@ public class RedactionServiceSettings {
private int droolsExecutionTimeoutSecs = 300;
private boolean ruleExecutionSecured = true;
}
@@ -16,7 +16,7 @@ public class RedisCachingConfiguration {
public RedisCacheManagerBuilderCustomizer redisCacheManagerBuilderCustomizer() {
return (builder) -> builder.withCacheConfiguration("documentDataCache",
RedisCacheConfiguration.defaultCacheConfig().entryTtl(Duration.ofMinutes(30)).disableCachingNullValues());
RedisCacheConfiguration.defaultCacheConfig().entryTtl(Duration.ofMinutes(30)).disableCachingNullValues());
}
@@ -1,6 +1,5 @@
package com.iqser.red.service.redaction.v1.server.client.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
@@ -3,10 +3,10 @@ package com.iqser.red.service.redaction.v1.server.controller;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.service.drools.DroolsSyntaxValidationService;
import com.iqser.red.service.redaction.v1.server.service.drools.DroolsValidationService;
import com.iqser.red.service.redaction.v1.server.utils.exception.RulesValidationException;
import lombok.RequiredArgsConstructor;
@@ -17,14 +17,14 @@ import lombok.extern.slf4j.Slf4j;
@RequiredArgsConstructor
public class RedactionController implements RedactionResource {
private final DroolsSyntaxValidationService droolsSyntaxValidationService;
private final DroolsValidationService droolsValidationService;
@Override
public DroolsSyntaxValidation testRules(@RequestBody RuleValidationModel rulesValidationModel) {
public DroolsValidation testRules(@RequestBody RuleValidationModel rulesValidationModel) {
try {
return droolsSyntaxValidationService.testRules(rulesValidationModel);
return droolsValidationService.testRules(rulesValidationModel);
} catch (Exception e) {
throw new RulesValidationException("Could not test rules: " + e.getMessage(), e);
}
@@ -13,7 +13,6 @@ import lombok.RequiredArgsConstructor;
@RequiredArgsConstructor
public class RuleBuilderController implements RuleBuilderResource {
@Override
public RuleBuilderModel getRuleBuilderModel() {
@@ -92,11 +92,16 @@ public class LegacyRedactionLogMergeService {
return redactionLog;
}
public long getNumberOfAffectedAnnotations(ManualRedactions manualRedactions) {
return createManualRedactionWrappers(manualRedactions).stream().map(ManualRedactionWrapper::getId).distinct().count();
return createManualRedactionWrappers(manualRedactions).stream()
.map(ManualRedactionWrapper::getId)
.distinct()
.count();
}
private List<ManualRedactionWrapper> createManualRedactionWrappers(ManualRedactions manualRedactions) {
List<ManualRedactionWrapper> manualRedactionWrappers = new ArrayList<>();
@@ -197,7 +202,12 @@ public class LegacyRedactionLogMergeService {
}
redactionLogEntry.getManualChanges()
.add(ManualChange.from(imageRecategorization).withManualRedactionType(ManualRedactionType.RECATEGORIZE).withChange("type", imageRecategorization.getType()));
.add(ManualChange.from(imageRecategorization)
.withManualRedactionType(ManualRedactionType.RECATEGORIZE)
.withChange("type", imageRecategorization.getType())
.withChange("section", imageRecategorization.getSection())
.withChange("legalBasis", imageRecategorization.getLegalBasis())
.withChange("value", imageRecategorization.getValue()));
}
@@ -21,7 +21,9 @@ public class LegacyVersion0MigrationService {
public RedactionLog mergeDuplicateAnnotationIds(RedactionLog redactionLog) {
List<RedactionLogEntry> mergedEntries = new LinkedList<>();
Map<String, List<RedactionLogEntry>> entriesById = redactionLog.getRedactionLogEntry().stream().collect(Collectors.groupingBy(RedactionLogEntry::getId));
Map<String, List<RedactionLogEntry>> entriesById = redactionLog.getRedactionLogEntry()
.stream()
.collect(Collectors.groupingBy(RedactionLogEntry::getId));
for (List<RedactionLogEntry> entries : entriesById.values()) {
if (entries.isEmpty()) {
@@ -33,7 +35,10 @@ public class LegacyVersion0MigrationService {
continue;
}
List<RedactionLogEntry> sortedEntries = entries.stream().sorted(Comparator.comparing(entry -> entry.getChanges().get(0).getDateTime())).toList();
List<RedactionLogEntry> sortedEntries = entries.stream()
.sorted(Comparator.comparing(entry -> entry.getChanges()
.get(0).getDateTime()))
.toList();
RedactionLogEntry initialEntry = sortedEntries.get(0);
for (RedactionLogEntry entry : sortedEntries.subList(1, sortedEntries.size())) {
@@ -1,9 +1,9 @@
package com.iqser.red.service.redaction.v1.server.migration;
import java.util.Collections;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ChangeType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualChange;
@@ -70,13 +70,21 @@ public class MigrationMapper {
public static Set<com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine> getMigratedEngines(RedactionLogEntry entry) {
if (entry.getEngines() == null) {
return Collections.emptySet();
Set<com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine> engines = new HashSet<>();
if (entry.isImported()) {
engines.add(com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine.IMPORTED);
}
return entry.getEngines()
if (entry.getEngines() == null) {
return engines;
}
entry.getEngines()
.stream()
.map(MigrationMapper::toEntityLogEngine)
.collect(Collectors.toSet());
.forEach(engines::add);
return engines;
}
@@ -57,7 +57,7 @@ public class MigrationMessageReceiver {
if (redactionLog.getAnalysisVersion() == 0) {
redactionLog = legacyVersion0MigrationService.mergeDuplicateAnnotationIds(redactionLog);
} else if (migrationRequest.getManualRedactions() != null) {
} else {
redactionLog = legacyRedactionLogMergeService.addManualAddEntriesAndRemoveSkippedImported(redactionLog,
migrationRequest.getManualRedactions(),
migrationRequest.getDossierTemplateId());
@@ -67,9 +67,12 @@ public class MigrationMessageReceiver {
document,
migrationRequest.getDossierTemplateId(),
migrationRequest.getManualRedactions(),
migrationRequest.getFileId());
migrationRequest.getFileId(),
migrationRequest.getEntitiesWithComments(),
migrationRequest.isFileIsApproved());
log.info("Storing migrated entityLog and ids to migrate in DB for file {}", migrationRequest.getFileId());
redactionStorageService.storeObject(migrationRequest.getDossierId(), migrationRequest.getFileId(), FileType.ENTITY_LOG, migratedEntityLog.getEntityLog());
redactionStorageService.storeObject(migrationRequest.getDossierId(), migrationRequest.getFileId(), FileType.MIGRATED_IDS, migratedEntityLog.getMigratedIds());
@@ -8,6 +8,7 @@ import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.Set;
import java.util.function.Function;
import java.util.stream.Collectors;
import java.util.stream.Stream;
@@ -19,7 +20,9 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.migration.MigratedIds;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.BaseAnnotation;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRedactionEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualResizeRedaction;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
@@ -34,7 +37,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.service.ManualChangesApplicationService;
import com.iqser.red.service.redaction.v1.server.service.document.EntityEnrichmentService;
import com.iqser.red.service.redaction.v1.server.service.document.EntityFindingUtility;
import com.iqser.red.service.redaction.v1.server.service.document.EntityFromPrecursorCreationService;
import com.iqser.red.service.redaction.v1.server.utils.IdBuilder;
@@ -54,12 +56,17 @@ public class RedactionLogToEntityLogMigrationService {
private static final double MATCH_THRESHOLD = 10;
EntityFindingUtility entityFindingUtility;
EntityEnrichmentService entityEnrichmentService;
DictionaryService dictionaryService;
ManualChangesApplicationService manualChangesApplicationService;
public MigratedEntityLog migrate(RedactionLog redactionLog, Document document, String dossierTemplateId, ManualRedactions manualRedactions, String fileId) {
public MigratedEntityLog migrate(RedactionLog redactionLog,
Document document,
String dossierTemplateId,
ManualRedactions manualRedactions,
String fileId,
Set<String> entitiesWithComments,
boolean fileIsApproved) {
log.info("Migrating entities for file {}", fileId);
List<MigrationEntity> entitiesToMigrate = calculateMigrationEntitiesFromRedactionLog(redactionLog, document, dossierTemplateId, fileId);
@@ -67,8 +74,8 @@ public class RedactionLogToEntityLogMigrationService {
MigratedIds migratedIds = entitiesToMigrate.stream()
.collect(new MigratedIdsCollector());
applyManualChanges(entitiesToMigrate, manualRedactions);
log.info("applying manual changes to migrated entities for file {}", fileId);
applyLocalProcessedManualChanges(entitiesToMigrate, manualRedactions, fileIsApproved);
EntityLog entityLog = new EntityLog();
entityLog.setAnalysisNumber(redactionLog.getAnalysisNumber());
@@ -89,7 +96,7 @@ public class RedactionLogToEntityLogMigrationService {
.map(migrationEntity -> migrationEntity.toEntityLogEntry(oldToNewIDMapping))
.toList());
if (getNumberOfApprovedEntries(redactionLog) != entityLog.getEntityLogEntry().size()) {
if (getNumberOfApprovedEntries(redactionLog, document.getNumberOfPages()) != entityLog.getEntityLogEntry().size()) {
String message = String.format("Not all entities have been found during the migration redactionLog has %d entries and new entityLog %d",
redactionLog.getRedactionLogEntry().size(),
entityLog.getEntityLogEntry().size());
@@ -97,8 +104,14 @@ public class RedactionLogToEntityLogMigrationService {
throw new AssertionError(message);
}
Set<String> entitiesWithUnprocessedChanges = manualRedactions.buildAll()
.stream()
.filter(manualRedaction -> manualRedaction.getProcessedDate() == null)
.map(BaseAnnotation::getAnnotationId)
.collect(Collectors.toSet());
MigratedIds idsToMigrateInDb = entitiesToMigrate.stream()
.filter(MigrationEntity::hasManualChangesOrComments)
.filter(migrationEntity -> migrationEntity.hasManualChangesOrComments(entitiesWithComments, entitiesWithUnprocessedChanges))
.filter(m -> !m.getOldId().equals(m.getNewId()))
.collect(new MigratedIdsCollector());
@@ -113,20 +126,29 @@ public class RedactionLogToEntityLogMigrationService {
}
private void applyManualChanges(List<MigrationEntity> entitiesToMigrate, ManualRedactions manualRedactions) {
private void applyLocalProcessedManualChanges(List<MigrationEntity> entitiesToMigrate, ManualRedactions manualRedactions, boolean fileIsApproved) {
if (manualRedactions == null) {
return;
}
Map<String, List<BaseAnnotation>> manualChangesPerAnnotationId;
Map<String, List<BaseAnnotation>> manualChangesPerAnnotationId = Stream.of(manualRedactions.getIdsToRemove(),
manualRedactions.getEntriesToAdd(),
manualRedactions.getForceRedactions(),
manualRedactions.getResizeRedactions(),
manualRedactions.getLegalBasisChanges(),
manualRedactions.getRecategorizations())
.flatMap(Collection::stream)
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
if (fileIsApproved) {
manualChangesPerAnnotationId = manualRedactions.buildAll()
.stream()
.filter(manualChange -> (manualChange.getProcessedDate() != null && manualChange.isLocal()) //
// unprocessed dict change of type IdRemoval or ManualResize must be applied for approved documents
|| (manualChange.getProcessedDate() == null && !manualChange.isLocal() //
&& (manualChange instanceof IdRemoval || manualChange instanceof ManualResizeRedaction)))
.map(this::convertPendingDictChangesToLocal)
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
} else {
manualChangesPerAnnotationId = manualRedactions.buildAll()
.stream()
.filter(manualChange -> manualChange.getProcessedDate() != null)
.filter(BaseAnnotation::isLocal)
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
}
entitiesToMigrate.forEach(migrationEntity -> migrationEntity.applyManualChanges(manualChangesPerAnnotationId.getOrDefault(migrationEntity.getOldId(),
Collections.emptyList()),
@@ -135,15 +157,40 @@ public class RedactionLogToEntityLogMigrationService {
}
private static long getNumberOfApprovedEntries(RedactionLog redactionLog) {
private BaseAnnotation convertPendingDictChangesToLocal(BaseAnnotation baseAnnotation) {
return redactionLog.getRedactionLogEntry().size();
if (baseAnnotation.getProcessedDate() != null) {
return baseAnnotation;
}
if (baseAnnotation.isLocal()) {
return baseAnnotation;
}
if (baseAnnotation instanceof ManualResizeRedaction manualResizeRedaction) {
manualResizeRedaction.setAddToAllDossiers(false);
manualResizeRedaction.setUpdateDictionary(false);
} else if (baseAnnotation instanceof IdRemoval idRemoval) {
idRemoval.setRemoveFromAllDossiers(false);
idRemoval.setRemoveFromDictionary(false);
}
return baseAnnotation;
}
private long getNumberOfApprovedEntries(RedactionLog redactionLog, int numberOfPages) {
return redactionLog.getRedactionLogEntry()
.stream()
.filter(redactionLogEntry -> isOnExistingPage(redactionLogEntry, numberOfPages))
.count();
}
private List<MigrationEntity> calculateMigrationEntitiesFromRedactionLog(RedactionLog redactionLog, Document document, String dossierTemplateId, String fileId) {
List<MigrationEntity> images = getImageBasedMigrationEntities(redactionLog, document, fileId);
List<MigrationEntity> images = getImageBasedMigrationEntities(redactionLog, document, fileId, dossierTemplateId);
List<MigrationEntity> textMigrationEntities = getTextBasedMigrationEntities(redactionLog, document, dossierTemplateId, fileId);
return Stream.of(textMigrationEntities.stream(), images.stream())
.flatMap(Function.identity())
@@ -157,7 +204,7 @@ public class RedactionLogToEntityLogMigrationService {
}
private List<MigrationEntity> getImageBasedMigrationEntities(RedactionLog redactionLog, Document document, String fileId) {
private List<MigrationEntity> getImageBasedMigrationEntities(RedactionLog redactionLog, Document document, String fileId, String dossierTemplateId) {
List<Image> images = document.streamAllImages()
.collect(Collectors.toList());
@@ -204,7 +251,7 @@ public class RedactionLogToEntityLogMigrationService {
} else {
closestImage.skip(ruleIdentifier, reason);
}
migrationEntities.add(MigrationEntity.fromRedactionLogImage(redactionLogImage, closestImage, fileId));
migrationEntities.add(MigrationEntity.fromRedactionLogImage(redactionLogImage, closestImage, fileId, dictionaryService, dossierTemplateId));
}
return migrationEntities;
}
@@ -250,7 +297,8 @@ public class RedactionLogToEntityLogMigrationService {
List<MigrationEntity> entitiesToMigrate = redactionLog.getRedactionLogEntry()
.stream()
.filter(redactionLogEntry -> !redactionLogEntry.isImage())
.map(entry -> MigrationEntity.fromRedactionLogEntry(entry, dictionaryService.isHint(entry.getType(), dossierTemplateId), fileId))
.filter(redactionLogEntry -> isOnExistingPage(redactionLogEntry, document.getNumberOfPages()))
.map(entry -> MigrationEntity.fromRedactionLogEntry(entry, fileId, dictionaryService, dossierTemplateId))
.toList();
List<PrecursorEntity> precursorEntities = entitiesToMigrate.stream()
@@ -287,4 +335,20 @@ public class RedactionLogToEntityLogMigrationService {
return entitiesToMigrate;
}
private boolean isOnExistingPage(RedactionLogEntry redactionLogEntry, int numberOfPages) {
var pages = redactionLogEntry.getPositions()
.stream()
.map(Rectangle::getPage)
.collect(Collectors.toSet());
for (int page : pages) {
if (page > numberOfPages) {
return false;
}
}
return true;
}
}
@@ -14,4 +14,5 @@ public record KieWrapper(KieContainer container, long rulesVersion) {
return container != null && rulesVersion >= 0;
}
}
@@ -19,4 +19,5 @@ public class MigratedEntityLog {
MigratedIds migratedIds;
EntityLog entityLog;
}
@@ -4,9 +4,11 @@ import static com.iqser.red.service.redaction.v1.server.service.EntityLogCreator
import static com.iqser.red.service.redaction.v1.server.service.EntityLogCreatorService.buildEntryType;
import java.awt.geom.Rectangle2D;
import java.time.OffsetDateTime;
import java.util.Collections;
import java.util.LinkedList;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.Optional;
import java.util.Set;
@@ -16,10 +18,13 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryState;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Position;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualChangeFactory;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.Rectangle;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.BaseAnnotation;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRecategorization;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRedactionEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualResizeRedaction;
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.type.DictionaryEntryType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.ManualRedactionType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.migration.MigrationMapper;
@@ -28,7 +33,8 @@ import com.iqser.red.service.redaction.v1.server.model.document.entity.IEntity;
import com.iqser.red.service.redaction.v1.server.model.document.entity.ManualChangeOverwrite;
import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntity;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
import com.iqser.red.service.redaction.v1.server.service.ManualChangeFactory;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.service.ManualChangesApplicationService;
import lombok.AllArgsConstructor;
@@ -46,6 +52,8 @@ public final class MigrationEntity {
private final PrecursorEntity precursorEntity;
private final RedactionLogEntry redactionLogEntry;
private final DictionaryService dictionaryService;
private final String dossierTemplateId;
private IEntity migratedEntity;
private String oldId;
private String newId;
@@ -55,10 +63,10 @@ public final class MigrationEntity {
List<BaseAnnotation> manualChanges = new LinkedList<>();
public static MigrationEntity fromRedactionLogEntry(RedactionLogEntry redactionLogEntry, boolean hint, String fileId) {
public static MigrationEntity fromRedactionLogEntry(RedactionLogEntry redactionLogEntry, String fileId, DictionaryService dictionaryService, String dossierTemplateId) {
boolean hint = dictionaryService.isHint(redactionLogEntry.getType(), dossierTemplateId);
PrecursorEntity precursorEntity = createPrecursorEntity(redactionLogEntry, hint);
if (precursorEntity.getEntityType().equals(EntityType.HINT) && !redactionLogEntry.isHint() && !redactionLogEntry.isRedacted()) {
precursorEntity.ignore(precursorEntity.getRuleIdentifier(), precursorEntity.getReason());
} else if (redactionLogEntry.lastChangeIsRemoved()) {
@@ -73,13 +81,32 @@ public final class MigrationEntity {
precursorEntity.skip(precursorEntity.getRuleIdentifier(), precursorEntity.getReason());
}
return MigrationEntity.builder().precursorEntity(precursorEntity).redactionLogEntry(redactionLogEntry).oldId(redactionLogEntry.getId()).fileId(fileId).build();
return MigrationEntity.builder()
.precursorEntity(precursorEntity)
.redactionLogEntry(redactionLogEntry)
.oldId(redactionLogEntry.getId())
.fileId(fileId)
.dictionaryService(dictionaryService)
.dossierTemplateId(dossierTemplateId)
.build();
}
public static MigrationEntity fromRedactionLogImage(RedactionLogEntry redactionLogImage, Image image, String fileId) {
public static MigrationEntity fromRedactionLogImage(RedactionLogEntry redactionLogImage,
Image image,
String fileId,
DictionaryService dictionaryService,
String dossierTemplateId) {
return MigrationEntity.builder().redactionLogEntry(redactionLogImage).migratedEntity(image).oldId(redactionLogImage.getId()).newId(image.getId()).fileId(fileId).build();
return MigrationEntity.builder()
.redactionLogEntry(redactionLogImage)
.migratedEntity(image)
.oldId(redactionLogImage.getId())
.newId(image.getId())
.fileId(fileId)
.dictionaryService(dictionaryService)
.dossierTemplateId(dossierTemplateId)
.build();
}
@@ -156,18 +183,6 @@ public final class MigrationEntity {
}
private static EntryType getEntryType(EntityType entityType) {
return switch (entityType) {
case ENTITY -> EntryType.ENTITY;
case HINT -> EntryType.HINT;
case FALSE_POSITIVE -> EntryType.FALSE_POSITIVE;
case RECOMMENDATION -> EntryType.RECOMMENDATION;
case FALSE_RECOMMENDATION -> EntryType.FALSE_RECOMMENDATION;
};
}
public EntityLogEntry toEntityLogEntry(Map<String, String> oldToNewIdMapping) {
EntityLogEntry entityLogEntry;
@@ -181,7 +196,7 @@ public final class MigrationEntity {
throw new UnsupportedOperationException("Unknown subclass " + migratedEntity.getClass());
}
entityLogEntry.setManualChanges(ManualChangeFactory.toManualChangeList(migratedEntity.getManualOverwrite().getManualChangeLog(), redactionLogEntry.isHint()));
entityLogEntry.setManualChanges(ManualChangeFactory.toLocalManualChangeList(migratedEntity.getManualOverwrite().getManualChangeLog(), true));
entityLogEntry.setColor(redactionLogEntry.getColor());
entityLogEntry.setChanges(redactionLogEntry.getChanges()
.stream()
@@ -190,14 +205,15 @@ public final class MigrationEntity {
entityLogEntry.setReference(migrateSetOfIds(redactionLogEntry.getReference(), oldToNewIdMapping));
entityLogEntry.setImportedRedactionIntersections(migrateSetOfIds(redactionLogEntry.getImportedRedactionIntersections(), oldToNewIdMapping));
entityLogEntry.setEngines(MigrationMapper.getMigratedEngines(redactionLogEntry));
if (redactionLogEntry.getLegalBasis() != null) {
entityLogEntry.setLegalBasis(redactionLogEntry.getLegalBasis());
}
if (entityLogEntry.getEntryType().equals(EntryType.HINT) && lastManualChangeIsRemoveLocally(entityLogEntry)) {
entityLogEntry.setState(EntryState.IGNORED);
}
if (redactionLogEntry.isImported() && redactionLogEntry.getValue() == null) {
entityLogEntry.setValue("Imported Redaction");
}
return entityLogEntry;
}
@@ -225,13 +241,15 @@ public final class MigrationEntity {
public EntityLogEntry createEntityLogEntry(Image image) {
String imageType = image.getImageType().equals(ImageType.OTHER) ? "image" : image.getImageType().toString().toLowerCase(Locale.ENGLISH);
List<Position> positions = getPositionsFromOverride(image).orElse(List.of(new Position(image.getPosition(), image.getPage().getNumber())));
return EntityLogEntry.builder()
.id(image.getId())
.value(image.value())
.type(image.type())
.value(image.getValue())
.type(imageType)
.reason(image.buildReasonWithManualChangeDescriptions())
.legalBasis(image.legalBasis())
.legalBasis(image.getManualOverwrite().getLegalBasis()
.orElse(redactionLogEntry.getLegalBasis()))
.matchedRule(image.getMatchedRule().getRuleIdentifier().toString())
.dictionaryEntry(false)
.positions(positions)
@@ -243,7 +261,7 @@ public final class MigrationEntity {
.textBefore(redactionLogEntry.getTextBefore())
.imageHasTransparency(image.isTransparent())
.state(buildEntryState(image))
.entryType(redactionLogEntry.isHint() ? EntryType.IMAGE_HINT : EntryType.IMAGE)
.entryType(dictionaryService.isHint(imageType, dossierTemplateId) ? EntryType.IMAGE_HINT : EntryType.IMAGE)
.build();
}
@@ -254,7 +272,8 @@ public final class MigrationEntity {
return EntityLogEntry.builder()
.id(precursorEntity.getId())
.reason(precursorEntity.buildReasonWithManualChangeDescriptions())
.legalBasis(precursorEntity.legalBasis())
.legalBasis(precursorEntity.getManualOverwrite().getLegalBasis()
.orElse(redactionLogEntry.getLegalBasis()))
.value(precursorEntity.value())
.type(precursorEntity.type())
.state(buildEntryState(precursorEntity))
@@ -288,7 +307,8 @@ public final class MigrationEntity {
.id(entity.getId())
.positions(rectanglesPerLine)
.reason(entity.buildReasonWithManualChangeDescriptions())
.legalBasis(entity.legalBasis())
.legalBasis(entity.getManualOverwrite().getLegalBasis()
.orElse(redactionLogEntry.getLegalBasis()))
.value(entity.getManualOverwrite().getValue()
.orElse(entity.getMatchedRule().isWriteValueWithLineBreaks() ? entity.getValueWithLineBreaks() : entity.getValue()))
.type(entity.type())
@@ -331,11 +351,11 @@ public final class MigrationEntity {
}
public boolean hasManualChangesOrComments() {
public boolean hasManualChangesOrComments(Set<String> entitiesWithComments, Set<String> entitiesWithUnprocessedChanges) {
return !(redactionLogEntry.getManualChanges() == null || redactionLogEntry.getManualChanges().isEmpty()) || //
!(redactionLogEntry.getComments() == null || redactionLogEntry.getComments().isEmpty()) //
|| hasManualChanges();
|| hasManualChanges() || entitiesWithComments.contains(oldId) || entitiesWithUnprocessedChanges.contains(oldId);
}
@@ -350,17 +370,11 @@ public final class MigrationEntity {
manualChanges.addAll(manualChangesToApply);
manualChangesToApply.forEach(manualChange -> {
if (manualChange instanceof ManualResizeRedaction manualResizeRedaction && migratedEntity instanceof TextEntity textEntity) {
// Due to the value in the old redaction log already being resized, there is no way to find the original entity ID and therefore to migrate the resize annotation correctly.
// Instead, we add an add_locally change to the db.
ManualResizeRedaction migratedManualResizeRedaction = ManualResizeRedaction.builder()
.positions(manualResizeRedaction.getPositions())
.annotationId(getNewId())
.updateDictionary(manualResizeRedaction.getUpdateDictionary())
.addToAllDossiers(manualResizeRedaction.isAddToAllDossiers())
.textAfter(manualResizeRedaction.getTextAfter())
.textBefore(manualResizeRedaction.getTextBefore())
.build();
manualChangesApplicationService.resize(textEntity, migratedManualResizeRedaction);
manualResizeRedaction.setAnnotationId(newId);
manualChangesApplicationService.resize(textEntity, manualResizeRedaction);
} else if (manualChange instanceof ManualRecategorization manualRecategorization && migratedEntity instanceof Image image) {
image.setImageType(ImageType.fromString(manualRecategorization.getType()));
migratedEntity.getManualOverwrite().addChange(manualChange);
} else {
migratedEntity.getManualOverwrite().addChange(manualChange);
}
@@ -378,19 +392,25 @@ public final class MigrationEntity {
.findFirst()
.orElse(manualChanges.get(0)).getUser();
OffsetDateTime requestDate = manualChanges.get(0).getRequestDate();
return ManualRedactionEntry.builder()
.annotationId(newId)
.fileId(fileId)
.user(user)
.requestDate(requestDate)
.type(redactionLogEntry.getType())
.value(redactionLogEntry.getValue())
.reason(redactionLogEntry.getReason())
.legalBasis(redactionLogEntry.getLegalBasis())
.section(redactionLogEntry.getSection())
.rectangle(false)
.addToDictionary(false)
.addToDossierDictionary(false)
.rectangle(false)
.positions(buildPositions(migratedEntity))
.user(user)
.textAfter(redactionLogEntry.getTextAfter())
.textBefore(redactionLogEntry.getTextBefore())
.dictionaryEntryType(DictionaryEntryType.ENTRY)
.build();
}
@@ -11,6 +11,10 @@ import lombok.AllArgsConstructor;
import lombok.Getter;
import lombok.experimental.FieldDefaults;
/**
* Represents a collection of named entity recognition (NER) entities.
* This class provides methods to manage and query NER entities.
*/
@Getter
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE, makeFinal = true)
@@ -25,18 +29,35 @@ public class NerEntities {
}
/**
* Checks if there are any entities of a specified type.
*
* @param type The type of entity to check for.
* @return true if there is at least one entity of the specified type, false otherwise.
*/
public boolean hasEntitiesOfType(String type) {
return nerEntityList.stream().anyMatch(nerEntity -> nerEntity.type.equals(type));
return nerEntityList.stream()
.anyMatch(nerEntity -> nerEntity.type.equals(type));
}
/**
* Returns a stream of NER entities of a specified type.
*
* @param type The type of entities to return.
* @return a stream of {@link NerEntity} objects of the specified type.
*/
public Stream<NerEntity> streamEntitiesOfType(String type) {
return nerEntityList.stream().filter(nerEntity -> nerEntity.type().equals(type));
return nerEntityList.stream()
.filter(nerEntity -> nerEntity.type().equals(type));
}
/**
* Represents a single NER entity with its value, text range, and type.
*/
public record NerEntity(String value, TextRange textRange, String type) {
}
@@ -88,7 +88,8 @@ public class Entity {
.textAfter(e.getTextAfter())
.startOffset(e.getStartOffset())
.endOffset(e.getEndOffset())
.length(Optional.ofNullable(e.getValue()).orElse("").length())
.length(Optional.ofNullable(e.getValue())
.orElse("").length())
.imageHasTransparency(e.isImageHasTransparency())
.isDictionaryEntry(e.isDictionaryEntry())
.isDossierDictionaryEntry(e.isDossierDictionaryEntry())
@@ -23,6 +23,9 @@ import com.iqser.red.service.redaction.v1.server.utils.exception.NotFoundExcepti
import lombok.Data;
import lombok.Getter;
/**
* A class representing a dictionary used for redaction processes, containing various dictionary models and their versions.
*/
@Data
public class Dictionary {
@@ -51,9 +54,15 @@ public class Dictionary {
}
/**
* Checks if the dictionary contains local entries.
*
* @return true if any dictionary model contains local entries, false otherwise.
*/
public boolean hasLocalEntries() {
return dictionaryModels.stream().anyMatch(dm -> !dm.getLocalEntriesWithMatchedRules().isEmpty());
return dictionaryModels.stream()
.anyMatch(dm -> !dm.getLocalEntriesWithMatchedRules().isEmpty());
}
@@ -63,6 +72,13 @@ public class Dictionary {
}
/**
* Retrieves the {@link DictionaryModel} of a specified type.
*
* @param type The type of dictionary model to retrieve.
* @return The {@link DictionaryModel} of the specified type.
* @throws NotFoundException If the specified type is not found in the dictionary.
*/
public DictionaryModel getType(String type) {
DictionaryModel model = localAccessMap.get(type);
@@ -73,6 +89,12 @@ public class Dictionary {
}
/**
* Checks if the dictionary of a specific type is considered a hint.
*
* @param type The type of dictionary to check.
* @return true if the dictionary model is marked as a hint, false otherwise.
*/
public boolean isHint(String type) {
DictionaryModel model = localAccessMap.get(type);
@@ -83,6 +105,12 @@ public class Dictionary {
}
/**
* Checks if the dictionary of a specific type is case-insensitive.
*
* @param type The type of dictionary to check.
* @return true if the dictionary is case-insensitive, false otherwise.
*/
public boolean isCaseInsensitiveDictionary(String type) {
DictionaryModel dictionaryModel = localAccessMap.get(type);
@@ -93,6 +121,18 @@ public class Dictionary {
}
/**
* Adds a local dictionary entry of a specific type.
*
* @param type The type of dictionary to add the entry to.
* @param value The value of the entry.
* @param matchedRules A collection of {@link MatchedRule} associated with the entry.
* @param alsoAddLastname Indicates whether to also add the lastname separately as an entry.
* @throws IllegalArgumentException If the specified type does not exist within the dictionary, if the type
* does not have any local entries defined, or if the provided value is
* blank. This ensures that only valid, non-empty entries
* are added to the dictionary.
*/
private void addLocalDictionaryEntry(String type, String value, Collection<MatchedRule> matchedRules, boolean alsoAddLastname) {
if (value.isBlank()) {
@@ -116,28 +156,49 @@ public class Dictionary {
}
localAccessMap.get(type)
.getLocalEntriesWithMatchedRules()
.merge(cleanedValue.trim(), matchedRulesSet, (set1, set2) -> Stream.concat(set1.stream(), set2.stream()).collect(Collectors.toSet()));
.merge(cleanedValue.trim(),
matchedRulesSet,
(set1, set2) -> Stream.concat(set1.stream(), set2.stream())
.collect(Collectors.toSet()));
if (alsoAddLastname) {
String lastname = cleanedValue.split(" ")[0];
localAccessMap.get(type)
.getLocalEntriesWithMatchedRules()
.merge(lastname, matchedRulesSet, (set1, set2) -> Stream.concat(set1.stream(), set2.stream()).collect(Collectors.toSet()));
.merge(lastname,
matchedRulesSet,
(set1, set2) -> Stream.concat(set1.stream(), set2.stream())
.collect(Collectors.toSet()));
}
}
/**
* Recommends a text entity for inclusion in every dictionary model without separating the last name.
*
* @param textEntity The {@link TextEntity} to be recommended.
*/
public void recommendEverywhere(TextEntity textEntity) {
addLocalDictionaryEntry(textEntity.type(), textEntity.getValue(), textEntity.getMatchedRuleList(), false);
}
/**
* Recommends a text entity for inclusion in every dictionary model with the last name added separately.
*
* @param textEntity The {@link TextEntity} to be recommended.
*/
public void recommendEverywhereWithLastNameSeparately(TextEntity textEntity) {
addLocalDictionaryEntry(textEntity.type(), textEntity.getValue(), textEntity.getMatchedRuleList(), true);
}
/**
* Adds multiple author names contained within a text entity as recommendations in the dictionary.
*
* @param textEntity The {@link TextEntity} containing author names to be added.
*/
public void addMultipleAuthorsAsRecommendation(TextEntity textEntity) {
splitIntoAuthorNames(textEntity).forEach(authorName -> addLocalDictionaryEntry(textEntity.type(), authorName, textEntity.getMatchedRuleList(), true));
@@ -145,6 +206,12 @@ public class Dictionary {
}
/**
* Splits a {@link TextEntity} into individual author names based on commas or new lines.
*
* @param textEntity The {@link TextEntity} to split.
* @return A list of strings where each string is an author name.
*/
public static List<String> splitIntoAuthorNames(TextEntity textEntity) {
List<String> splitAuthorNames;
@@ -153,7 +220,10 @@ public class Dictionary {
} else {
splitAuthorNames = Arrays.asList(textEntity.getValueWithLineBreaks().split("\n"));
}
return splitAuthorNames.stream().map(String::trim).filter(authorName -> Patterns.AUTHOR_NAME_PATTERN.matcher(authorName).matches()).toList();
return splitAuthorNames.stream()
.map(String::trim)
.filter(authorName -> Patterns.AUTHOR_NAME_PATTERN.matcher(authorName).matches())
.toList();
}
}
@@ -2,7 +2,6 @@ package com.iqser.red.service.redaction.v1.server.model.dictionary;
import java.io.Serializable;
import java.util.HashMap;
import java.util.List;
import java.util.Locale;
import java.util.Set;
import java.util.stream.Collectors;
@@ -11,11 +10,17 @@ import com.iqser.red.service.dictionarymerge.commons.DictionaryEntry;
import com.iqser.red.service.dictionarymerge.commons.DictionaryEntryModel;
import com.iqser.red.service.redaction.v1.server.model.document.entity.MatchedRule;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
/**
* Represents a model of a dictionary containing entries for redaction processes.
* It includes various types of entries such as standard entries, false positives,
* and false recommendations. Additionally, it manages local entries with matched
* rules for enhanced search and matching capabilities.
*/
@Data
@AllArgsConstructor
@Slf4j
public class DictionaryModel implements Serializable {
private final String type;
@@ -29,6 +34,7 @@ public class DictionaryModel implements Serializable {
private final Set<DictionaryEntryModel> falseRecommendations;
private transient SearchImplementation entriesSearch;
private transient SearchImplementation deletionEntriesSearch;
private transient SearchImplementation falsePositiveSearch;
private transient SearchImplementation falseRecommendationsSearch;
@@ -36,6 +42,19 @@ public class DictionaryModel implements Serializable {
private transient SearchImplementation localSearch;
/**
* Constructs a new DictionaryModel with specified parameters.
*
* @param type The type of the dictionary model.
* @param rank The rank order of the dictionary model.
* @param color An array representing the color associated with this model.
* @param caseInsensitive Flag indicating whether the dictionary is case-insensitive.
* @param hint Flag indicating whether this model should be used as a hint.
* @param entries Set of dictionary entry models representing the entries.
* @param falsePositives Set of dictionary entry models representing false positives.
* @param falseRecommendations Set of dictionary entry models representing false recommendations.
* @param isDossierDictionary Flag indicating whether this model is for a dossier dictionary.
*/
public DictionaryModel(String type,
int rank,
float[] color,
@@ -52,23 +71,17 @@ public class DictionaryModel implements Serializable {
this.caseInsensitive = caseInsensitive;
this.hint = hint;
this.isDossierDictionary = isDossierDictionary;
this.entries = entries;
this.falsePositives = falsePositives;
this.falseRecommendations = falseRecommendations;
this.entriesSearch = new SearchImplementation(this.entries.stream().filter(e -> !e.isDeleted()).map(DictionaryEntryModel::getValue).collect(Collectors.toList()),
caseInsensitive);
this.falsePositiveSearch = new SearchImplementation(this.falsePositives.stream().filter(e -> !e.isDeleted()).map(DictionaryEntryModel::getValue).collect(Collectors.toList()),
caseInsensitive);
this.falseRecommendationsSearch = new SearchImplementation(this.falseRecommendations.stream()
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
}
/**
* Returns the search implementation for local entries.
*
* @return The {@link SearchImplementation} for local entries.
*/
public SearchImplementation getLocalSearch() {
if (this.localSearch == null || this.localSearch.getValues().size() != this.localEntriesWithMatchedRules.size()) {
@@ -78,44 +91,85 @@ public class DictionaryModel implements Serializable {
}
/**
* Returns the search implementation for non-deleted dictionary entries.
*
* @return The {@link SearchImplementation} for non-deleted dictionary entries.
*/
public SearchImplementation getEntriesSearch() {
if (entriesSearch == null) {
this.entriesSearch = new SearchImplementation(this.entries.stream().filter(e -> !e.isDeleted()).map(DictionaryEntry::getValue).collect(Collectors.toList()),
caseInsensitive);
this.entriesSearch = new SearchImplementation(this.entries.stream()
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
}
return entriesSearch;
}
/**
* Returns the search implementation for deleted dictionary entries.
*
* @return The {@link SearchImplementation} for deleted dictionary entries.
*/
public SearchImplementation getDeletionEntriesSearch() {
if (deletionEntriesSearch == null) {
this.deletionEntriesSearch = new SearchImplementation(this.entries.stream()
.filter(DictionaryEntry::isDeleted)
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
}
return deletionEntriesSearch;
}
/**
* Returns the search implementation for non-deleted false positive entries.
*
* @return The {@link SearchImplementation} for non-deleted false positive entries.
*/
public SearchImplementation getFalsePositiveSearch() {
if (falsePositiveSearch == null) {
this.falsePositiveSearch = new SearchImplementation(this.falsePositives.stream()
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
}
return falsePositiveSearch;
}
/**
* Returns the search implementation for non-deleted false recommendation entries.
*
* @return The {@link SearchImplementation} for non-deleted false recommendation entries.
*/
public SearchImplementation getFalseRecommendationsSearch() {
if (falseRecommendationsSearch == null) {
this.falseRecommendationsSearch = new SearchImplementation(this.falseRecommendations.stream()
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
.filter(e -> !e.isDeleted())
.map(DictionaryEntry::getValue)
.collect(Collectors.toList()), caseInsensitive);
}
return falseRecommendationsSearch;
}
/**
* Retrieves the matched rules for a given value from the local dictionary entries.
* The value is processed based on the case sensitivity of the dictionary.
*
* @param value The value for which to retrieve the matched rules.
* @return A set of {@link MatchedRule} associated with the given value, or null if no rules are found.
*/
public Set<MatchedRule> getMatchedRulesForLocalDictionaryEntry(String value) {
var cleanedValue = isCaseInsensitive() ? value.toLowerCase(Locale.US) : value;
return localEntriesWithMatchedRules.get(cleanedValue);
}
}
@@ -76,7 +76,9 @@ public class SearchImplementation {
if (ignoreCase) {
textToCheck = textToCheck.toLowerCase(Locale.ROOT);
}
return this.pattern.matcher(textToCheck).results().findAny().isPresent();
return this.pattern.matcher(textToCheck).results()
.findAny()
.isPresent();
} else {
return this.trie.containsMatch(textToCheck);
}
@@ -89,9 +91,14 @@ public class SearchImplementation {
return new ArrayList<>();
}
if (this.pattern != null) {
return this.pattern.matcher(text).results().map(r -> new TextRange(r.start(), r.end())).collect(Collectors.toList());
return this.pattern.matcher(text).results()
.map(r -> new TextRange(r.start(), r.end()))
.collect(Collectors.toList());
} else {
return this.trie.parseText(text).stream().map(r -> new TextRange(r.getStart(), r.getEnd() + 1)).collect(Collectors.toList());
return this.trie.parseText(text)
.stream()
.map(r -> new TextRange(r.getStart(), r.getEnd() + 1))
.collect(Collectors.toList());
}
}
@@ -103,9 +110,14 @@ public class SearchImplementation {
}
CharSequence subSequence = text.subSequence(region.start(), region.end());
if (this.pattern != null) {
return this.pattern.matcher(subSequence).results().map(r -> new TextRange(r.start() + region.start(), r.end() + region.start())).collect(Collectors.toList());
return this.pattern.matcher(subSequence).results()
.map(r -> new TextRange(r.start() + region.start(), r.end() + region.start()))
.collect(Collectors.toList());
} else {
return this.trie.parseText(subSequence).stream().map(r -> new TextRange(r.getStart() + region.start(), r.getEnd() + region.start() + 1)).collect(Collectors.toList());
return this.trie.parseText(subSequence)
.stream()
.map(r -> new TextRange(r.getStart() + region.start(), r.getEnd() + region.start() + 1))
.collect(Collectors.toList());
}
}
@@ -120,9 +132,14 @@ public class SearchImplementation {
if (ignoreCase) {
textToCheck = textToCheck.toLowerCase(Locale.ROOT);
}
return this.pattern.matcher(textToCheck).results().map(r -> new MatchPosition(r.start(), r.end())).collect(Collectors.toList());
return this.pattern.matcher(textToCheck).results()
.map(r -> new MatchPosition(r.start(), r.end()))
.collect(Collectors.toList());
} else {
return this.trie.parseText(textToCheck).stream().map(r -> new MatchPosition(r.getStart(), r.getEnd() + 1)).collect(Collectors.toList());
return this.trie.parseText(textToCheck)
.stream()
.map(r -> new MatchPosition(r.getStart(), r.getEnd() + 1))
.collect(Collectors.toList());
}
}
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.server.model.document;
import static java.lang.String.format;
import java.util.ArrayList;
import java.util.Collections;
import java.util.LinkedList;
import java.util.List;
@@ -40,7 +41,10 @@ public class DocumentTree {
public TextBlock buildTextBlock() {
return allEntriesInOrder().map(Entry::getNode).filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
return allEntriesInOrder().map(Entry::getNode)
.filter(SemanticNode::isLeaf)
.map(SemanticNode::getLeafTextBlock)
.collect(new TextBlockCollector());
}
@@ -89,8 +93,8 @@ public class DocumentTree {
if (treeId.isEmpty()) {
return root != null;
}
Entry entry = root.children.get(treeId.get(0));
for (int id : treeId.subList(1, treeId.size())) {
Entry entry = root;
for (int id : treeId) {
if (id >= entry.children.size() || 0 > id) {
return false;
}
@@ -114,13 +118,78 @@ public class DocumentTree {
public Stream<SemanticNode> childNodes(List<Integer> treeId) {
return getEntryById(treeId).children.stream().map(Entry::getNode);
return getEntryById(treeId).children.stream()
.map(Entry::getNode);
}
/**
* Finds all child nodes of the specified entry, whose nodes textRange intersects the given textRange. It achieves this by finding the first entry, whose textRange contains the start idx of the TextRange using a binary search.
* It then iterates over the remaining children adding them to the intersections, until one does not contain the end of the TextRange. All intersected Entries are returned as SemanticNodes.
*
* @param treeId the treeId of the Entry whose children shall be checked.
* @param textRange The TextRange to find intersecting childNodes for.
* @return A list of all SemanticNodes, that are direct children of the specified Entry, whose TextRange intersects the given TextRange
*/
public List<SemanticNode> findIntersectingChildNodes(List<Integer> treeId, TextRange textRange) {
List<Entry> childEntries = getEntryById(treeId).getChildren();
List<SemanticNode> intersectingChildEntries = new LinkedList<>();
int startIdx = findFirstIdxOfContainingChildBinarySearch(childEntries, textRange.start());
if (startIdx < 0) {
return intersectingChildEntries;
}
for (int i = startIdx; i < childEntries.size(); i++) {
if (childEntries.get(i).getNode().getTextRange().start() < textRange.end()) {
intersectingChildEntries.add(childEntries.get(i).getNode());
} else {
break;
}
}
return intersectingChildEntries;
}
public Optional<SemanticNode> findFirstContainingChild(List<Integer> treeId, TextRange textRange) {
List<Entry> childEntries = getEntryById(treeId).getChildren();
int startIdx = findFirstIdxOfContainingChildBinarySearch(childEntries, textRange.start());
if (startIdx < 0) {
return Optional.empty();
}
if (childEntries.get(startIdx).getNode().getTextRange().contains(textRange.end())) {
return Optional.of(childEntries.get(startIdx).getNode());
}
return Optional.empty();
}
private int findFirstIdxOfContainingChildBinarySearch(List<Entry> childNodes, int start) {
int low = 0;
int high = childNodes.size() - 1;
while (low <= high) {
int mid = low + (high - low) / 2;
TextRange range = childNodes.get(mid).getNode().getTextRange();
if (range.start() > start) {
high = mid - 1;
} else if (range.end() <= start) {
low = mid + 1;
} else {
return mid;
}
}
return -1;
}
public Stream<SemanticNode> childNodesOfType(List<Integer> treeId, NodeType nodeType) {
return getEntryById(treeId).children.stream().filter(entry -> entry.node.getType().equals(nodeType)).map(Entry::getNode);
return getEntryById(treeId).children.stream()
.filter(entry -> entry.node.getType().equals(nodeType))
.map(Entry::getNode);
}
@@ -199,26 +268,32 @@ public class DocumentTree {
public Stream<Entry> allEntriesInOrder() {
return Stream.of(root).flatMap(DocumentTree::flatten);
return Stream.of(root)
.flatMap(DocumentTree::flatten);
}
public Stream<Entry> allSubEntriesInOrder(List<Integer> parentId) {
return getEntryById(parentId).children.stream().flatMap(DocumentTree::flatten);
return getEntryById(parentId).children.stream()
.flatMap(DocumentTree::flatten);
}
@Override
public String toString() {
return String.join("\n", allEntriesInOrder().map(Entry::toString).toList());
return String.join("\n",
allEntriesInOrder().map(Entry::toString)
.toList());
}
private static Stream<Entry> flatten(Entry entry) {
return Stream.concat(Stream.of(entry), entry.children.stream().flatMap(DocumentTree::flatten));
return Stream.concat(Stream.of(entry),
entry.children.stream()
.flatMap(DocumentTree::flatten));
}
@@ -240,7 +315,7 @@ public class DocumentTree {
List<Integer> treeId;
SemanticNode node;
@Builder.Default
List<Entry> children = new LinkedList<>();
List<Entry> children = new ArrayList<>();
@Override
@@ -11,6 +11,10 @@ import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBl
import lombok.EqualsAndHashCode;
import lombok.Setter;
/**
* Represents a range of text defined by a start and end index.
* Provides functionality to check containment, intersection, and to adjust ranges based on specified conditions.
*/
@Setter
@EqualsAndHashCode
@SuppressWarnings("PMD.AvoidFieldNameMatchingMethodName")
@@ -20,6 +24,13 @@ public class TextRange implements Comparable<TextRange> {
private int end;
/**
* Constructs a TextRange with specified start and end indexes.
*
* @param start The starting index of the range.
* @param end The ending index of the range.
* @throws IllegalArgumentException If start is greater than end.
*/
public TextRange(int start, int end) {
if (start > end) {
@@ -30,6 +41,11 @@ public class TextRange implements Comparable<TextRange> {
}
/**
* Returns the length of the text range.
*
* @return The length of the range.
*/
public int length() {
return end - start;
@@ -48,18 +64,38 @@ public class TextRange implements Comparable<TextRange> {
}
/**
* Checks if this {@link TextRange} fully contains another TextRange.
*
* @param textRange The {@link TextRange} to check.
* @return true if this range contains the specified range, false otherwise.
*/
public boolean contains(TextRange textRange) {
return start <= textRange.start() && textRange.end() <= end;
}
/**
* Checks if this {@link TextRange} is fully contained by another TextRange.
*
* @param textRange The {@link TextRange} to check against.
* @return true if this range is contained by the specified range, false otherwise.
*/
public boolean containedBy(TextRange textRange) {
return textRange.contains(this);
}
/**
* Checks if this {@link TextRange} contains another range specified by start and end indices.
*
* @param start The starting index of the range to check.
* @param end The ending index of the range to check.
* @return true if this range fully contains the specified range, false otherwise.
* @throws IllegalArgumentException If the start index is greater than the end index.
*/
public boolean contains(int start, int end) {
if (start > end) {
@@ -69,6 +105,14 @@ public class TextRange implements Comparable<TextRange> {
}
/**
* Checks if this {@link TextRange} is fully contained within another range specified by start and end indices.
*
* @param start The starting index of the outer range.
* @param end The ending index of the outer range.
* @return true if this range is fully contained within the specified range, false otherwise.
* @throws IllegalArgumentException If the start index is greater than the end index.
*/
public boolean containedBy(int start, int end) {
if (start > end) {
@@ -78,26 +122,51 @@ public class TextRange implements Comparable<TextRange> {
}
/**
* Determines if the specified index is within this {@link TextRange}.
*
* @param index The index to check.
* @return true if the index is within the range (inclusive of the start and exclusive of the end), false otherwise.
*/
public boolean contains(int index) {
return start <= index && index < end;
}
/**
* Checks if this {@link TextRange} intersects with another {@link TextRange}.
*
* @param textRange The {@link TextRange} to check for intersection.
* @return true if the ranges intersect, false otherwise.
*/
public boolean intersects(TextRange textRange) {
return textRange.start() < this.end && this.start < textRange.end();
}
/**
* Splits this TextRange into multiple ranges based on a list of indices.
*
* @param splitIndices The indices at which to split the range.
* @return A list of TextRanges resulting from the split.
* @throws IndexOutOfBoundsException If any split index is outside this TextRange.
*/
public List<TextRange> split(List<Integer> splitIndices) {
if (splitIndices.stream().anyMatch(idx -> !this.contains(idx))) {
throw new IndexOutOfBoundsException(format("%s splitting indices are out of range for %s", splitIndices.stream().filter(idx -> !this.contains(idx)).toList(), this));
if (splitIndices.stream()
.anyMatch(idx -> !this.contains(idx))) {
throw new IndexOutOfBoundsException(format("%s splitting indices are out of range for %s",
splitIndices.stream()
.filter(idx -> !this.contains(idx))
.toList(),
this));
}
List<TextRange> splitBoundaries = new LinkedList<>();
int previousIndex = start;
for (int splitIndex : splitIndices) {
for (int i = 0, splitIndicesSize = splitIndices.size(); i < splitIndicesSize; i++) {
int splitIndex = splitIndices.get(i);
// skip split if it would produce a boundary of length 0
if (splitIndex == previousIndex) {
@@ -111,10 +180,23 @@ public class TextRange implements Comparable<TextRange> {
}
/**
* Merges a collection of TextRanges into a single Text range encompassing all.
*
* @param boundaries The collection of TextRanges to merge.
* @return A new TextRange covering the entire span of the given ranges.
* @throws IllegalArgumentException If boundaries are empty.
*/
public static TextRange merge(Collection<TextRange> boundaries) {
int minStart = boundaries.stream().mapToInt(TextRange::start).min().orElseThrow(IllegalArgumentException::new);
int maxEnd = boundaries.stream().mapToInt(TextRange::end).max().orElseThrow(IllegalArgumentException::new);
int minStart = boundaries.stream()
.mapToInt(TextRange::start)
.min()
.orElseThrow(IllegalArgumentException::new);
int maxEnd = boundaries.stream()
.mapToInt(TextRange::end)
.max()
.orElseThrow(IllegalArgumentException::new);
return new TextRange(minStart, maxEnd);
}
@@ -141,16 +223,17 @@ public class TextRange implements Comparable<TextRange> {
/**
* shrinks the boundary, such that textBlock.subSequence(boundary) returns a string without trailing or preceding whitespaces.
* Shrinks the boundary, such that textBlock.subSequence(boundary) returns a string without trailing or preceding whitespaces.
*
* @param textBlock TextBlock to check whitespaces against
* @return trimmed boundary
* @return Trimmed boundary
*/
public TextRange trim(TextBlock textBlock) {
if (this.length() == 0) {
return this;
}
int trimmedStart = this.start;
while (textBlock.containsIndex(trimmedStart) && trimmedStart < end && Character.isWhitespace(textBlock.charAt(trimmedStart))) {
trimmedStart++;
@@ -5,5 +5,6 @@ public enum EntityType {
HINT,
RECOMMENDATION,
FALSE_POSITIVE,
FALSE_RECOMMENDATION
FALSE_RECOMMENDATION,
DICTIONARY_REMOVAL
}
@@ -12,82 +12,173 @@ import lombok.NonNull;
public interface IEntity {
/**
* Gets the list of rules matched against this entity.
*
* @return A priority queue of matched rules.
*/
PriorityQueue<MatchedRule> getMatchedRuleList();
/**
* Gets the manual overwrite actions applied to this entity, if any.
*
* @return The manual overwrite details.
*/
ManualChangeOverwrite getManualOverwrite();
/**
* Gets the value of this entity as a string.
*
* @return The string value.
*/
String getValue();
/**
* Gets the range of text in the document associated with this entity.
*
* @return The text range.
*/
TextRange getTextRange();
/**
* Gets the type of this entity.
*
* @return The entity type.
*/
String type();
/**
* Calculates the length of the entity's value.
*
* @return The length of the value.
*/
default int length() {
return value().length();
}
/**
* Retrieves the value of the entity, considering any manual overwrite.
* If no manual overwrite value is found, return the value of the entity or an empty string
* if that value is null.
*
* @return The possibly overwritten value
*/
default String value() {
return getManualOverwrite().getValue().orElse(getValue() == null ? "" : getValue());
return getManualOverwrite().getValue()
.orElse(getValue() == null ? "" : getValue());
}
/**
* Determines if the entity has been applied, considering manual overwrites.
*
* @return True if applied, false otherwise.
*/
// Don't use default accessor pattern (e.g. isApplied()), as it might lead to errors in drools due to property-specific optimization of the drools planner.
default boolean applied() {
return getManualOverwrite().getApplied().orElse(getMatchedRule().isApplied());
return getManualOverwrite().getApplied()
.orElse(getMatchedRule().isApplied());
}
/**
* Determines if the entity has been skipped, based on its applied status.
*
* @return True if skipped, false otherwise.
*/
default boolean skipped() {
return !applied();
}
/**
* Determines if the entity has been ignored, considering manual overwrites.
*
* @return True if ignored, false otherwise.
*/
default boolean ignored() {
return getManualOverwrite().getIgnored().orElse(getMatchedRule().isIgnored());
return getManualOverwrite().getIgnored()
.orElse(getMatchedRule().isIgnored());
}
/**
* Determines if the entity has been removed, considering manual overwrites.
*
* @return True if removed, false otherwise.
*/
default boolean removed() {
return getManualOverwrite().getRemoved().orElse(getMatchedRule().isRemoved());
return getManualOverwrite().getRemoved()
.orElse(getMatchedRule().isRemoved());
}
/**
* Checks if the entity has been resized, considering manual overwrites.
*
* @return True if resized, false otherwise.
*/
default boolean resized() {
return getManualOverwrite().getResized().orElse(false);
return getManualOverwrite().getResized()
.orElse(false);
}
/**
* Checks if the entity is considered active, based on its removed and ignored status.
* An active entry is not removed or ignored.
*
* @return True if active, false otherwise.
*/
default boolean active() {
return !(removed() || ignored());
}
/**
* Checks if there are any manual changes applied to the entity.
*
* @return True if there are manual changes, false otherwise.
*/
default boolean hasManualChanges() {
return !getManualOverwrite().getManualChangeLog().isEmpty();
}
/**
* Retrieves a set of references associated with the entity's matched rule.
*
* @return A set of references.
*/
default Set<TextEntity> references() {
return getMatchedRule().getReferences();
}
/**
* Applies a redaction to the entity with a specified legal basis.
*
* @param ruleIdentifier The identifier of the rule being applied.
* @param reason The reason for the redaction.
* @param legalBasis The legal basis for the redaction, which must not be blank or empty.
* @throws IllegalArgumentException If the legal basis is blank or empty.
*/
default void redact(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
@@ -97,78 +188,143 @@ public interface IEntity {
}
/**
* Applies a rule to the entity with an optional legal basis.
*
* @param ruleIdentifier The identifier of the rule being applied.
* @param reason The reason for applying the rule.
* @param legalBasis The legal basis for the application, can be a default or unspecified value.
*/
default void apply(@NonNull String ruleIdentifier, String reason, String legalBasis) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).legalBasis(legalBasis).applied(true).build());
}
/**
* Applies a rule to the entity without specifying a legal basis, which will be replaced by "n-a".
*
* @param ruleIdentifier The identifier of the rule being applied.
* @param reason The reason for applying the rule.
*/
default void apply(@NonNull String ruleIdentifier, String reason) {
apply(ruleIdentifier, reason, "n-a");
}
/**
* Marks the entity as skipped according to a specific rule.
*
* @param ruleIdentifier The identifier of the rule being skipped.
* @param reason The reason for skipping the rule.
*/
default void skip(@NonNull String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).build());
}
/**
* Marks the entity as removed according to a specific rule.
*
* @param ruleIdentifier The identifier of the rule based on which the entity is removed.
* @param reason The reason for the removal.
*/
default void remove(String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).removed(true).build());
}
/**
* Marks the entity as ignored according to a specific rule.
*
* @param ruleIdentifier The identifier of the rule based on which the entity is removed.
* @param reason The reason for the removal.
*/
default void ignore(String ruleIdentifier, String reason) {
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).ignored(true).build());
}
/**
* Applies a rule to the entity, indicating that the value should be written with line breaks.
*
* @param ruleIdentifier The identifier of the rule being applied.
* @param reason The reason for the rule application.
* @param legalBasis The legal basis for the rule, which must not be empty.
* @throws IllegalArgumentException If the legal basis is blank or empty.
*/
default void applyWithLineBreaks(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
}
getMatchedRuleList().add(MatchedRule.builder()
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.writeValueWithLineBreaks(true)
.build());
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.writeValueWithLineBreaks(true)
.build());
}
/**
* Applies a rule to the entity with a collection of references.
*
* @param ruleIdentifier The identifier of the rule being applied.
* @param reason The reason for the rule application.
* @param legalBasis The legal basis for the rule, which must not be empty.
* @param references A collection of text entities that are referenced by this rule application.
* @throws IllegalArgumentException If the legal basis is blank or empty.
*/
default void applyWithReferences(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis, Collection<TextEntity> references) {
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
}
getMatchedRuleList().add(MatchedRule.builder()
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.references(new HashSet<>(references))
.build());
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
.reason(reason)
.legalBasis(legalBasis)
.applied(true)
.references(new HashSet<>(references))
.build());
}
/**
* Marks the entity as skipped for a specific rule and associates a collection of references.
*
* @param ruleIdentifier The identifier of the rule being skipped.
* @param reason The reason for skipping the rule.
* @param references A collection of text entities that are referenced by the skipped rule.
*/
default void skipWithReferences(@NonNull String ruleIdentifier, String reason, Collection<TextEntity> references) {
getMatchedRuleList().add(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).references(new HashSet<>(references)).build());
}
/**
* Adds a single matched rule to this entity.
*
* @param matchedRule The matched rule to add.
*/
default void addMatchedRule(MatchedRule matchedRule) {
getMatchedRuleList().add(matchedRule);
}
/**
* Adds a collection of matched rules to this entity.
*
* @param matchedRules The collection of matched rules to add.
*/
default void addMatchedRules(Collection<MatchedRule> matchedRules) {
if (getMatchedRuleList().equals(matchedRules)) {
@@ -178,12 +334,22 @@ public interface IEntity {
}
/**
* Retrieves the 'unit' value of the highest priority matched rule.
*
* @return The unit value of the matched rule.
*/
default int getMatchedRuleUnit() {
return getMatchedRule().getRuleIdentifier().unit();
}
/**
* Gets the highest priority matched rule for this entity.
*
* @return The matched rule.
*/
default MatchedRule getMatchedRule() {
if (getMatchedRuleList().isEmpty()) {
@@ -193,6 +359,11 @@ public interface IEntity {
}
/**
* Builds a reason string for this entity, incorporating descriptions from manual changes.
*
* @return The built reason string.
*/
default String buildReasonWithManualChangeDescriptions() {
if (getManualOverwrite().getDescriptions().isEmpty()) {
@@ -205,9 +376,15 @@ public interface IEntity {
}
/**
* Retrieves the legal basis for the action taken on this entity, considering any manual overwrite.
*
* @return The legal basis.
*/
default String legalBasis() {
return getManualOverwrite().getLegalBasis().orElse(getMatchedRule().getLegalBasis());
return getManualOverwrite().getLegalBasis()
.orElse(getMatchedRule().getLegalBasis());
}
}
@@ -130,8 +130,16 @@ public class ManualChangeOverwrite {
if (manualChange instanceof ManualRecategorization recategorization) {
recategorized = true;
type = recategorization.getType();
if (recategorization.getLegalBasis() != null && !recategorization.getLegalBasis().isEmpty()) {
if (recategorization.getType() != null) {
type = recategorization.getType();
}
if (recategorization.getSection() != null) {
section = recategorization.getSection();
}
if (recategorization.getValue() != null) {
value = recategorization.getValue();
}
if (recategorization.getLegalBasis() != null) {
legalBasis = recategorization.getLegalBasis();
}
}
@@ -15,6 +15,9 @@ import lombok.EqualsAndHashCode;
import lombok.Getter;
import lombok.experimental.FieldDefaults;
/**
* Represents a rule that has been matched during the document redaction process.
*/
@Getter
@Builder
@AllArgsConstructor
@@ -25,7 +28,8 @@ public final class MatchedRule implements Comparable<MatchedRule> {
public static final RuleType FINAL_TYPE = RuleType.fromString("FINAL");
public static final RuleType ELIMINATION_RULE_TYPE = RuleType.fromString("X");
public static final RuleType IMPORTED_TYPE = RuleType.fromString("IMP");
private static final List<RuleType> RULE_TYPE_PRIORITIES = List.of(FINAL_TYPE, ELIMINATION_RULE_TYPE, IMPORTED_TYPE);
public static final RuleType DICTIONARY_TYPE = RuleType.fromString("DICT");
private static final List<RuleType> RULE_TYPE_PRIORITIES = List.of(FINAL_TYPE, ELIMINATION_RULE_TYPE, IMPORTED_TYPE, DICTIONARY_TYPE);
RuleIdentifier ruleIdentifier;
@Builder.Default
@@ -41,18 +45,33 @@ public final class MatchedRule implements Comparable<MatchedRule> {
Set<TextEntity> references = Collections.emptySet();
/**
* Creates an empty instance of {@link MatchedRule}.
* This can be used as a placeholder or when no rule is actually matched.
*
* @return An empty {@link MatchedRule} instance.
*/
public static MatchedRule empty() {
return MatchedRule.builder().ruleIdentifier(RuleIdentifier.empty()).build();
}
/**
* Returns a modified instance of {@link MatchedRule} based on its applied status.
* If the rule has been applied, it returns a new {@link MatchedRule} instance that retains all properties of the original
* except for the 'applied' status, which is set to false.
* If the rule has not been applied, it returns the original instance.
*
* @return A {@link MatchedRule} instance with 'applied' set to false.
*/
public MatchedRule asSkippedIfApplied() {
if (!this.isApplied()) {
return this;
}
return MatchedRule.builder().ruleIdentifier(getRuleIdentifier())
return MatchedRule.builder()
.ruleIdentifier(getRuleIdentifier())
.writeValueWithLineBreaks(this.isWriteValueWithLineBreaks())
.legalBasis(this.getLegalBasis())
.reason(this.getReason())
@@ -61,6 +80,13 @@ public final class MatchedRule implements Comparable<MatchedRule> {
}
/**
* Compares this rule with another {@link MatchedRule} to establish a priority order.
* The comparison is based on the rule type, unit, and ID, in that order.
*
* @param matchedRule The {@link MatchedRule} to compare against.
* @return A negative integer, zero, or a positive integer as this rule is less than, equal to, or greater than the specified rule.
*/
@Override
public int compareTo(MatchedRule matchedRule) {
@@ -97,7 +123,19 @@ public final class MatchedRule implements Comparable<MatchedRule> {
@Override
public String toString() {
return "MatchedRule[ruleIdentifier=" + ruleIdentifier + ", reason=" + reason + ", legalBasis=" + legalBasis + ", applied=" + applied + ", writeValueWithLineBreaks=" + writeValueWithLineBreaks + ", references=" + references + ']';
return "MatchedRule[ruleIdentifier="
+ ruleIdentifier
+ ", reason="
+ reason
+ ", legalBasis="
+ legalBasis
+ ", applied="
+ applied
+ ", writeValueWithLineBreaks="
+ writeValueWithLineBreaks
+ ", references="
+ references
+ ']';
}
}
@@ -40,7 +40,7 @@ public class TextEntity implements IEntity {
TextRange textRange;
@Builder.Default
List<TextRange> duplicateTextRanges = new ArrayList<>();
String type; // TODO: make final once ManualChangesApplicatioService recategorize is deleted
String type; // TODO: make final once ManualChangesApplicationService::recategorize is deleted
final EntityType entityType;
@Builder.Default
@@ -67,7 +67,13 @@ public class TextEntity implements IEntity {
public static TextEntity initialEntityNode(TextRange textRange, String type, EntityType entityType, SemanticNode node) {
return TextEntity.builder().id(buildId(node, textRange, type, entityType)).type(type).entityType(entityType).textRange(textRange).manualOverwrite(new ManualChangeOverwrite(entityType)).build();
return TextEntity.builder()
.id(buildId(node, textRange, type, entityType))
.type(type)
.entityType(entityType)
.textRange(textRange)
.manualOverwrite(new ManualChangeOverwrite(entityType))
.build();
}
@@ -80,7 +86,13 @@ public class TextEntity implements IEntity {
private static String buildId(SemanticNode node, TextRange textRange, String type, EntityType entityType) {
Map<Page, List<Rectangle2D>> rectanglesPerLinePerPage = node.getPositionsPerPage(textRange);
return IdBuilder.buildId(rectanglesPerLinePerPage.keySet(), rectanglesPerLinePerPage.values().stream().flatMap(Collection::stream).toList(), type, entityType.name());
return IdBuilder.buildId(rectanglesPerLinePerPage.keySet(),
rectanglesPerLinePerPage.values()
.stream()
.flatMap(Collection::stream)
.toList(),
type,
entityType.name());
}
@@ -89,15 +101,18 @@ public class TextEntity implements IEntity {
duplicateTextRanges.add(textRange);
}
public boolean occursInNodeOfType(Class<? extends SemanticNode> clazz) {
return intersectingNodes.stream().anyMatch(clazz::isInstance);
return intersectingNodes.stream()
.anyMatch(clazz::isInstance);
}
public boolean occursInNode(SemanticNode semanticNode) {
return intersectingNodes.stream().anyMatch(node -> node.equals(semanticNode));
return intersectingNodes.stream()
.anyMatch(node -> node.equals(semanticNode));
}
@@ -146,7 +161,10 @@ public class TextEntity implements IEntity {
.min(Comparator.comparingInt(Page::getNumber))
.orElseThrow(() -> new RuntimeException("No Positions found on any page!"));
positionsOnPagePerPage = rectanglesPerLinePerPage.entrySet().stream().map(entry -> buildPositionOnPage(firstPage, id, entry)).toList();
positionsOnPagePerPage = rectanglesPerLinePerPage.entrySet()
.stream()
.map(entry -> buildPositionOnPage(firstPage, id, entry))
.toList();
}
return positionsOnPagePerPage;
}
@@ -164,19 +182,37 @@ public class TextEntity implements IEntity {
public boolean containedBy(TextEntity textEntity) {
return this.textRange.containedBy(textEntity.getTextRange());
return this.textRange.containedBy(textEntity.getTextRange()) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> duplicateTextRange.containedBy(textEntity.textRange)) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
.stream()
.anyMatch(duplicateTextRange::containedBy));
}
public boolean contains(TextEntity textEntity) {
return this.textRange.contains(textEntity.getTextRange());
return this.textRange.contains(textEntity.getTextRange()) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> duplicateTextRange.contains(textEntity.textRange)) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
.stream()
.anyMatch(duplicateTextRange::contains));
}
public boolean intersects(TextEntity textEntity) {
return this.textRange.intersects(textEntity.getTextRange());
return this.textRange.intersects(textEntity.getTextRange()) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> duplicateTextRange.intersects(textEntity.textRange)) //
|| duplicateTextRanges.stream()
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
.stream()
.anyMatch(duplicateTextRange::intersects));
}
@@ -194,7 +230,8 @@ public class TextEntity implements IEntity {
public boolean matchesAnnotationId(String manualRedactionId) {
return getPositionsOnPagePerPage().stream().anyMatch(entityPosition -> entityPosition.getId().equals(manualRedactionId));
return getPositionsOnPagePerPage().stream()
.anyMatch(entityPosition -> entityPosition.getId().equals(manualRedactionId));
}
@@ -224,14 +261,16 @@ public class TextEntity implements IEntity {
@Override
public String type() {
return getManualOverwrite().getType().orElse(type);
return getManualOverwrite().getType()
.orElse(type);
}
@Override
public String value() {
return getManualOverwrite().getValue().orElse(getMatchedRule().isWriteValueWithLineBreaks() ? getValueWithLineBreaks() : value);
return getManualOverwrite().getValue()
.orElse(getMatchedRule().isWriteValueWithLineBreaks() ? getValueWithLineBreaks() : value);
}
}
@@ -24,6 +24,9 @@ import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
/**
* Represents the entire document as a node within the document's semantic structure.
*/
@Data
@Builder
@AllArgsConstructor
@@ -57,21 +60,32 @@ public class Document implements GenericSemanticNode {
public TextBlock getTextBlock() {
if (textBlock == null) {
textBlock = streamTerminalTextBlocksInOrder().collect(new TextBlockCollector());
textBlock = GenericSemanticNode.super.getTextBlock();
}
return textBlock;
}
/**
* Gets the main sections of the document as a list.
*
* @return A list of main sections within the document.
*/
public List<Section> getMainSections() {
return streamChildrenOfType(NodeType.SECTION).map(node -> (Section) node).collect(Collectors.toList());
return streamChildrenOfType(NodeType.SECTION).map(node -> (Section) node)
.collect(Collectors.toList());
}
/**
* Streams all terminal (leaf) text blocks within the document in their natural order.
*
* @return A stream of terminal {@link TextBlock}.
*/
public Stream<TextBlock> streamTerminalTextBlocksInOrder() {
return streamAllNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock);
return streamAllNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getTextBlock);
}
@@ -92,16 +106,29 @@ public class Document implements GenericSemanticNode {
@Override
public Headline getHeadline() {
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node).findFirst().orElseGet(Headline::empty);
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node)
.findFirst()
.orElseGet(Headline::empty);
}
/**
* Streams all nodes within the document, regardless of type, in their natural order.
*
* @return A stream of all {@link SemanticNode} within the document.
*/
private Stream<SemanticNode> streamAllNodes() {
return documentTree.allEntriesInOrder().map(DocumentTree.Entry::getNode);
return documentTree.allEntriesInOrder()
.map(DocumentTree.Entry::getNode);
}
/**
* Streams all image nodes contained within the document.
*
* @return A stream of {@link Image} nodes.
*/
public Stream<Image> streamAllImages() {
return streamAllSubNodesOfType(NodeType.IMAGE).map(node -> (Image) node);
@@ -0,0 +1,34 @@
package com.iqser.red.service.redaction.v1.server.model.document.nodes;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlockCollector;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.SuperBuilder;
@Data
@EqualsAndHashCode(callSuper = true)
@SuperBuilder
public class DuplicatedParagraph extends Paragraph {
TextBlock unsortedLeafTextBlock;
@Override
public TextBlock getTextBlock() {
return Stream.of(leafTextBlock, unsortedLeafTextBlock).collect(new TextBlockCollector());
}
@Override
public String toString() {
return super.toString();
}
}
@@ -19,6 +19,9 @@ import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
/**
* Represents the header part of a document page.
*/
@Data
@Builder
@AllArgsConstructor
@@ -20,6 +20,9 @@ import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
/**
* Represents a headline in a document.
*/
@Data
@Builder
@AllArgsConstructor
@@ -98,15 +101,27 @@ public class Headline implements GenericSemanticNode {
}
/**
* Creates an empty headline with no text content.
*
* @return An empty {@link Headline} instance.
*/
public static Headline empty() {
return Headline.builder().leafTextBlock(AtomicTextBlock.empty(-1L, 0, new Page(), -1, null)).build();
}
/**
* Checks if this headline is associated with any paragraphs within its parent section or node.
*
* @return True if there are paragraphs associated with this headline, false otherwise.
*/
public boolean hasParagraphs() {
return getParent().streamAllSubNodesOfType(NodeType.PARAGRAPH).findFirst().isPresent();
return getParent().streamAllSubNodesOfType(NodeType.PARAGRAPH)
.findFirst()
.isPresent();
}
}
@@ -28,6 +28,10 @@ import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
import lombok.experimental.FieldDefaults;
/**
*
Represents an image within the document.
*/
@Data
@Builder
@AllArgsConstructor
@@ -43,6 +47,8 @@ public class Image implements GenericSemanticNode, IEntity {
List<Integer> treeId;
String id;
TextBlock leafTextBlock;
ImageType imageType;
boolean transparent;
Rectangle2D position;
@@ -53,14 +59,11 @@ public class Image implements GenericSemanticNode, IEntity {
@Builder.Default
ManualChangeOverwrite manualOverwrite = new ManualChangeOverwrite();
@EqualsAndHashCode.Exclude
Page page;
@EqualsAndHashCode.Exclude
DocumentTree documentTree;
@Builder.Default
@EqualsAndHashCode.Exclude
Set<TextEntity> entities = new HashSet<>();
@@ -74,9 +77,7 @@ public class Image implements GenericSemanticNode, IEntity {
@Override
public TextBlock getTextBlock() {
return streamAllSubNodes().filter(SemanticNode::isLeaf)
.map(SemanticNode::getLeafTextBlock)
.collect(new TextBlockCollector());
return leafTextBlock;
}
@@ -90,22 +91,28 @@ public class Image implements GenericSemanticNode, IEntity {
@Override
public TextRange getTextRange() {
return GenericSemanticNode.super.getTextRange();
return leafTextBlock.getTextRange();
}
@Override
public int length() {
return getTextRange().length();
}
@Override
public String type() {
return getManualOverwrite().getType()
.orElse(imageType.toString());
return getManualOverwrite().getType().orElse(imageType.toString().toLowerCase(Locale.ENGLISH));
}
@Override
public String toString() {
return treeId + ": " + NodeType.IMAGE + ": " + imageType.toString() + " " + position;
return treeId + ": " + getValue() + " " + position;
}
@@ -136,7 +143,7 @@ public class Image implements GenericSemanticNode, IEntity {
Map<Page, Rectangle2D> bboxImage = image.getBBox();
Map<Page, Rectangle2D> bbox = this.getBBox();
//image needs to be on the same page
if(bboxImage.get(this.page) != null) {
if (bboxImage.get(this.page) != null) {
Rectangle2D intersection = bboxImage.get(this.page).createIntersection(bbox.get(this.page));
double calculatedIntersection = intersection.getWidth() * intersection.getHeight();
double area = bbox.get(this.page).getWidth() * bbox.get(this.page).getHeight();
@@ -156,10 +163,4 @@ public class Image implements GenericSemanticNode, IEntity {
return (area / calculatedIntersection) > containmentThreshold;
}
public int length() {
return 0;
}
}
@@ -6,20 +6,9 @@ public enum ImageType {
LOGO,
FORMULA,
SIGNATURE,
OTHER {
@Override
public String toString() {
return "image";
}
},
OCR;
public String toString() {
return name().toLowerCase(Locale.ENGLISH);
}
OTHER,
OCR,
GRAPHIC;
public static ImageType fromString(String imageType) {
@@ -29,6 +18,7 @@ public enum ImageType {
case "formula" -> ImageType.FORMULA;
case "signature" -> ImageType.SIGNATURE;
case "ocr" -> ImageType.OCR;
case "graphic" -> ImageType.GRAPHIC;
default -> ImageType.OTHER;
};
}
@@ -17,6 +17,9 @@ import lombok.NoArgsConstructor;
import lombok.Setter;
import lombok.experimental.FieldDefaults;
/**
* Represents a single page in a document.
*/
@Getter
@Setter
@Builder
@@ -43,9 +46,17 @@ public class Page {
Set<Image> images = new HashSet<>();
/**
* Constructs and returns a {@link TextBlock} representing the concatenated text of all leaf semantic nodes in the main body.
*
* @return The main body text block.
*/
public TextBlock getMainBodyTextBlock() {
return mainBody.stream().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
return mainBody.stream()
.filter(SemanticNode::isLeaf)
.map(SemanticNode::getLeafTextBlock)
.collect(new TextBlockCollector());
}
@@ -54,4 +65,5 @@ public class Page {
return String.valueOf(number);
}
}
@@ -17,11 +17,15 @@ import lombok.Builder;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
import lombok.experimental.SuperBuilder;
/**
* Represents a paragraph in the document.
*/
@Data
@Builder
@SuperBuilder
@AllArgsConstructor
@FieldDefaults(level = AccessLevel.PRIVATE)
@FieldDefaults(level = AccessLevel.PROTECTED)
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
public class Paragraph implements GenericSemanticNode {
@@ -21,6 +21,9 @@ import lombok.RequiredArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
/**
* Represents a section within a document, encapsulating both its textual content and semantic structure.
*/
@Slf4j
@Data
@Builder
@@ -51,9 +54,15 @@ public class Section implements GenericSemanticNode {
}
/**
* Checks if this section contains any tables.
*
* @return True if the section contains at least one table, false otherwise.
*/
public boolean hasTables() {
return streamAllSubNodesOfType(NodeType.TABLE).findAny().isPresent();
return streamAllSubNodesOfType(NodeType.TABLE).findAny()
.isPresent();
}
@@ -68,7 +77,7 @@ public class Section implements GenericSemanticNode {
public TextBlock getTextBlock() {
if (textBlock == null) {
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
textBlock = GenericSemanticNode.super.getTextBlock();
}
return textBlock;
}
@@ -90,12 +99,24 @@ public class Section implements GenericSemanticNode {
}
/**
* Checks if any headline within this section or its sub-nodes contains a given string.
*
* @param value The string to search for within headlines, case-sensitive.
* @return True if at least one headline contains the specified string, false otherwise.
*/
public boolean anyHeadlineContainsString(String value) {
return streamAllSubNodesOfType(NodeType.HEADLINE).anyMatch(h -> h.containsString(value));
}
/**
* Checks if any headline within this section or its sub-nodes contains a given string, case-insensitive.
*
* @param value The string to search for within headlines, case-insensitive.
* @return True if at least one headline contains the specified string, false otherwise.
*/
public boolean anyHeadlineContainsStringIgnoreCase(String value) {
return streamAllSubNodesOfType(NodeType.HEADLINE).anyMatch(h -> h.containsStringIgnoreCase(value));
@@ -10,6 +10,9 @@ import lombok.AccessLevel;
import lombok.AllArgsConstructor;
import lombok.experimental.FieldDefaults;
/**
* Represents a unique identifier for a section within a document.
*/
@AllArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class SectionIdentifier {
@@ -28,6 +31,12 @@ public class SectionIdentifier {
boolean asChild;
/**
* Generates a SectionIdentifier from the headline text of a section, determining its format and structure.
*
* @param headline The headline text from which to generate the section identifier.
* @return A {@link SectionIdentifier} instance corresponding to the headline text.
*/
public static SectionIdentifier fromSearchText(String headline) {
if (headline == null || headline.isEmpty() || headline.isBlank()) {
@@ -43,18 +52,34 @@ public class SectionIdentifier {
}
/**
* Marks the current section identifier as a child of another section.
*
* @param sectionIdentifier The parent section identifier.
* @return A new {@link SectionIdentifier} instance marked as a child.
*/
public static SectionIdentifier asChildOf(SectionIdentifier sectionIdentifier) {
return new SectionIdentifier(sectionIdentifier.format, sectionIdentifier.toString(), sectionIdentifier.identifiers, true);
}
/**
* Generates a SectionIdentifier that represents the entire document.
*
* @return A {@link SectionIdentifier} with a document-wide scope.
*/
public static SectionIdentifier document() {
return new SectionIdentifier(Format.DOCUMENT, "document", Collections.emptyList(), false);
}
/**
* Generates an empty SectionIdentifier.
*
* @return An empty {@link SectionIdentifier} instance.
*/
public static SectionIdentifier empty() {
return new SectionIdentifier(Format.EMPTY, "empty", Collections.emptyList(), false);
@@ -72,7 +97,11 @@ public class SectionIdentifier {
}
identifiers.add(Integer.parseInt(numericalIdentifier.trim()));
}
return new SectionIdentifier(Format.NUMERICAL, identifierString, identifiers.stream().toList(), false);
return new SectionIdentifier(Format.NUMERICAL,
identifierString,
identifiers.stream()
.toList(),
false);
}
@@ -105,6 +134,12 @@ public class SectionIdentifier {
}
/**
* Determines if the current section is a child of the given section, based on their identifiers.
*
* @param sectionIdentifier The section identifier to compare against.
* @return True if the current section is a child of the given section, false otherwise.
*/
public boolean isChildOf(SectionIdentifier sectionIdentifier) {
if (this.format.equals(Format.DOCUMENT) || this.format.equals(Format.EMPTY)) {
@@ -19,6 +19,7 @@ import com.iqser.red.service.redaction.v1.server.model.document.TextRange;
import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntity;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlockCollector;
import com.iqser.red.service.redaction.v1.server.service.document.NodeVisitor;
import com.iqser.red.service.redaction.v1.server.utils.RectangleTransformations;
import com.iqser.red.service.redaction.v1.server.utils.RedactionSearchUtility;
@@ -41,7 +42,12 @@ public interface SemanticNode {
*
* @return TextBlock containing all AtomicTextBlocks that are located under this Node.
*/
TextBlock getTextBlock();
default TextBlock getTextBlock() {
return streamAllSubNodes().filter(SemanticNode::isLeaf)
.map(SemanticNode::getTextBlock)
.collect(new TextBlockCollector());
}
/**
@@ -71,7 +77,10 @@ public interface SemanticNode {
*/
default Page getFirstPage() {
return getTextBlock().getPages().stream().min(Comparator.comparingInt(Page::getNumber)).orElseThrow();
return getTextBlock().getPages()
.stream()
.min(Comparator.comparingInt(Page::getNumber))
.orElseThrow();
}
@@ -97,7 +106,8 @@ public interface SemanticNode {
*/
default boolean onPage(int pageNumber) {
return getPages().stream().anyMatch(page -> page.getNumber() == pageNumber);
return getPages().stream()
.anyMatch(page -> page.getNumber() == pageNumber);
}
@@ -249,7 +259,9 @@ public interface SemanticNode {
*/
default boolean hasEntitiesOfType(String type) {
return getEntities().stream().filter(TextEntity::active).anyMatch(redactionEntity -> redactionEntity.type().equals(type));
return getEntities().stream()
.filter(TextEntity::active)
.anyMatch(redactionEntity -> redactionEntity.type().equals(type));
}
@@ -262,7 +274,10 @@ public interface SemanticNode {
*/
default boolean hasEntitiesOfAnyType(String... types) {
return getEntities().stream().filter(TextEntity::active).anyMatch(redactionEntity -> Arrays.stream(types).anyMatch(type -> redactionEntity.type().equals(type)));
return getEntities().stream()
.filter(TextEntity::active)
.anyMatch(redactionEntity -> Arrays.stream(types)
.anyMatch(type -> redactionEntity.type().equals(type)));
}
@@ -275,7 +290,12 @@ public interface SemanticNode {
*/
default boolean hasEntitiesOfAllTypes(String... types) {
return getEntities().stream().filter(TextEntity::active).map(TextEntity::type).collect(Collectors.toUnmodifiableSet()).containsAll(Arrays.stream(types).toList());
return getEntities().stream()
.filter(TextEntity::active)
.map(TextEntity::type)
.collect(Collectors.toUnmodifiableSet())
.containsAll(Arrays.stream(types)
.toList());
}
@@ -288,7 +308,10 @@ public interface SemanticNode {
*/
default List<TextEntity> getEntitiesOfType(String type) {
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.type().equals(type)).toList();
return getEntities().stream()
.filter(TextEntity::active)
.filter(redactionEntity -> redactionEntity.type().equals(type))
.toList();
}
@@ -301,7 +324,10 @@ public interface SemanticNode {
*/
default List<TextEntity> getEntitiesOfType(List<String> types) {
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.isAnyType(types)).toList();
return getEntities().stream()
.filter(TextEntity::active)
.filter(redactionEntity -> redactionEntity.isAnyType(types))
.toList();
}
@@ -314,7 +340,11 @@ public interface SemanticNode {
*/
default List<TextEntity> getEntitiesOfType(String... types) {
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.isAnyType(Arrays.stream(types).toList())).toList();
return getEntities().stream()
.filter(TextEntity::active)
.filter(redactionEntity -> redactionEntity.isAnyType(Arrays.stream(types)
.toList()))
.toList();
}
@@ -328,7 +358,8 @@ public interface SemanticNode {
TextBlock textBlock = getTextBlock();
if (!textBlock.getAtomicTextBlocks().isEmpty()) {
return getTextBlock().getAtomicTextBlocks().get(0).getNumberOnPage();
return getTextBlock().getAtomicTextBlocks()
.get(0).getNumberOnPage();
} else {
return -1;
}
@@ -357,14 +388,16 @@ public interface SemanticNode {
return getTextBlock().getSearchText().contains(string);
}
Set<LayoutEngine> getEngines();
default void addEngine(LayoutEngine engine) {
getEngines().add(engine);
}
/**
* Checks whether this SemanticNode contains all the provided Strings.
*
@@ -373,7 +406,8 @@ public interface SemanticNode {
*/
default boolean containsAllStrings(String... strings) {
return Arrays.stream(strings).allMatch(this::containsString);
return Arrays.stream(strings)
.allMatch(this::containsString);
}
@@ -385,7 +419,8 @@ public interface SemanticNode {
*/
default boolean containsAnyString(String... strings) {
return Arrays.stream(strings).anyMatch(this::containsString);
return Arrays.stream(strings)
.anyMatch(this::containsString);
}
@@ -397,15 +432,16 @@ public interface SemanticNode {
*/
default boolean containsAnyString(List<String> strings) {
return strings.stream().anyMatch(this::containsString);
return strings.stream()
.anyMatch(this::containsString);
}
/**
* Checks whether this SemanticNode contains all the provided Strings ignoring case.
* Checks whether this SemanticNode contains all the provided Strings case-insensitive.
*
* @param string A String which the TextBlock might contain
* @return true, if this node's TextBlock contains the string ignoring case
* @return true, if this node's TextBlock contains the string case-insensitive
*/
default boolean containsStringIgnoreCase(String string) {
@@ -414,26 +450,28 @@ public interface SemanticNode {
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
* Checks whether this SemanticNode contains any of the provided Strings case-insensitive.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
default boolean containsAnyStringIgnoreCase(String... strings) {
return Arrays.stream(strings).anyMatch(this::containsStringIgnoreCase);
return Arrays.stream(strings)
.anyMatch(this::containsStringIgnoreCase);
}
/**
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
* Checks whether this SemanticNode contains any of the provided Strings case-insensitive.
*
* @param strings A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the strings
*/
default boolean containsAllStringsIgnoreCase(String... strings) {
return Arrays.stream(strings).allMatch(this::containsStringIgnoreCase);
return Arrays.stream(strings)
.allMatch(this::containsStringIgnoreCase);
}
@@ -445,19 +483,24 @@ public interface SemanticNode {
*/
default boolean containsWord(String word) {
return getTextBlock().getWords().stream().anyMatch(s -> s.equals(word));
return getTextBlock().getWords()
.stream()
.anyMatch(s -> s.equals(word));
}
/**
* Checks whether this SemanticNode contains exactly the provided String as a word ignoring case.
* Checks whether this SemanticNode contains exactly the provided String as a word case-insensitive.
*
* @param word - String which the TextBlock might contain
* @return true, if this node's TextBlock contains string
*/
default boolean containsWordIgnoreCase(String word) {
return getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(s -> s.equals(word.toLowerCase(Locale.ENGLISH)));
return getTextBlock().getWords()
.stream()
.map(String::toLowerCase)
.anyMatch(s -> s.equals(word.toLowerCase(Locale.ENGLISH)));
}
@@ -469,19 +512,27 @@ public interface SemanticNode {
*/
default boolean containsAnyWord(String... words) {
return Arrays.stream(words).anyMatch(word -> getTextBlock().getWords().stream().anyMatch(word::equals));
return Arrays.stream(words)
.anyMatch(word -> getTextBlock().getWords()
.stream()
.anyMatch(word::equals));
}
/**
* Checks whether this SemanticNode contains any of the provided Strings as a word ignoring case.
* Checks whether this SemanticNode contains any of the provided Strings as a word case-insensitive.
*
* @param words - A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains any of the provided strings
*/
default boolean containsAnyWordIgnoreCase(String... words) {
return Arrays.stream(words).map(String::toLowerCase).anyMatch(word -> getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(word::equals));
return Arrays.stream(words)
.map(String::toLowerCase)
.anyMatch(word -> getTextBlock().getWords()
.stream()
.map(String::toLowerCase)
.anyMatch(word::equals));
}
@@ -493,19 +544,27 @@ public interface SemanticNode {
*/
default boolean containsAllWords(String... words) {
return Arrays.stream(words).allMatch(word -> getTextBlock().getWords().stream().anyMatch(word::equals));
return Arrays.stream(words)
.allMatch(word -> getTextBlock().getWords()
.stream()
.anyMatch(word::equals));
}
/**
* Checks whether this SemanticNode contains all the provided Strings as word ignoring case.
* Checks whether this SemanticNode contains all the provided Strings as word case-insensitive.
*
* @param words - A List of Strings which the TextBlock might contain
* @return true, if this node's TextBlock contains all the provided strings
*/
default boolean containsAllWordsIgnoreCase(String... words) {
return Arrays.stream(words).map(String::toLowerCase).allMatch(word -> getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(word::equals));
return Arrays.stream(words)
.map(String::toLowerCase)
.allMatch(word -> getTextBlock().getWords()
.stream()
.map(String::toLowerCase)
.anyMatch(word::equals));
}
@@ -522,10 +581,10 @@ public interface SemanticNode {
/**
* Checks whether this SemanticNode matches the provided regex pattern ignoring case.
* Checks whether this SemanticNode matches the provided regex pattern case-insensitive.
*
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
* @return true, if this node's TextBlock contains the regex pattern ignoring case
* @return true, if this node's TextBlock contains the regex pattern case-insensitive
*/
default boolean matchesRegexIgnoreCase(String regexPattern) {
@@ -545,7 +604,11 @@ public interface SemanticNode {
*/
default boolean intersectsRectangle(int x, int y, int w, int h, int pageNumber) {
return getBBox().entrySet().stream().filter(entry -> entry.getKey().getNumber() == pageNumber).map(Map.Entry::getValue).anyMatch(rect -> rect.intersects(x, y, w, h));
return getBBox().entrySet()
.stream()
.filter(entry -> entry.getKey().getNumber() == pageNumber)
.map(Map.Entry::getValue)
.anyMatch(rect -> rect.intersects(x, y, w, h));
}
@@ -563,7 +626,7 @@ public interface SemanticNode {
textEntity.setDeepestFullyContainingNode(this);
}
textEntity.addIntersectingNode(this);
streamChildren().filter(semanticNode -> semanticNode.getTextRange().intersects(textEntity.getTextRange()))
getDocumentTree().findIntersectingChildNodes(getTreeId(), textEntity.getTextRange())
.forEach(node -> node.addThisToEntityIfIntersects(textEntity));
}
}
@@ -598,7 +661,8 @@ public interface SemanticNode {
*/
default Stream<SemanticNode> streamAllSubNodes() {
return getDocumentTree().allSubEntriesInOrder(getTreeId()).map(DocumentTree.Entry::getNode);
return getDocumentTree().allSubEntriesInOrder(getTreeId())
.map(DocumentTree.Entry::getNode);
}
@@ -609,7 +673,9 @@ public interface SemanticNode {
*/
default Stream<SemanticNode> streamAllSubNodesOfType(NodeType nodeType) {
return getDocumentTree().allSubEntriesInOrder(getTreeId()).filter(entry -> entry.getType().equals(nodeType)).map(DocumentTree.Entry::getNode);
return getDocumentTree().allSubEntriesInOrder(getTreeId())
.filter(entry -> entry.getType().equals(nodeType))
.map(DocumentTree.Entry::getNode);
}
@@ -648,7 +714,7 @@ public interface SemanticNode {
if (isLeaf()) {
return getTextBlock().getPositionsPerPage(textRange);
}
Optional<SemanticNode> containingChildNode = streamChildren().filter(child -> child.getTextRange().contains(textRange)).findFirst();
Optional<SemanticNode> containingChildNode = getDocumentTree().findFirstContainingChild(getTreeId(), textRange);
if (containingChildNode.isEmpty()) {
return getTextBlock().getPositionsPerPage(textRange);
}
@@ -698,8 +764,12 @@ public interface SemanticNode {
private Map<Page, Rectangle2D> getBBoxFromChildren() {
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
List<Map<Page, Rectangle2D>> childrenBBoxes = streamChildren().map(SemanticNode::getBBox).toList();
Set<Page> pages = childrenBBoxes.stream().flatMap(map -> map.keySet().stream()).collect(Collectors.toSet());
List<Map<Page, Rectangle2D>> childrenBBoxes = streamChildren().map(SemanticNode::getBBox)
.toList();
Set<Page> pages = childrenBBoxes.stream()
.flatMap(map -> map.keySet()
.stream())
.collect(Collectors.toSet());
for (Page page : pages) {
Rectangle2D bBoxOnPage = childrenBBoxes.stream()
.filter(childBboxPerPage -> childBboxPerPage.containsKey(page))
@@ -717,7 +787,9 @@ public interface SemanticNode {
private Map<Page, Rectangle2D> getBBoxFromLeafTextBlock() {
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
Map<Page, List<AtomicTextBlock>> atomicTextBlockPerPage = getTextBlock().getAtomicTextBlocks().stream().collect(Collectors.groupingBy(AtomicTextBlock::getPage));
Map<Page, List<AtomicTextBlock>> atomicTextBlockPerPage = getTextBlock().getAtomicTextBlocks()
.stream()
.collect(Collectors.groupingBy(AtomicTextBlock::getPage));
atomicTextBlockPerPage.forEach((page, atomicTextBlocks) -> bBoxPerPage.put(page, RectangleTransformations.atomicTextBlockBBox(atomicTextBlocks)));
return bBoxPerPage;
}
@@ -26,6 +26,9 @@ import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
/**
* Represents a table within a document.
*/
@Data
@Builder
@AllArgsConstructor
@@ -408,9 +411,7 @@ public class Table implements SemanticNode {
public TextBlock getTextBlock() {
if (textBlock == null) {
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf)
.map(SemanticNode::getLeafTextBlock)
.collect(new TextBlockCollector());
textBlock = SemanticNode.super.getTextBlock();
}
return textBlock;
}
@@ -20,6 +20,9 @@ import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.experimental.FieldDefaults;
/**
* Represents a single table cell within a table.
*/
@Data
@Builder
@AllArgsConstructor
@@ -79,7 +82,9 @@ public class TableCell implements GenericSemanticNode {
}
if (textBlock == null) {
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf)
.map(SemanticNode::getLeafTextBlock)
.collect(new TextBlockCollector());
}
return textBlock;
}
@@ -61,6 +61,7 @@ public class AtomicTextBlock implements TextBlock {
return lineBreaks.size() + 1;
}
public static AtomicTextBlock empty(Long textBlockIdx, int stringOffset, Page page, int numberOnPage, SemanticNode parent) {
return AtomicTextBlock.builder()
@@ -77,10 +78,7 @@ public class AtomicTextBlock implements TextBlock {
}
public static AtomicTextBlock fromAtomicTextBlockData(DocumentTextData atomicTextBlockData,
DocumentPositionData atomicPositionBlockData,
SemanticNode parent,
Page page) {
public static AtomicTextBlock fromAtomicTextBlockData(DocumentTextData atomicTextBlockData, DocumentPositionData atomicPositionBlockData, SemanticNode parent, Page page) {
return AtomicTextBlock.builder()
.id(atomicTextBlockData.getId())
@@ -88,8 +86,10 @@ public class AtomicTextBlock implements TextBlock {
.page(page)
.textRange(new TextRange(atomicTextBlockData.getStart(), atomicTextBlockData.getEnd()))
.searchText(atomicTextBlockData.getSearchText())
.lineBreaks(Arrays.stream(atomicTextBlockData.getLineBreaks()).boxed().toList())
.stringIdxToPositionIdx(Arrays.stream(atomicPositionBlockData.getStringIdxToPositionIdx()).boxed().toList())
.lineBreaks(Arrays.stream(atomicTextBlockData.getLineBreaks()).boxed()
.toList())
.stringIdxToPositionIdx(Arrays.stream(atomicPositionBlockData.getStringIdxToPositionIdx()).boxed()
.toList())
.positions(toRectangle2DList(atomicPositionBlockData.getPositions()))
.parent(parent)
.build();
@@ -98,7 +98,9 @@ public class AtomicTextBlock implements TextBlock {
private static List<Rectangle2D> toRectangle2DList(float[][] positions) {
return Arrays.stream(positions).map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3])).toList();
return Arrays.stream(positions)
.map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3]))
.toList();
}
@@ -118,6 +120,7 @@ public class AtomicTextBlock implements TextBlock {
return new TextRange(lineBreaks.get(lineNumber - 1) + textRange.start(), lineBreaks.get(lineNumber) + textRange.start());
}
public List<String> getWords() {
if (words == null) {
@@ -144,9 +147,9 @@ public class AtomicTextBlock implements TextBlock {
public int getNextLinebreak(int fromIndex) {
return lineBreaks.stream()//
.filter(linebreak -> linebreak > fromIndex - textRange.start()) //
.findFirst() //
.orElse(searchText.length()) + textRange.start();
.filter(linebreak -> linebreak > fromIndex - textRange.start()) //
.findFirst() //
.orElse(searchText.length()) + textRange.start();
}
@@ -154,9 +157,9 @@ public class AtomicTextBlock implements TextBlock {
public int getPreviousLinebreak(int fromIndex) {
return lineBreaks.stream()//
.filter(linebreak -> linebreak <= fromIndex - textRange.start())//
.reduce((a, b) -> b)//
.orElse(0) + textRange.start();
.filter(linebreak -> linebreak <= fromIndex - textRange.start())//
.reduce((a, b) -> b)//
.orElse(0) + textRange.start();
}
@@ -209,7 +212,10 @@ public class AtomicTextBlock implements TextBlock {
return "";
}
Set<Integer> lbInBoundary = lineBreaks.stream().map(i -> i + textRange.start()).filter(textRange::contains).collect(Collectors.toSet());
Set<Integer> lbInBoundary = lineBreaks.stream()
.map(i -> i + textRange.start())
.filter(textRange::contains)
.collect(Collectors.toSet());
if (textRange.end() == getTextRange().end()) {
lbInBoundary.add(getTextRange().end());
}
@@ -235,7 +241,10 @@ public class AtomicTextBlock implements TextBlock {
private List<Integer> getAllLineBreaksInBoundary(TextRange textRange) {
return getLineBreaks().stream().map(linebreak -> linebreak + this.textRange.start()).filter(textRange::contains).toList();
return getLineBreaks().stream()
.map(linebreak -> linebreak + this.textRange.start())
.filter(textRange::contains)
.toList();
}
@@ -44,7 +44,8 @@ public class ConcatenatedTextBlock implements TextBlock {
this.atomicTextBlocks.add(firstTextBlock);
textRange = new TextRange(firstTextBlock.getTextRange().start(), firstTextBlock.getTextRange().end());
atomicTextBlocks.subList(1, atomicTextBlocks.size()).forEach(this::concat);
atomicTextBlocks.subList(1, atomicTextBlocks.size())
.forEach(this::concat);
}
@@ -65,7 +66,10 @@ public class ConcatenatedTextBlock implements TextBlock {
private AtomicTextBlock getAtomicTextBlockByStringIndex(int stringIdx) {
return atomicTextBlocks.stream().filter(textBlock -> textBlock.getTextRange().contains(stringIdx)).findAny().orElseThrow(IndexOutOfBoundsException::new);
return atomicTextBlocks.stream()
.filter(textBlock -> textBlock.getTextRange().contains(stringIdx))
.findAny()
.orElseThrow(IndexOutOfBoundsException::new);
}
@@ -99,14 +103,18 @@ public class ConcatenatedTextBlock implements TextBlock {
@Override
public List<String> getWords() {
return atomicTextBlocks.stream().map(AtomicTextBlock::getWords).flatMap(Collection::stream).toList();
return atomicTextBlocks.stream()
.map(AtomicTextBlock::getWords)
.flatMap(Collection::stream)
.toList();
}
@Override
public int numberOfLines() {
return atomicTextBlocks.stream().mapToInt(AtomicTextBlock::numberOfLines).sum();
return atomicTextBlocks.stream()
.mapToInt(AtomicTextBlock::numberOfLines).sum();
}
@@ -127,7 +135,10 @@ public class ConcatenatedTextBlock implements TextBlock {
@Override
public List<Integer> getLineBreaks() {
return getAtomicTextBlocks().stream().flatMap(atomicTextBlock -> atomicTextBlock.getLineBreaks().stream()).toList();
return getAtomicTextBlocks().stream()
.flatMap(atomicTextBlock -> atomicTextBlock.getLineBreaks()
.stream())
.toList();
}
@@ -202,7 +213,8 @@ public class ConcatenatedTextBlock implements TextBlock {
AtomicTextBlock lastTextBlock = textBlocks.get(textBlocks.size() - 1);
rectanglesPerLinePerPage = mergeEntityPositionsWithSamePageNode(rectanglesPerLinePerPage,
lastTextBlock.getPositionsPerPage(new TextRange(lastTextBlock.getTextRange().start(), stringTextRange.end())));
lastTextBlock.getPositionsPerPage(new TextRange(lastTextBlock.getTextRange().start(),
stringTextRange.end())));
return rectanglesPerLinePerPage;
}
@@ -239,7 +251,10 @@ public class ConcatenatedTextBlock implements TextBlock {
private Map<Page, List<Rectangle2D>> mergeEntityPositionsWithSamePageNode(Map<Page, List<Rectangle2D>> map1, Map<Page, List<Rectangle2D>> map2) {
Map<Page, List<Rectangle2D>> mergedMap = new HashMap<>(map1);
map2.forEach((pageNode, rectangles) -> mergedMap.merge(pageNode, rectangles, (l1, l2) -> Stream.concat(l1.stream(), l2.stream()).toList()));
map2.forEach((pageNode, rectangles) -> mergedMap.merge(pageNode,
rectangles,
(l1, l2) -> Stream.concat(l1.stream(), l2.stream())
.toList()));
return mergedMap;
}
@@ -18,8 +18,10 @@ public interface TextBlock extends CharSequence {
String getSearchText();
List<String> getWords();
List<AtomicTextBlock> getAtomicTextBlocks();
@@ -35,7 +37,6 @@ public interface TextBlock extends CharSequence {
TextRange getLineTextRange(int lineNumber);
List<Integer> getLineBreaks();
@@ -71,6 +72,7 @@ public interface TextBlock extends CharSequence {
return RectangleTransformations.rectangle2DBBox(getLinePositions(lineNumber));
}
default String searchTextWithLineBreaks() {
return subSequenceWithLineBreaks(getTextRange());
@@ -85,7 +87,9 @@ public interface TextBlock extends CharSequence {
default Set<Page> getPages() {
return getAtomicTextBlocks().stream().map(AtomicTextBlock::getPage).collect(Collectors.toUnmodifiableSet());
return getAtomicTextBlocks().stream()
.map(AtomicTextBlock::getPage)
.collect(Collectors.toUnmodifiableSet());
}
@@ -9,7 +9,8 @@ public record RuleClass(RuleType ruleType, List<RuleUnit> ruleUnits) {
public Optional<RuleUnit> findRuleUnitByInteger(Integer unit) {
return ruleUnits.stream()
.filter(ruleUnit -> Objects.equals(ruleUnit.unit(), unit)).findFirst();
.filter(ruleUnit -> Objects.equals(ruleUnit.unit(), unit))
.findFirst();
}
}
@@ -10,7 +10,7 @@ import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
import lombok.AccessLevel;
import lombok.AllArgsConstructor;
@@ -32,18 +32,22 @@ public final class RuleFileBluePrint {
int globalsLine;
List<BasicQuery> queries;
List<RuleClass> ruleClasses;
DroolsSyntaxValidation droolsSyntaxValidation;
DroolsValidation droolsValidation;
public Optional<RuleClass> findRuleClassByType(RuleType ruleType) {
return ruleClasses.stream().filter(ruleClass -> Objects.equals(ruleClass.ruleType(), ruleType)).findFirst();
return ruleClasses.stream()
.filter(ruleClass -> Objects.equals(ruleClass.ruleType(), ruleType))
.findFirst();
}
public Set<String> getImportSplitByKeyword() {
return Arrays.stream(imports.replaceAll("\n", "").split("import")).map(String::trim).collect(Collectors.toSet());
return Arrays.stream(imports.replaceAll("\n", "").split("import"))
.map(String::trim)
.collect(Collectors.toSet());
}
@@ -53,11 +57,15 @@ public final class RuleFileBluePrint {
return findRuleClassByType(ruleIdentifier.type()).map(RuleClass::ruleUnits)
.orElse(Collections.emptyList())
.stream()
.flatMap(ruleUnit -> ruleUnit.rules().stream().filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
.flatMap(ruleUnit -> ruleUnit.rules()
.stream()
.filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
.toList();
}
return findRuleClassByType(ruleIdentifier.type()).flatMap(ruleClass -> ruleClass.findRuleUnitByInteger(ruleIdentifier.unit()))
.map(ruleUnit -> ruleUnit.rules().stream().filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
.map(ruleUnit -> ruleUnit.rules()
.stream()
.filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
.orElse(Stream.empty())
.toList();
}
@@ -65,13 +73,18 @@ public final class RuleFileBluePrint {
public List<RuleIdentifier> getAllRuleIdentifiers() {
return streamAllRules().map(BasicRule::getIdentifier).collect(Collectors.toList());
return streamAllRules().map(BasicRule::getIdentifier)
.collect(Collectors.toList());
}
public Stream<BasicRule> streamAllRules() {
return getRuleClasses().stream().map(RuleClass::ruleUnits).flatMap(Collection::stream).map(RuleUnit::rules).flatMap(Collection::stream);
return getRuleClasses().stream()
.map(RuleClass::ruleUnits)
.flatMap(Collection::stream)
.map(RuleUnit::rules)
.flatMap(Collection::stream);
}
@@ -42,8 +42,8 @@ public record RuleIdentifier(@NonNull RuleType type, Integer unit, Integer id) {
public boolean matches(RuleIdentifier ruleIdentifier) {
return ruleIdentifier.type().equals(this.type()) && //
(Objects.isNull(ruleIdentifier.unit()) || Objects.isNull(this.unit()) || Objects.equals(this.unit(), ruleIdentifier.unit())) && //
(Objects.isNull(ruleIdentifier.id()) || Objects.isNull(this.id()) || Objects.equals(this.id(), ruleIdentifier.id()));
(Objects.isNull(ruleIdentifier.unit()) || Objects.isNull(this.unit()) || Objects.equals(this.unit(), ruleIdentifier.unit())) && //
(Objects.isNull(ruleIdentifier.id()) || Objects.isNull(this.id()) || Objects.equals(this.id(), ruleIdentifier.id()));
}
@@ -17,7 +17,7 @@ public class MessageReceiver {
@RabbitHandler
@RabbitListener(queues = REDACTION_QUEUE)
@RabbitListener(queues = REDACTION_QUEUE, concurrency = "1")
public void receiveAnalyzeRequest(Message message) {
redactionMessageReceiver.receiveAnalyzeRequest(message, false);
@@ -70,6 +70,7 @@ public class MessagingConfiguration {
.build();
}
@Bean
public Queue redactionAnalysisResponseQueue() {
@@ -17,7 +17,7 @@ public class PriorityMessageReceiver {
@RabbitHandler
@RabbitListener(queues = REDACTION_PRIORITY_QUEUE)
@RabbitListener(queues = REDACTION_PRIORITY_QUEUE, concurrency = "1")
public void receiveAnalyzeRequest(Message message) {
redactionMessageReceiver.receiveAnalyzeRequest(message, true);
@@ -51,14 +51,14 @@ public class RedactionMessageReceiver {
// This prevents from endless retries oom errors.
if (message.getMessageProperties().isRedelivered()) {
var errorMessage = format("Error during last processing of request with dossierId: %s and fileId: %s, do not retry.",
analyzeRequest.getDossierId(),
analyzeRequest.getFileId());
analyzeRequest.getDossierId(),
analyzeRequest.getFileId());
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
new FileErrorInfo(errorMessage,
priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE,
"redaction-service",
OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS)));
analyzeRequest.getFileId(),
new FileErrorInfo(errorMessage,
priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE,
"redaction-service",
OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS)));
throw new AmqpRejectAndDontRequeueException(errorMessage);
}
@@ -84,9 +84,9 @@ public class RedactionMessageReceiver {
log.debug(analyzeRequest.getManualRedactions().toString());
result = analyzeService.analyze(analyzeRequest);
log.info("Successfully analyzed dossier {} file {} took: {} s",
analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
format("%.2f", result.getDuration() / 1000.0));
analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
format("%.2f", result.getDuration() / 1000.0));
log.info("----------------------------------------------------------------------------------");
break;
@@ -96,9 +96,9 @@ public class RedactionMessageReceiver {
log.debug(analyzeRequest.getManualRedactions().toString());
result = analyzeService.reanalyze(analyzeRequest);
log.info("Successfully reanalyzed dossier {} file {} took: {} s",
analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
format("%.2f", result.getDuration() / 1000.0));
analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
format("%.2f", result.getDuration() / 1000.0));
log.info("----------------------------------------------------------------------------------");
break;
case SURROUNDING_TEXT_ANALYSIS:
@@ -106,9 +106,7 @@ public class RedactionMessageReceiver {
log.info("Starting Surrounding Text Analysis for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
log.debug(analyzeRequest.getManualRedactions().toString());
unprocessedChangesService.analyseSurroundingText(analyzeRequest);
log.info("Successful Surrounding Text Analysis dossier {} file {} ",
analyzeRequest.getDossierId(),
analyzeRequest.getFileId());
log.info("Successful Surrounding Text Analysis dossier {} file {} ", analyzeRequest.getDossierId(), analyzeRequest.getFileId());
log.info("-------------------------------------------------------------------------------------------------");
shouldRespond = false;
break;
@@ -137,8 +135,8 @@ public class RedactionMessageReceiver {
log.warn("Failed to process analyze request: {}", analyzeRequest, e);
var timestamp = OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS);
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
new FileErrorInfo(e.getMessage(), priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE, "redaction-service", timestamp));
analyzeRequest.getFileId(),
new FileErrorInfo(e.getMessage(), priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE, "redaction-service", timestamp));
}
@@ -153,8 +151,8 @@ public class RedactionMessageReceiver {
timestamp = timestamp != null ? timestamp : OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS);
log.info("Failed to process analyze request, errorCause: {}, timestamp: {}", errorCause, timestamp);
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
new FileErrorInfo(errorCause, REDACTION_DQL, "redaction-service", timestamp));
analyzeRequest.getFileId(),
new FileErrorInfo(errorCause, REDACTION_DQL, "redaction-service", timestamp));
}
}
@@ -1,5 +1,8 @@
package com.iqser.red.service.redaction.v1.server.service;
import static com.iqser.red.service.redaction.v1.server.service.document.SectionFinderService.getRelevantManuallyModifiedAnnotationIds;
import java.util.ArrayList;
import java.util.Collection;
import java.util.Collections;
import java.util.HashSet;
@@ -22,16 +25,9 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogChanges;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.legalbasis.LegalBasis;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogChanges;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogLegalBasis;
import com.iqser.red.service.redaction.v1.server.RedactionServiceSettings;
import com.iqser.red.service.redaction.v1.server.client.LegalBasisClient;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntitiesModel;
import com.iqser.red.service.redaction.v1.server.model.KieWrapper;
import com.iqser.red.service.redaction.v1.server.model.PrecursorEntity;
import com.iqser.red.service.redaction.v1.server.model.NerEntities;
import com.iqser.red.service.redaction.v1.server.model.component.Component;
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
@@ -87,7 +83,7 @@ public class AnalyzeService {
public AnalyzeResult reanalyze(@RequestBody AnalyzeRequest analyzeRequest) {
long startTime = System.currentTimeMillis();
EntityLog previousEntityLog = redactionStorageService.getEntityLog(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
EntityLog entityLogWithoutEntries = redactionStorageService.getEntityLogWithoutEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
log.info("Loaded previous entity log for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
Document document = DocumentGraphMapper.toDocumentGraph(observedStorageService.getDocumentData(analyzeRequest.getDossierId(), analyzeRequest.getFileId()));
@@ -97,25 +93,36 @@ public class AnalyzeService {
log.info("Loaded Imported Redactions for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
// not yet ready for reanalysis
if (previousEntityLog == null || document == null || document.getNumberOfPages() == 0) {
if (entityLogWithoutEntries == null || document == null || document.getNumberOfPages() == 0) {
return analyze(analyzeRequest);
}
DictionaryIncrement dictionaryIncrement = dictionaryService.getDictionaryIncrements(analyzeRequest.getDossierTemplateId(),
new DictionaryVersion(previousEntityLog.getDictionaryVersion(),
previousEntityLog.getDossierDictionaryVersion()),
new DictionaryVersion(entityLogWithoutEntries.getDictionaryVersion(),
entityLogWithoutEntries.getDossierDictionaryVersion()),
analyzeRequest.getDossierId());
Set<Integer> sectionsToReanalyseIds = getSectionsToReanalyseIds(analyzeRequest, previousEntityLog, document, dictionaryIncrement, importedRedactions);
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
Set<Integer> sectionsToReanalyseIds = redactionStorageService.findIdsOfSectionsToReanalyse(analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
relevantManuallyModifiedAnnotationIds);
sectionsToReanalyseIds.addAll(getSectionsToReanalyseIds(analyzeRequest,
document,
dictionaryIncrement,
importedRedactions,
relevantManuallyModifiedAnnotationIds));
List<SemanticNode> sectionsToReAnalyse = getSectionsToReAnalyse(document, sectionsToReanalyseIds);
log.info("{} Sections to reanalyze found for file {} in dossier {}", sectionsToReanalyseIds.size(), analyzeRequest.getFileId(), analyzeRequest.getDossierId());
if (sectionsToReAnalyse.isEmpty()) {
EntityLogChanges entityLogChanges = entityLogCreatorService.updateVersionsAndReturnChanges(previousEntityLog,
EntityLogChanges entityLogChanges = entityLogCreatorService.updateVersionsAndReturnChanges(entityLogWithoutEntries,
dictionaryIncrement.getDictionaryVersion(),
analyzeRequest,
false);
new ArrayList<>(),
new ArrayList<>());
return finalizeAnalysis(analyzeRequest,
startTime,
@@ -160,8 +167,8 @@ public class AnalyzeService {
EntityLogChanges entityLogChanges = entityLogCreatorService.updatePreviousEntityLog(analyzeRequest,
document,
entityLogWithoutEntries,
notFoundManualOrImportedEntries,
previousEntityLog,
sectionsToReanalyseIds,
dictionary.getVersion());
@@ -224,18 +231,18 @@ public class AnalyzeService {
nerEntities);
log.info("Finished entity rule execution for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
EntityLog entityLog = entityLogCreatorService.createInitialEntityLog(analyzeRequest,
document,
notFoundManualOrImportedEntries,
dictionary.getVersion(),
kieWrapperEntityRules.rulesVersion());
EntityLogChanges entityLogChanges = entityLogCreatorService.createInitialEntityLog(analyzeRequest,
document,
notFoundManualOrImportedEntries,
dictionary.getVersion(),
kieWrapperEntityRules.rulesVersion());
notFoundImportedEntitiesService.processEntityLog(entityLog, analyzeRequest, notFoundImportedEntries);
notFoundImportedEntitiesService.processEntityLog(entityLogChanges.getEntityLog(), analyzeRequest, notFoundImportedEntries);
return finalizeAnalysis(analyzeRequest,
startTime,
kieWrapperComponentRules,
new EntityLogChanges(entityLog, false),
entityLogChanges,
document,
document.getNumberOfPages(),
dictionary.getVersion(),
@@ -255,10 +262,25 @@ public class AnalyzeService {
Set<FileAttribute> addedFileAttributes) {
EntityLog entityLog = entityLogChanges.getEntityLog();
redactionStorageService.storeObject(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.ENTITY_LOG, entityLogChanges.getEntityLog());
// as workaround for duplicate key exceptions occurring due to simultaneous analyses and reanalyses save instead of insert is used
// also analysis numbers should be incremented in every follow-up request, so checking if the log exists is not needed
if (!redactionStorageService.entityLogExists(analyzeRequest.getDossierId(), analyzeRequest.getFileId())) {
redactionStorageService.saveEntityLog(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLog);
} else {
redactionStorageService.updateEntityLogWithoutEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLog);
if (!entityLogChanges.getNewEntityLogEntries().isEmpty()) {
redactionStorageService.saveEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLogChanges.getNewEntityLogEntries());
}
if (!entityLogChanges.getUpdatedEntityLogEntries().isEmpty()) {
redactionStorageService.updateEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLogChanges.getUpdatedEntityLogEntries());
}
}
log.info("Created entity log for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
if (entityLogChanges.isHasChanges() || !isReanalysis) {
if (entityLogChanges.hasChanges() || !isReanalysis) {
computeComponentsWhenRulesArePresent(analyzeRequest, kieWrapperComponentRules, document, addedFileAttributes, entityLogChanges, dictionaryVersion);
}
@@ -273,7 +295,7 @@ public class AnalyzeService {
.fileId(analyzeRequest.getFileId())
.duration(duration)
.numberOfPages(numberOfPages)
.hasUpdates(entityLogChanges.isHasChanges())
.hasUpdates(entityLogChanges.hasChanges())
.analysisVersion(redactionServiceSettings.getAnalysisVersion())
.analysisNumber(analyzeRequest.getAnalysisNumber())
.rulesVersion(entityLog.getRulesVersion())
@@ -323,12 +345,16 @@ public class AnalyzeService {
private Set<Integer> getSectionsToReanalyseIds(AnalyzeRequest analyzeRequest,
EntityLog entityLog,
Document document,
DictionaryIncrement dictionaryIncrement,
ImportedRedactions importedRedactions) {
ImportedRedactions importedRedactions,
Set<String> relevantManuallyModifiedAnnotationIds) {
return sectionFinderService.findSectionsToReanalyse(dictionaryIncrement, entityLog, document, analyzeRequest, importedRedactions);
return sectionFinderService.findSectionsToReanalyse(dictionaryIncrement,
document,
analyzeRequest,
importedRedactions,
relevantManuallyModifiedAnnotationIds);
}
@@ -23,13 +23,15 @@ public class ComponentLogCreatorService {
public ComponentLog buildComponentLog(int analysisNumber, List<Component> components, long componentRulesVersion) {
Map<String, List<ComponentLogEntryValue>> map = new HashMap<>();
components.stream().sorted(ComponentComparator.first()).forEach(component -> {
ComponentLogEntryValue componentLogEntryValue = buildComponentLogEntry(component);
map.computeIfAbsent(component.getName(), k -> new ArrayList<>()).add(componentLogEntryValue);
});
List<ComponentLogEntry> componentLogComponents = map
.entrySet()
.stream().map(entry -> new ComponentLogEntry(entry.getKey(), entry.getValue()))
components.stream()
.sorted(ComponentComparator.first())
.forEach(component -> {
ComponentLogEntryValue componentLogEntryValue = buildComponentLogEntry(component);
map.computeIfAbsent(component.getName(), k -> new ArrayList<>()).add(componentLogEntryValue);
});
List<ComponentLogEntry> componentLogComponents = map.entrySet()
.stream()
.map(entry -> new ComponentLogEntry(entry.getKey(), entry.getValue()))
.toList();
return new ComponentLog(analysisNumber, componentRulesVersion, componentLogComponents);
}
@@ -38,24 +40,36 @@ public class ComponentLogCreatorService {
private ComponentLogEntryValue buildComponentLogEntry(Component component) {
return ComponentLogEntryValue.builder()
.value(component.getValue()).originalValue(component.getValue())
.value(component.getValue())
.originalValue(component.getValue())
.componentRuleId(component.getMatchedRule().toString())
.valueDescription(component.getValueDescription())
.componentLogEntityReferences(toComponentEntityReferences(component.getReferences().stream().sorted(EntityComparators.first()).toList()))
.componentLogEntityReferences(toComponentEntityReferences(component.getReferences()
.stream()
.sorted(EntityComparators.first())
.toList()))
.build();
}
private List<ComponentLogEntityReference> toComponentEntityReferences(List<Entity> references) {
return references.stream().map(this::toComponentEntityReference).toList();
return references.stream()
.map(this::toComponentEntityReference)
.toList();
}
private ComponentLogEntityReference toComponentEntityReference(Entity entity) {
return ComponentLogEntityReference.builder().id(entity.getId())
.page(entity.getPositions().stream().findFirst().map(Position::getPageNumber).orElse(0)).entityRuleId(entity.getMatchedRule())
return ComponentLogEntityReference.builder()
.id(entity.getId())
.page(entity.getPositions()
.stream()
.findFirst()
.map(Position::getPageNumber)
.orElse(0))
.entityRuleId(entity.getMatchedRule())
.type(entity.getType())
.build();
}
@@ -6,10 +6,11 @@ import java.util.Set;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine;
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
import com.iqser.red.service.redaction.v1.server.model.dictionary.DictionaryModel;
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
import com.iqser.red.service.redaction.v1.server.model.document.entity.EntityType;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNode;
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
import com.iqser.red.service.redaction.v1.server.service.document.EntityCreationService;
import com.iqser.red.service.redaction.v1.server.service.document.EntityEnrichmentService;
@@ -38,10 +39,13 @@ public class DictionarySearchService {
@Observed(name = "DictionarySearchService", contextualName = "add-dictionary-entries")
public void addDictionaryEntities(Dictionary dictionary, SemanticNode node) {
for (var model : dictionary.getDictionaryModels()) {
for (DictionaryModel model : dictionary.getDictionaryModels()) {
bySearchImplementationAsDictionary(model.getEntriesSearch(), model.getType(), model.isHint() ? EntityType.HINT : EntityType.ENTITY, node, model.isDossierDictionary());
bySearchImplementationAsDictionary(model.getFalsePositiveSearch(), model.getType(), EntityType.FALSE_POSITIVE, node, model.isDossierDictionary());
bySearchImplementationAsDictionary(model.getFalseRecommendationsSearch(), model.getType(), EntityType.FALSE_RECOMMENDATION, node, model.isDossierDictionary());
if (model.isDossierDictionary()) {
bySearchImplementationAsDictionary(model.getDeletionEntriesSearch(), model.getType(), EntityType.DICTIONARY_REMOVAL, node, model.isDossierDictionary());
}
}
}
@@ -52,14 +56,16 @@ public class DictionarySearchService {
SemanticNode node,
boolean isDossierDictionaryEntry) {
Set<Engine> engines = isDossierDictionaryEntry ? Set.of(Engine.DOSSIER_DICTIONARY) : Set.of(Engine.DICTIONARY);
EntityCreationService entityCreationService = new EntityCreationService(entityEnrichmentService);
searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
.stream()
.filter(boundary -> entityCreationService.isValidEntityTextRange(node.getTextBlock(), boundary))
.forEach(bounds -> entityCreationService.byTextRangeWithEngine(bounds, type, entityType, node, Set.of(Engine.DICTIONARY)).ifPresent(entity -> {
entity.setDictionaryEntry(true);
entity.setDossierDictionaryEntry(isDossierDictionaryEntry);
}));
.forEach(bounds -> entityCreationService.byTextRangeWithEngine(bounds, type, entityType, node, engines)
.ifPresent(entity -> {
entity.setDictionaryEntry(true);
entity.setDossierDictionaryEntry(isDossierDictionaryEntry);
}));
}
}
@@ -4,6 +4,7 @@ import java.awt.Color;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashSet;
import java.util.LinkedList;
import java.util.List;
import java.util.Locale;
import java.util.Optional;
@@ -91,7 +92,8 @@ public class DictionaryService {
updateDictionaryEntry(dossierTemplateId, dossierDictionaryVersion, getVersion(dossierDictionary), dossierId);
}
return DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateDictionaryVersion).dossierVersion(dossierDictionaryVersion).build();
return DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateDictionaryVersion).dossierVersion(dossierDictionaryVersion)
.build();
}
@@ -106,41 +108,47 @@ public class DictionaryService {
List<DictionaryModel> dictionaryModels = getDossierTemplateDictionary(dossierTemplateId).getDictionary();
dictionaryModels.forEach(dictionaryModel -> {
dictionaryModel.getEntries().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getEntries()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
});
if (dossierDictionaryExists(dossierId)) {
dictionaryModels = getDossierDictionary(dossierId).getDictionary();
dictionaryModels.forEach(dictionaryModel -> {
dictionaryModel.getEntries().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getEntries()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalsePositives()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
dictionaryModel.getFalseRecommendations()
.forEach(dictionaryEntry -> {
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
}
});
});
}
@@ -155,84 +163,120 @@ public class DictionaryService {
DictionaryRepresentation dictionaryRepresentation = new DictionaryRepresentation();
var typeResponse = dossierId == null ? dictionaryClient.getAllTypesForDossierTemplate(dossierTemplateId, true) : dictionaryClient.getAllTypesForDossier(dossierId,
true);
true);
if (CollectionUtils.isNotEmpty(typeResponse)) {
List<DictionaryModel> dictionary = typeResponse.stream().map(t -> {
List<DictionaryModel> dictionary = typeResponse.stream()
.map(t -> {
Optional<DictionaryModel> optionalOldModel;
if (dossierId == null) {
var representation = getDossierTemplateDictionary(dossierTemplateId);
optionalOldModel = representation != null ? representation.getDictionary()
.stream()
.filter(f -> f.getType().equals(t.getType()))
.findAny() : Optional.empty();
} else {
var representation = getDossierDictionary(dossierId);
optionalOldModel = representation != null ? representation.getDictionary()
.stream()
.filter(f -> f.getType().equals(t.getType()))
.findAny() : Optional.empty();
}
Optional<DictionaryModel> optionalOldModel;
if (dossierId == null) {
var representation = getDossierTemplateDictionary(dossierTemplateId);
optionalOldModel = representation != null ? representation.getDictionary()
.stream()
.filter(f -> f.getType().equals(t.getType()))
.findAny() : Optional.empty();
} else {
var representation = getDossierDictionary(dossierId);
optionalOldModel = representation != null ? representation.getDictionary()
.stream()
.filter(f -> f.getType().equals(t.getType()))
.findAny() : Optional.empty();
}
Set<DictionaryEntryModel> entries = new HashSet<>();
Set<DictionaryEntryModel> falsePositives = new HashSet<>();
Set<DictionaryEntryModel> falseRecommendations = new HashSet<>();
Set<DictionaryEntryModel> entries = new HashSet<>();
Set<DictionaryEntryModel> falsePositives = new HashSet<>();
Set<DictionaryEntryModel> falseRecommendations = new HashSet<>();
DictionaryEntries newEntries = getEntries(t.getId(), currentVersion);
DictionaryEntries newEntries = getEntries(t.getId(), currentVersion);
var newValues = newEntries.getEntries().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
var newFalsePositivesValues = newEntries.getFalsePositives().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
var newFalseRecommendationsValues = newEntries.getFalseRecommendations().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
optionalOldModel.ifPresent(oldDictionaryModel -> {
});
if (optionalOldModel.isPresent()) {
var oldModel = optionalOldModel.get();
if (oldModel.isCaseInsensitive() && !t.isCaseInsensitive()) {
// add old entries from existing DictionaryModel but exclude lower case representation
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue())).toList());
falsePositives.addAll(oldModel.getFalsePositives()
var newValues = newEntries.getEntries()
.stream()
.filter(f -> !newFalsePositivesValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue()))
.toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations()
.map(DictionaryEntry::getValue)
.collect(Collectors.toSet());
var newFalsePositivesValues = newEntries.getFalsePositives()
.stream()
.filter(f -> !newFalseRecommendationsValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue()))
.toList());
} else if (!oldModel.isCaseInsensitive() && t.isCaseInsensitive()) {
// add old entries from existing DictionaryModel but exclude upper case representation
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.contains(f.getValue().toLowerCase(Locale.ROOT))).toList());
falsePositives.addAll(oldModel.getFalsePositives().stream().filter(f -> !newFalsePositivesValues.contains(f.getValue().toLowerCase(Locale.ROOT))).toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations()
.map(DictionaryEntry::getValue)
.collect(Collectors.toSet());
var newFalseRecommendationsValues = newEntries.getFalseRecommendations()
.stream()
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
.toList());
.map(DictionaryEntry::getValue)
.collect(Collectors.toSet());
} else {
// add old entries from existing DictionaryModel
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.contains(f.getValue())).toList());
falsePositives.addAll(oldModel.getFalsePositives().stream().filter(f -> !newFalsePositivesValues.contains(f.getValue())).toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations().stream().filter(f -> !newFalseRecommendationsValues.contains(f.getValue())).toList());
}
}
optionalOldModel.ifPresent(oldDictionaryModel -> {
// Add Increments
entries.addAll(newEntries.getEntries());
falsePositives.addAll(newEntries.getFalsePositives());
falseRecommendations.addAll(newEntries.getFalseRecommendations());
});
if (optionalOldModel.isPresent()) {
var oldModel = optionalOldModel.get();
if (oldModel.isCaseInsensitive() && !t.isCaseInsensitive()) {
// add old entries from existing DictionaryModel but exclude lower case representation
entries.addAll(oldModel.getEntries()
.stream()
.filter(f -> !newValues.stream()
.map(s -> s.toLowerCase(Locale.ROOT))
.toList().contains(f.getValue()))
.toList());
falsePositives.addAll(oldModel.getFalsePositives()
.stream()
.filter(f -> !newFalsePositivesValues.stream()
.map(s -> s.toLowerCase(Locale.ROOT))
.toList().contains(f.getValue()))
.toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations()
.stream()
.filter(f -> !newFalseRecommendationsValues.stream()
.map(s -> s.toLowerCase(Locale.ROOT))
.toList().contains(f.getValue()))
.toList());
} else if (!oldModel.isCaseInsensitive() && t.isCaseInsensitive()) {
// add old entries from existing DictionaryModel but exclude upper case representation
entries.addAll(oldModel.getEntries()
.stream()
.filter(f -> !newValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
.toList());
falsePositives.addAll(oldModel.getFalsePositives()
.stream()
.filter(f -> !newFalsePositivesValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
.toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations()
.stream()
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
.toList());
return new DictionaryModel(t.getType(),
t.getRank(),
convertColor(t.getHexColor()),
t.isCaseInsensitive(),
t.isHint(),
entries,
falsePositives,
falseRecommendations,
dossierId != null);
}).sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed()).collect(Collectors.toList());
} else {
// add old entries from existing DictionaryModel
entries.addAll(oldModel.getEntries()
.stream()
.filter(f -> !newValues.contains(f.getValue()))
.toList());
falsePositives.addAll(oldModel.getFalsePositives()
.stream()
.filter(f -> !newFalsePositivesValues.contains(f.getValue()))
.toList());
falseRecommendations.addAll(oldModel.getFalseRecommendations()
.stream()
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue()))
.toList());
}
}
// Add Increments
entries.addAll(newEntries.getEntries());
falsePositives.addAll(newEntries.getFalsePositives());
falseRecommendations.addAll(newEntries.getFalseRecommendations());
return new DictionaryModel(t.getType(),
t.getRank(),
convertColor(t.getHexColor()),
t.isCaseInsensitive(),
t.isHint(),
entries,
falsePositives,
falseRecommendations,
dossierId != null);
})
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
.collect(Collectors.toList());
dictionary.forEach(dm -> dictionaryRepresentation.getLocalAccessMap().put(dm.getType(), dm));
@@ -264,17 +308,17 @@ public class DictionaryService {
var type = dictionaryClient.getDictionaryForType(typeId, fromVersion);
Set<DictionaryEntryModel> entries = type.getEntries() != null ? new HashSet<>(type.getEntries()
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
Set<DictionaryEntryModel> falsePositives = type.getFalsePositiveEntries() != null ? new HashSet<>(type.getFalsePositiveEntries()
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
Set<DictionaryEntryModel> falseRecommendations = type.getFalseRecommendationEntries() != null ? new HashSet<>(type.getFalseRecommendationEntries()
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
.stream()
.map(DictionaryEntryModel::new)
.collect(Collectors.toSet())) : new HashSet<>();
if (type.isCaseInsensitive()) {
entries.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
@@ -282,10 +326,10 @@ public class DictionaryService {
falseRecommendations.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
}
log.debug("Dictionary update returned {} entries {} falsePositives and {} falseRecommendations for type {}",
entries.size(),
falsePositives.size(),
falseRecommendations.size(),
typeId);
entries.size(),
falsePositives.size(),
falseRecommendations.size(),
typeId);
return new DictionaryEntries(entries, falsePositives, falseRecommendations);
}
@@ -300,7 +344,8 @@ public class DictionaryService {
@SneakyThrows
public float[] getColor(String type, String dossierTemplateId) {
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap().get(type);
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap()
.get(type);
if (model != null) {
return model.getColor();
}
@@ -311,7 +356,8 @@ public class DictionaryService {
@SneakyThrows
public boolean isHint(String type, String dossierTemplateId) {
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap().get(type);
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap()
.get(type);
if (model != null) {
return model.isHint();
}
@@ -324,26 +370,33 @@ public class DictionaryService {
@Observed(name = "DictionaryService", contextualName = "deep-copy-dictionary")
public Dictionary getDeepCopyDictionary(String dossierTemplateId, String dossierId) {
List<DictionaryModel> mergedDictionaries;
List<DictionaryModel> mergedDictionaries = new LinkedList<>();
var dossierTemplateRepresentation = getDossierTemplateDictionary(dossierTemplateId);
var dossierTemplateDictionaries = dossierTemplateRepresentation.getDictionary();
DictionaryRepresentation dossierTemplateRepresentation = getDossierTemplateDictionary(dossierTemplateId);
List<DictionaryModel> dossierTemplateDictionaries = dossierTemplateRepresentation.getDictionary();
dossierTemplateDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
// merge dictionaries if they have same names
// add dossier
long dossierDictionaryVersion = -1;
if (dossierDictionaryExists(dossierId)) {
var dossierRepresentation = getDossierDictionary(dossierId);
var dossierDictionaries = dossierRepresentation.getDictionary();
mergedDictionaries = convertCommonsDictionaryModel(dictionaryMergeService.getMergedDictionary(convertDictionaryModel(dossierTemplateDictionaries),
convertDictionaryModel(dossierDictionaries)));
dossierDictionaryVersion = dossierRepresentation.getDictionaryVersion();
DictionaryRepresentation dossierRepresentation = getDossierDictionary(dossierId);
List<DictionaryModel> dossierDictionaries = dossierRepresentation.getDictionary();
dossierDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
return getDictionary(mergedDictionaries, dossierTemplateRepresentation, dossierRepresentation.getDictionaryVersion());
} else {
mergedDictionaries = new ArrayList<>();
dossierTemplateDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
return getDictionary(mergedDictionaries, dossierTemplateRepresentation, dossierDictionaryVersion);
}
return new Dictionary(mergedDictionaries.stream().sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed()).collect(Collectors.toList()),
DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateRepresentation.getDictionaryVersion()).dossierVersion(dossierDictionaryVersion).build());
}
private Dictionary getDictionary(List<DictionaryModel> mergedDictionaries, DictionaryRepresentation dossierTemplateRepresentation, long dossierDictionaryVersion) {
return new Dictionary(mergedDictionaries.stream()
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
.collect(Collectors.toList()),
DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateRepresentation.getDictionaryVersion()).dossierVersion(dossierDictionaryVersion)
.build());
}
@@ -371,14 +424,16 @@ public class DictionaryService {
@SneakyThrows
private DictionaryRepresentation getDossierTemplateDictionary(String dossierTemplateId) {
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossierTemplate().get(dossierTemplateId);
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossierTemplate()
.get(dossierTemplateId);
}
@SneakyThrows
private DictionaryRepresentation getDossierDictionary(String dossierId) {
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossier().get(dossierId);
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossier()
.get(dossierId);
}
@@ -421,14 +476,14 @@ public class DictionaryService {
return commonsDictionaries.stream()
.map(cd -> new DictionaryModel(cd.getType(),
cd.getRank(),
cd.getColor(),
cd.isCaseInsensitive(),
cd.isHint(),
cd.getEntries(),
cd.getFalsePositives(),
cd.getFalseRecommendations(),
cd.isDossierDictionary()))
cd.getRank(),
cd.getColor(),
cd.isCaseInsensitive(),
cd.isHint(),
cd.getEntries(),
cd.getFalsePositives(),
cd.getFalseRecommendations(),
cd.isDossierDictionary()))
.collect(Collectors.toList());
}
@@ -1,7 +1,7 @@
package com.iqser.red.service.redaction.v1.server.service;
import java.time.OffsetDateTime;
import java.util.Comparator;
import java.util.ArrayList;
import java.util.List;
import java.util.Optional;
import java.util.Set;
@@ -13,9 +13,6 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ChangeType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryState;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualChange;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
import io.micrometer.core.annotation.Timed;
import lombok.AccessLevel;
@@ -30,75 +27,60 @@ import lombok.extern.slf4j.Slf4j;
public class EntityChangeLogService {
@Timed("redactmanager_computeChanges")
public boolean computeChanges(List<EntityLogEntry> previousEntityLogEntries, List<EntityLogEntry> newEntityLogEntries, ManualRedactions manualRedactions, int analysisNumber) {
public EntryChanges computeChanges(List<EntityLogEntry> previousEntityLogEntries, List<EntityLogEntry> newEntityLogEntries, int analysisNumber) {
var now = OffsetDateTime.now();
if (previousEntityLogEntries.isEmpty()) {
newEntityLogEntries.forEach(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.ADDED, now)));
return true;
return new EntryChanges(newEntityLogEntries, new ArrayList<>());
}
boolean hasChanges = false;
List<EntityLogEntry> toInsert = new ArrayList<>();
List<EntityLogEntry> toUpdate = new ArrayList<>();
for (EntityLogEntry entityLogEntry : newEntityLogEntries) {
Optional<EntityLogEntry> optionalPreviousEntity = previousEntityLogEntries.stream()
.filter(entry -> entry.getId().equals(entityLogEntry.getId()))
.findAny();
if (optionalPreviousEntity.isEmpty()) {
hasChanges = true;
entityLogEntry.getChanges().add(new Change(analysisNumber, ChangeType.ADDED, now));
toInsert.add(entityLogEntry);
continue;
}
EntityLogEntry previousEntity = optionalPreviousEntity.get();
entityLogEntry.getChanges().addAll(previousEntity.getChanges());
if (!previousEntity.getState().equals(entityLogEntry.getState())) {
hasChanges = true;
ChangeType changeType = calculateChangeType(entityLogEntry.getState(), previousEntity.getState());
entityLogEntry.getChanges().add(new Change(analysisNumber, changeType, now));
if (!previousEntity.equals(entityLogEntry)) {
if(!previousEntity.getState().equals(entityLogEntry.getState())) {
ChangeType changeType = calculateChangeType(entityLogEntry.getState(), previousEntity.getState());
entityLogEntry.getChanges().add(new Change(analysisNumber, changeType, now));
}
toUpdate.add(entityLogEntry);
}
}
addRemovedEntriesAsRemoved(previousEntityLogEntries, newEntityLogEntries, manualRedactions, analysisNumber, now);
return hasChanges;
toUpdate.addAll(addRemovedEntriesAsRemoved(previousEntityLogEntries, newEntityLogEntries, analysisNumber, now));
return new EntryChanges(toInsert, toUpdate);
}
private void addRemovedEntriesAsRemoved(List<EntityLogEntry> previousEntityLogEntries,
List<EntityLogEntry> newEntityLogEntries,
ManualRedactions manualRedactions,
int analysisNumber,
OffsetDateTime now) {
private List<EntityLogEntry> addRemovedEntriesAsRemoved(List<EntityLogEntry> previousEntityLogEntries,
List<EntityLogEntry> newEntityLogEntries,
int analysisNumber,
OffsetDateTime now) {
Set<String> existingIds = newEntityLogEntries.stream()
.map(EntityLogEntry::getId)
.collect(Collectors.toSet());
List<EntityLogEntry> removedEntries = previousEntityLogEntries.stream()
.filter(entry -> !existingIds.contains(entry.getId()))
.collect(Collectors.toList());
List<EntityLogEntry> removedDossierRedaction = removedEntries.stream()
.filter(e -> e.getState() == EntryState.REMOVED && e.getType().equals("dossier_redaction"))
.toList();
previousEntityLogEntries.removeAll(removedDossierRedaction);
removedEntries.removeAll(removedDossierRedaction);
removedEntries.forEach(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.REMOVED, now)));
removedEntries.forEach(entry -> entry.setState(EntryState.REMOVED));
removedEntries.forEach(entry -> addManualChangeForDictionaryRemovals(entry, manualRedactions));
removedEntries.stream()
.filter(entry -> !entry.getState().equals(EntryState.REMOVED))
.peek(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.REMOVED, now)))
.forEach(entry -> entry.setState(EntryState.REMOVED));
newEntityLogEntries.addAll(removedEntries);
}
private void addManualChangeForDictionaryRemovals(EntityLogEntry entry, ManualRedactions manualRedactions) {
if (manualRedactions == null || manualRedactions.getIdsToRemove().isEmpty()) {
return;
}
manualRedactions.getIdsToRemove()
.stream()
.filter(IdRemoval::isRemoveFromDictionary)//
.filter(removed -> removed.getAnnotationId().equals(entry.getId()))//
.findFirst()//
.ifPresent(idRemove -> entry.getManualChanges().add(ManualChangeFactory.toManualChange(idRemove, false)));
return removedEntries;
}
@@ -122,4 +104,9 @@ public class EntityChangeLogService {
return (state.equals(EntryState.REMOVED) || state.equals(EntryState.IGNORED));
}
public record EntryChanges(List<EntityLogEntry> inserted, List<EntityLogEntry> updated) {
}
}
@@ -19,6 +19,7 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryState;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Position;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualChangeFactory;
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.legalbasis.LegalBasis;
import com.iqser.red.service.redaction.v1.server.RedactionServiceSettings;
import com.iqser.red.service.redaction.v1.server.client.LegalBasisClient;
@@ -32,6 +33,7 @@ import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntit
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Document;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
import com.iqser.red.service.redaction.v1.server.service.EntityChangeLogService.EntryChanges;
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
import lombok.AccessLevel;
@@ -52,17 +54,19 @@ public class EntityLogCreatorService {
RedactionStorageService redactionStorageService;
private static boolean notFalsePositiveOrFalseRecommendation(TextEntity textEntity) {
private static boolean notFalsePositiveOrFalseRecommendationOrRemoval(TextEntity textEntity) {
return !(textEntity.getEntityType().equals(EntityType.FALSE_POSITIVE) || textEntity.getEntityType().equals(EntityType.FALSE_RECOMMENDATION));
return !(textEntity.getEntityType().equals(EntityType.FALSE_POSITIVE) //
|| textEntity.getEntityType().equals(EntityType.FALSE_RECOMMENDATION) //
|| textEntity.getEntityType().equals(EntityType.DICTIONARY_REMOVAL));
}
public EntityLog createInitialEntityLog(AnalyzeRequest analyzeRequest,
Document document,
List<PrecursorEntity> notFoundEntities,
DictionaryVersion dictionaryVersion,
long rulesVersion) {
public EntityLogChanges createInitialEntityLog(AnalyzeRequest analyzeRequest,
Document document,
List<PrecursorEntity> notFoundEntities,
DictionaryVersion dictionaryVersion,
long rulesVersion) {
List<EntityLogEntry> entityLogEntries = createEntityLogEntries(document, analyzeRequest, notFoundEntities);
@@ -70,16 +74,20 @@ public class EntityLogCreatorService {
List<EntityLogEntry> previousExistingEntityLogEntries = getPreviousEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
entityChangeLogService.computeChanges(previousExistingEntityLogEntries, entityLogEntries, analyzeRequest.getManualRedactions(), analyzeRequest.getAnalysisNumber());
EntryChanges entryChanges = entityChangeLogService.computeChanges(previousExistingEntityLogEntries, entityLogEntries, analyzeRequest.getAnalysisNumber());
return new EntityLog(redactionServiceSettings.getAnalysisVersion(),
analyzeRequest.getAnalysisNumber(),
entityLogEntries,
toEntityLogLegalBasis(legalBasis),
dictionaryVersion.getDossierTemplateVersion(),
dictionaryVersion.getDossierVersion(),
rulesVersion,
legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId()));
return EntityLogChanges.builder()
.entityLog(new EntityLog(redactionServiceSettings.getAnalysisVersion(),
analyzeRequest.getAnalysisNumber(),
entityLogEntries,
toEntityLogLegalBasis(legalBasis),
dictionaryVersion.getDossierTemplateVersion(),
dictionaryVersion.getDossierVersion(),
rulesVersion,
legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId())))
.updatedEntityLogEntries(entryChanges.updated())
.newEntityLogEntries(entryChanges.inserted())
.build();
}
@@ -93,7 +101,11 @@ public class EntityLogCreatorService {
}
public EntityLogChanges updateVersionsAndReturnChanges(EntityLog entityLog, DictionaryVersion dictionaryVersion, AnalyzeRequest analyzeRequest, boolean hasChanges) {
public EntityLogChanges updateVersionsAndReturnChanges(EntityLog entityLog,
DictionaryVersion dictionaryVersion,
AnalyzeRequest analyzeRequest,
List<EntityLogEntry> newEntries,
List<EntityLogEntry> updatedEntries) {
List<LegalBasis> legalBasis = legalBasisClient.getLegalBasisMapping(analyzeRequest.getDossierTemplateId());
entityLog.setLegalBasisVersion(legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId()));
@@ -102,14 +114,14 @@ public class EntityLogCreatorService {
entityLog.setDossierDictionaryVersion(dictionaryVersion.getDossierVersion());
entityLog.setAnalysisNumber(analyzeRequest.getAnalysisNumber());
return new EntityLogChanges(entityLog, hasChanges);
return EntityLogChanges.builder().entityLog(entityLog).newEntityLogEntries(newEntries).updatedEntityLogEntries(updatedEntries).build();
}
public EntityLogChanges updatePreviousEntityLog(AnalyzeRequest analyzeRequest,
Document document,
EntityLog entityLogWithoutEntries,
List<PrecursorEntity> notFoundEntries,
EntityLog previousEntityLog,
Set<Integer> sectionsToReanalyseIds,
DictionaryVersion dictionaryVersion) {
@@ -117,24 +129,14 @@ public class EntityLogCreatorService {
.filter(entry -> entry.getContainingNodeId().isEmpty() || sectionsToReanalyseIds.contains(entry.getContainingNodeId()
.get(0)))
.collect(Collectors.toList());
Set<String> newEntityIds = newEntityLogEntries.stream()
.map(EntityLogEntry::getId)
.collect(Collectors.toSet());
List<EntityLogEntry> previousEntriesFromReAnalyzedSections = previousEntityLog.getEntityLogEntry()
.stream()
.filter(entry -> (newEntityIds.contains(entry.getId()) || entry.getContainingNodeId().isEmpty() || sectionsToReanalyseIds.contains(entry.getContainingNodeId()
.get(0))))
.collect(Collectors.toList());
previousEntityLog.getEntityLogEntry().removeAll(previousEntriesFromReAnalyzedSections);
List<EntityLogEntry> previousEntriesFromReAnalyzedSections = redactionStorageService.findEntriesContainedBySectionsOrNotContained(analyzeRequest.getDossierId(),
analyzeRequest.getFileId(),
sectionsToReanalyseIds);
boolean hasChanges = entityChangeLogService.computeChanges(previousEntriesFromReAnalyzedSections,
newEntityLogEntries,
analyzeRequest.getManualRedactions(),
analyzeRequest.getAnalysisNumber());
previousEntityLog.getEntityLogEntry().addAll(newEntityLogEntries);
EntryChanges entryChanges = entityChangeLogService.computeChanges(previousEntriesFromReAnalyzedSections, newEntityLogEntries, analyzeRequest.getAnalysisNumber());
return updateVersionsAndReturnChanges(previousEntityLog, dictionaryVersion, analyzeRequest, hasChanges);
return updateVersionsAndReturnChanges(entityLogWithoutEntries, dictionaryVersion, analyzeRequest, entryChanges.inserted(), entryChanges.updated());
}
@@ -147,7 +149,7 @@ public class EntityLogCreatorService {
document.getEntities()
.stream()
.filter(entity -> !entity.getValue().isEmpty())
.filter(EntityLogCreatorService::notFalsePositiveOrFalseRecommendation)
.filter(EntityLogCreatorService::notFalsePositiveOrFalseRecommendationOrRemoval)
.filter(entity -> !entity.removed())
.forEach(entityNode -> entries.addAll(toEntityLogEntries(entityNode)));
document.streamAllImages()
@@ -186,11 +188,12 @@ public class EntityLogCreatorService {
private EntityLogEntry createEntityLogEntry(Image image, String dossierTemplateId) {
boolean isHint = dictionaryService.isHint(image.type(), dossierTemplateId);
String imageType = image.getImageType().equals(ImageType.OTHER) ? "image" : image.getImageType().toString().toLowerCase(Locale.ENGLISH);
boolean isHint = dictionaryService.isHint(imageType, dossierTemplateId);
return EntityLogEntry.builder()
.id(image.getId())
.value(image.value())
.type(image.type())
.value(image.getValue())
.type(imageType)
.reason(image.buildReasonWithManualChangeDescriptions())
.legalBasis(image.legalBasis())
.matchedRule(image.getMatchedRule().getRuleIdentifier().toString())
@@ -201,7 +204,7 @@ public class EntityLogCreatorService {
.section(image.getManualOverwrite().getSection()
.orElse(image.getParent().toString()))
.imageHasTransparency(image.isTransparent())
.manualChanges(ManualChangeFactory.toManualChangeList(image.getManualOverwrite().getManualChangeLog(), isHint))
.manualChanges(ManualChangeFactory.toLocalManualChangeList(image.getManualOverwrite().getManualChangeLog(), true))
.state(buildEntryState(image))
.entryType(isHint ? EntryType.IMAGE_HINT : EntryType.IMAGE)
.engines(getEngines(null, image.getManualOverwrite()))
@@ -244,7 +247,7 @@ public class EntityLogCreatorService {
//(was .imported(precursorEntity.getEngines() != null && precursorEntity.getEngines().contains(Engine.IMPORTED)))
.imported(false)
.reference(Collections.emptySet())
.manualChanges(ManualChangeFactory.toManualChangeList(precursorEntity.getManualOverwrite().getManualChangeLog(), isHint))
.manualChanges(ManualChangeFactory.toLocalManualChangeList(precursorEntity.getManualOverwrite().getManualChangeLog(), true))
.build();
}
@@ -280,7 +283,7 @@ public class EntityLogCreatorService {
//(was .imported(entity.getEngines() != null && entity.getEngines().contains(Engine.IMPORTED)))
.imported(false)
.reference(referenceIds)
.manualChanges(ManualChangeFactory.toManualChangeList(entity.getManualOverwrite().getManualChangeLog(), isHint))
.manualChanges(ManualChangeFactory.toLocalManualChangeList(entity.getManualOverwrite().getManualChangeLog(), true))
.state(buildEntryState(entity))
.entryType(buildEntryType(entity))
.build();
@@ -342,6 +345,7 @@ public class EntityLogCreatorService {
case FALSE_POSITIVE -> EntryType.FALSE_POSITIVE;
case RECOMMENDATION -> EntryType.RECOMMENDATION;
case FALSE_RECOMMENDATION -> EntryType.FALSE_RECOMMENDATION;
case DICTIONARY_REMOVAL -> EntryType.FALSE_POSITIVE;
};
}
@@ -1,55 +0,0 @@
package com.iqser.red.service.redaction.v1.server.service;
import java.time.OffsetDateTime;
import java.util.List;
import java.util.stream.Collectors;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualChange;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualRedactionType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.BaseAnnotation;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualForceRedaction;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualLegalBasisChange;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRecategorization;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRedactionEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualResizeRedaction;
import lombok.experimental.UtilityClass;
@UtilityClass
public class ManualChangeFactory {
public List<ManualChange> toManualChangeList(List<BaseAnnotation> manualChanges, boolean isHint) {
return manualChanges.stream()
.map(baseAnnotation -> toManualChange(baseAnnotation, isHint))
.collect(Collectors.toList());
}
public ManualChange toManualChange(BaseAnnotation baseAnnotation, boolean isHint) {
ManualChange manualChange = ManualChange.from(baseAnnotation);
if (baseAnnotation instanceof ManualRecategorization imageRecategorization) {
manualChange.withManualRedactionType(ManualRedactionType.RECATEGORIZE).withChange("type", imageRecategorization.getType());
} else if (baseAnnotation instanceof IdRemoval manualRemoval) {
manualChange.withManualRedactionType(manualRemoval.isRemoveFromDictionary() ? ManualRedactionType.REMOVE_FROM_DICTIONARY : ManualRedactionType.REMOVE);
} else if (baseAnnotation instanceof ManualForceRedaction manualForceRedaction) {
manualChange.withManualRedactionType(ManualRedactionType.FORCE).withChange("legalBasis", manualForceRedaction.getLegalBasis());
} else if (baseAnnotation instanceof ManualResizeRedaction manualResizeRedact) {
manualChange.withManualRedactionType(manualResizeRedact.getUpdateDictionary() ? ManualRedactionType.RESIZE_IN_DICTIONARY : ManualRedactionType.RESIZE)
.withChange("value", manualResizeRedact.getValue());
} else if (baseAnnotation instanceof ManualRedactionEntry manualRedactionEntry) {
manualChange.withManualRedactionType(manualRedactionEntry.isAddToDictionary() ? ManualRedactionType.ADD_TO_DICTIONARY : ManualRedactionType.ADD)
.withChange("value", manualRedactionEntry.getValue());
} else if (baseAnnotation instanceof ManualLegalBasisChange manualLegalBasisChange) {
manualChange.withManualRedactionType(ManualRedactionType.LEGAL_BASIS_CHANGE)
.withChange("section", manualLegalBasisChange.getSection())
.withChange("value", manualLegalBasisChange.getValue())
.withChange("legalBasis", manualLegalBasisChange.getLegalBasis());
}
manualChange.setProcessedDate(OffsetDateTime.now());
return manualChange;
}
}
@@ -47,6 +47,10 @@ public class ManualChangesApplicationService {
entityToBeReCategorized.getMatchedRuleList().clear();
entityToBeReCategorized.getManualOverwrite().addChange(manualRecategorization);
if (manualRecategorization.getType() == null) {
return;
}
if (entityToBeReCategorized instanceof Image image) {
image.setImageType(ImageType.fromString(manualRecategorization.getType()));
return;
@@ -58,6 +62,12 @@ public class ManualChangesApplicationService {
}
/**
* Resizes a text entity based on manual resize redaction details.
*
* @param entityToBeResized The entity to resize.
* @param manualResizeRedaction The details of the resize operation.
*/
public void resize(TextEntity entityToBeResized, ManualResizeRedaction manualResizeRedaction) {
resizeEntityAndReinsert(entityToBeResized, manualResizeRedaction);
@@ -74,9 +84,9 @@ public class ManualChangesApplicationService {
.orElseThrow(() -> new NoSuchElementException("No redaction position with matching annotation id found!"));
positionOnPageToBeResized.setRectanglePerLine(manualResizeRedaction.getPositions()
.stream()
.map(ManualChangesApplicationService::toRectangle2D)
.collect(Collectors.toList()));
.stream()
.map(ManualChangesApplicationService::toRectangle2D)
.collect(Collectors.toList()));
entityToBeResized.getManualOverwrite().addChange(manualResizeRedaction);
@@ -90,11 +100,17 @@ public class ManualChangesApplicationService {
if (closestEntity.isPresent()) {
copyValuesFromClosestEntity(entityToBeResized, manualResizeRedaction, closestEntity.get());
possibleEntities.values().stream().flatMap(Collection::stream).forEach(TextEntity::removeFromGraph);
possibleEntities.values()
.stream()
.flatMap(Collection::stream)
.forEach(TextEntity::removeFromGraph);
return;
}
possibleEntities.values().stream().flatMap(Collection::stream).forEach(TextEntity::removeFromGraph);
possibleEntities.values()
.stream()
.flatMap(Collection::stream)
.forEach(TextEntity::removeFromGraph);
if (node.hasParent()) {
node = node.getParent();
@@ -110,14 +126,18 @@ public class ManualChangesApplicationService {
Set<SemanticNode> currentIntersectingNodes = new HashSet<>(entityToBeResized.getIntersectingNodes());
Set<SemanticNode> newIntersectingNodes = new HashSet<>(closestEntity.getIntersectingNodes());
Sets.difference(currentIntersectingNodes, newIntersectingNodes).forEach(removedNode -> removedNode.getEntities().remove(entityToBeResized));
Sets.difference(newIntersectingNodes, currentIntersectingNodes).forEach(addedNode -> addedNode.getEntities().add(entityToBeResized));
Sets.difference(currentIntersectingNodes, newIntersectingNodes)
.forEach(removedNode -> removedNode.getEntities().remove(entityToBeResized));
Sets.difference(newIntersectingNodes, currentIntersectingNodes)
.forEach(addedNode -> addedNode.getEntities().add(entityToBeResized));
Set<Page> currentIntersectingPages = new HashSet<>(entityToBeResized.getPages());
Set<Page> newIntersectingPages = new HashSet<>(closestEntity.getPages());
Sets.difference(currentIntersectingPages, newIntersectingPages).forEach(removedPage -> removedPage.getEntities().remove(entityToBeResized));
Sets.difference(newIntersectingPages, currentIntersectingPages).forEach(addedPage -> addedPage.getEntities().add(entityToBeResized));
Sets.difference(currentIntersectingPages, newIntersectingPages)
.forEach(removedPage -> removedPage.getEntities().remove(entityToBeResized));
Sets.difference(newIntersectingPages, currentIntersectingPages)
.forEach(addedPage -> addedPage.getEntities().add(entityToBeResized));
entityToBeResized.setDeepestFullyContainingNode(closestEntity.getDeepestFullyContainingNode());
entityToBeResized.setIntersectingNodes(new ArrayList<>(newIntersectingNodes));
@@ -130,12 +150,21 @@ public class ManualChangesApplicationService {
}
/**
* Resizes an image entity based on manual resize redaction instructions.
*
* @param image The image to resize.
* @param manualResizeRedaction The details of the resize operation.
*/
public void resizeImage(Image image, ManualResizeRedaction manualResizeRedaction) {
if (manualResizeRedaction.getPositions().isEmpty() || manualResizeRedaction.getPositions() == null) {
return;
}
var bBox = RectangleTransformations.rectangle2DBBox(manualResizeRedaction.getPositions().stream().map(ManualChangesApplicationService::toRectangle2D).toList());
var bBox = RectangleTransformations.rectangle2DBBox(manualResizeRedaction.getPositions()
.stream()
.map(ManualChangesApplicationService::toRectangle2D)
.toList());
image.setPosition(bBox);
image.getManualOverwrite().addChange(manualResizeRedaction);
}
@@ -53,10 +53,13 @@ public class NotFoundImportedEntitiesService {
if (!notFoundEntities.isEmpty()) {
// imported redactions present, intersections must be added with merged imported redactions
Map<Integer, List<PrecursorEntity>> importedRedactionsMap = mapImportedRedactionsOnPage(notFoundEntities);
entityLog.getEntityLogEntry().stream().filter(entry -> !entry.getEngines().contains(Engine.IMPORTED)).forEach(redactionLogEntry -> {
redactionLogEntry.setImportedRedactionIntersections(new HashSet<>());
addIntersections(redactionLogEntry, importedRedactionsMap, analysisNumber);
});
entityLog.getEntityLogEntry()
.stream()
.filter(entry -> !entry.getEngines().contains(Engine.IMPORTED))
.forEach(redactionLogEntry -> {
redactionLogEntry.setImportedRedactionIntersections(new HashSet<>());
addIntersections(redactionLogEntry, importedRedactionsMap, analysisNumber);
});
}
}
@@ -70,7 +73,10 @@ public class NotFoundImportedEntitiesService {
.map(RectangleWithPage::pageNumber)
.collect(Collectors.toSet());
pageNumbers.forEach(pageNumber -> importedRedactionsMap.put(pageNumber,
importedEntities.stream().filter(i -> pageNumber == i.getEntityPosition().get(0).pageNumber()).collect(Collectors.toList())));
importedEntities.stream()
.filter(i -> pageNumber == i.getEntityPosition()
.get(0).pageNumber())
.collect(Collectors.toList())));
return importedRedactionsMap;
}
@@ -15,8 +15,12 @@ public class ComponentComparator implements Comparator<Component> {
@Override
public int compare(Component component1, Component component2) {
var firstEntity1 = component1.getReferences().stream().min(EntityComparators.first());
var firstEntity2 = component2.getReferences().stream().min(EntityComparators.first());
var firstEntity1 = component1.getReferences()
.stream()
.min(EntityComparators.first());
var firstEntity2 = component2.getReferences()
.stream()
.min(EntityComparators.first());
if (firstEntity1.isEmpty() && firstEntity2.isEmpty()) {
return 0;
}
@@ -40,7 +40,8 @@ public class ComponentCreationService {
private static List<Entity> findEntitiesFromLongestSection(Collection<Entity> entities) {
var entitiesBySection = entities.stream().collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
var entitiesBySection = entities.stream()
.collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
Optional<SemanticNode> longestSection = entitiesBySection.entrySet()
.stream()
.sorted(Comparator.comparingInt(ComponentCreationService::getTotalLengthOfEntities).reversed())
@@ -79,14 +80,20 @@ public class ComponentCreationService {
public void firstOrElse(String ruleIdentifier, String name, Collection<Entity> entities, String fallback) {
String valueDescription = String.format("First found value of type %s or else '%s'", joinTypes(entities), fallback);
String value = entities.stream().min(EntityComparators.first()).map(Entity::getValue).orElse(fallback);
String value = entities.stream()
.min(EntityComparators.first())
.map(Entity::getValue)
.orElse(fallback);
create(ruleIdentifier, name, value, valueDescription, entities);
}
private static String joinTypes(Collection<Entity> entities) {
return entities.stream().map(Entity::getType).distinct().collect(Collectors.joining(", "));
return entities.stream()
.map(Entity::getType)
.distinct()
.collect(Collectors.joining(", "));
}
@@ -104,12 +111,12 @@ public class ComponentCreationService {
referencedEntities.addAll(references);
kieSession.insert(Component.builder()
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription(valueDescription)
.references(new LinkedList<>(references))
.build());
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription(valueDescription)
.references(new LinkedList<>(references))
.build());
}
@@ -142,8 +149,11 @@ public class ComponentCreationService {
private static List<Entity> findEntitiesFromFirstSection(Collection<Entity> entities) {
var entitiesBySection = entities.stream().collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
Optional<SemanticNode> firstSection = entitiesBySection.keySet().stream().min(SemanticNodeComparators.first());
var entitiesBySection = entities.stream()
.collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
Optional<SemanticNode> firstSection = entitiesBySection.keySet()
.stream()
.min(SemanticNodeComparators.first());
if (firstSection.isEmpty()) {
return Collections.emptyList();
}
@@ -188,7 +198,10 @@ public class ComponentCreationService {
public void joining(String ruleIdentifier, String name, Collection<Entity> entities, String delimiter) {
String valueDescription = String.format("Joining all values of type %s with '%s'", joinTypes(entities), delimiter);
String value = entities.stream().sorted(EntityComparators.first()).map(Entity::getValue).collect(Collectors.joining(delimiter));
String value = entities.stream()
.sorted(EntityComparators.first())
.map(Entity::getValue)
.collect(Collectors.joining(delimiter));
create(ruleIdentifier, name, value, valueDescription, entities);
}
@@ -231,14 +244,20 @@ public class ComponentCreationService {
public void joiningUnique(String ruleIdentifier, String name, Collection<Entity> entities, String delimiter) {
String valueDescription = String.format("Joining all unique values of type %s with '%s'", joinTypes(entities), delimiter);
String value = entities.stream().sorted(EntityComparators.first()).map(Entity::getValue).distinct().collect(Collectors.joining(delimiter));
String value = entities.stream()
.sorted(EntityComparators.first())
.map(Entity::getValue)
.distinct()
.collect(Collectors.joining(delimiter));
create(ruleIdentifier, name, value, valueDescription, entities);
}
private static int getTotalLengthOfEntities(Map.Entry<SemanticNode, List<Entity>> entry) {
return entry.getValue().stream().mapToInt(Entity::getLength).sum();
return entry.getValue()
.stream()
.mapToInt(Entity::getLength).sum();
}
@@ -293,7 +312,10 @@ public class ComponentCreationService {
*/
public void uniqueValueCount(String ruleIdentifier, String name, Collection<Entity> entities) {
long count = entities.stream().map(Entity::getValue).distinct().count();
long count = entities.stream()
.map(Entity::getValue)
.distinct()
.count();
create(ruleIdentifier, name, String.valueOf(count), "Number of unique values in the entity references", entities);
}
@@ -307,18 +329,20 @@ public class ComponentCreationService {
*/
public void rowValueCount(String ruleIdentifier, String name, Collection<Entity> entities) {
entities.stream().collect(Collectors.groupingBy(this::getFirstTable)).forEach((optionalTable, groupedEntities) -> {
entities.stream()
.collect(Collectors.groupingBy(this::getFirstTable))
.forEach((optionalTable, groupedEntities) -> {
if (optionalTable.isEmpty()) {
return;
}
if (optionalTable.isEmpty()) {
return;
}
long count = groupedEntities.stream()
.collect(Collectors.groupingBy(entity -> getFirstTableCell(entity).map(TableCell::getRow).orElse(-1)))
.size();
long count = groupedEntities.stream()
.collect(Collectors.groupingBy(entity -> getFirstTableCell(entity).map(TableCell::getRow)
.orElse(-1))).size();
create(ruleIdentifier, name, String.valueOf(count), "Count rows with values in the entity references in same table", entities);
});
create(ruleIdentifier, name, String.valueOf(count), "Count rows with values in the entity references in same table", entities);
});
}
@@ -334,18 +358,20 @@ public class ComponentCreationService {
if (entities.isEmpty()) {
return;
}
entities.stream().sorted(EntityComparators.first()).forEach(entity -> {
BreakIterator iterator = BreakIterator.getSentenceInstance(Locale.ENGLISH);
iterator.setText(entity.getValue());
int start = iterator.first();
for (int end = iterator.next(); end != BreakIterator.DONE; start = end, end = iterator.next()) {
create(ruleIdentifier,
name,
entity.getValue().substring(start, end).replaceAll("\\n", "").trim(),
String.format("Values of type '%s' as sentences", entity.getType()),
entity);
}
});
entities.stream()
.sorted(EntityComparators.first())
.forEach(entity -> {
BreakIterator iterator = BreakIterator.getSentenceInstance(Locale.ENGLISH);
iterator.setText(entity.getValue());
int start = iterator.first();
for (int end = iterator.next(); end != BreakIterator.DONE; start = end, end = iterator.next()) {
create(ruleIdentifier,
name,
entity.getValue().substring(start, end).replaceAll("\\n", "").trim(),
String.format("Values of type '%s' as sentences", entity.getType()),
entity);
}
});
}
@@ -366,12 +392,12 @@ public class ComponentCreationService {
List<Entity> referenceList = new LinkedList<>();
referenceList.add(reference);
kieSession.insert(Component.builder()
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription(valueDescription)
.references(referenceList)
.build());
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription(valueDescription)
.references(referenceList)
.build());
}
@@ -428,8 +454,10 @@ public class ComponentCreationService {
}
String formattedDateStrings = Stream.concat(//
dates.stream().sorted().map(date -> DateConverter.convertDate(date, resultFormat)), //
unparsedDates.stream())//
dates.stream()
.sorted()
.map(date -> DateConverter.convertDate(date, resultFormat)), //
unparsedDates.stream())//
.collect(Collectors.joining(", "));
create(ruleIdentifier, name, formattedDateStrings, valueDescription, entities);
@@ -445,26 +473,34 @@ public class ComponentCreationService {
*/
public void joiningFromSameTableRow(String ruleIdentifier, String name, Collection<Entity> entities) {
String types = entities.stream().map(Entity::getType).sorted(Comparator.reverseOrder()).distinct().collect(Collectors.joining(", "));
String types = entities.stream()
.map(Entity::getType)
.sorted(Comparator.reverseOrder())
.distinct()
.collect(Collectors.joining(", "));
String valueDescription = String.format("Combine values of %s that are in same table row", types);
entities.stream().collect(Collectors.groupingBy(this::getFirstTable)).forEach((optionalTable, groupedEntities) -> {
if (optionalTable.isEmpty()) {
groupedEntities.forEach(entity -> create(ruleIdentifier, name, entity.getValue(), valueDescription, entity));
}
entities.stream()
.collect(Collectors.groupingBy(this::getFirstTable))
.forEach((optionalTable, groupedEntities) -> {
if (optionalTable.isEmpty()) {
groupedEntities.forEach(entity -> create(ruleIdentifier, name, entity.getValue(), valueDescription, entity));
}
groupedEntities.stream()
.filter(entity -> entity.getContainingNode() instanceof TableCell)
.collect(Collectors.groupingBy(entity -> ((TableCell) entity.getContainingNode()).getRow()))
.entrySet()
.stream()
.sorted(Comparator.comparingInt(Map.Entry::getKey))
.map(Map.Entry::getValue)
.forEach(entitiesInSameRow -> create(ruleIdentifier,
name,
entitiesInSameRow.stream().sorted(EntityComparators.first()).map(Entity::getValue).collect(Collectors.joining(", ")),
valueDescription,
entitiesInSameRow));
});
groupedEntities.stream()
.filter(entity -> entity.getContainingNode() instanceof TableCell)
.collect(Collectors.groupingBy(entity -> ((TableCell) entity.getContainingNode()).getRow())).entrySet()
.stream()
.sorted(Comparator.comparingInt(Map.Entry::getKey))
.map(Map.Entry::getValue)
.forEach(entitiesInSameRow -> create(ruleIdentifier,
name,
entitiesInSameRow.stream()
.sorted(EntityComparators.first())
.map(Entity::getValue)
.collect(Collectors.joining(", ")),
valueDescription,
entitiesInSameRow));
});
}
@@ -521,12 +557,12 @@ public class ComponentCreationService {
public void create(String ruleIdentifier, String name, String value) {
kieSession.insert(Component.builder()
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription("")
.references(Collections.emptyList())
.build());
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
.name(name)
.value(value)
.valueDescription("")
.references(Collections.emptyList())
.build());
}
}
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server.service.document;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.HashSet;
import java.util.LinkedList;
@@ -8,22 +9,23 @@ import java.util.Map;
import java.util.NoSuchElementException;
import java.util.Set;
import com.iqser.red.service.redaction.v1.server.model.document.DocumentData;
import com.iqser.red.service.redaction.v1.server.model.document.DocumentTree;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.DuplicatedParagraph;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Document;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Footer;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Header;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Headline;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Page;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Paragraph;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Section;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNode;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Table;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.TableCell;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.AtomicTextBlock;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlockCollector;
import com.iqser.red.service.redaction.v1.server.model.document.DocumentData;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Document;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Headline;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Table;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPage;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentPositionData;
import com.knecon.fforesight.service.layoutparser.internal.api.data.redaction.DocumentStructure;
@@ -40,7 +42,9 @@ public class DocumentGraphMapper {
DocumentTree documentTree = new DocumentTree(document);
Context context = new Context(documentData, documentTree);
context.pageData.addAll(Arrays.stream(documentData.getDocumentPages()).map(DocumentGraphMapper::buildPage).toList());
context.pageData.addAll(Arrays.stream(documentData.getDocumentPages())
.map(DocumentGraphMapper::buildPage)
.toList());
context.documentTree.getRoot().getChildren().addAll(buildEntries(documentData.getDocumentStructure().getRoot().getChildren(), context));
@@ -55,14 +59,16 @@ public class DocumentGraphMapper {
private List<DocumentTree.Entry> buildEntries(List<DocumentStructure.EntryData> entries, Context context) {
List<DocumentTree.Entry> newEntries = new LinkedList<>();
List<DocumentTree.Entry> newEntries = new ArrayList<>(entries.size());
for (DocumentStructure.EntryData entryData : entries) {
List<Page> pages = Arrays.stream(entryData.getPageNumbers()).map(pageNumber -> getPage(pageNumber, context)).toList();
List<Page> pages = Arrays.stream(entryData.getPageNumbers())
.map(pageNumber -> getPage(pageNumber, context))
.toList();
SemanticNode node = switch (entryData.getType()) {
case SECTION -> buildSection(context);
case PARAGRAPH -> buildParagraph(context);
case PARAGRAPH -> buildParagraph(context, entryData.getProperties());
case HEADLINE -> buildHeadline(context);
case HEADER -> buildHeader(context);
case FOOTER -> buildFooter(context);
@@ -76,8 +82,10 @@ public class DocumentGraphMapper {
TextBlock textBlock = toTextBlock(entryData.getAtomicBlockIds(), context, node);
node.setLeafTextBlock(textBlock);
}
List<Integer> treeId = Arrays.stream(entryData.getTreeId()).boxed().toList();
entryData.getEngines().forEach(engine -> node.addEngine(engine));
List<Integer> treeId = Arrays.stream(entryData.getTreeId()).boxed()
.toList();
entryData.getEngines()
.forEach(engine -> node.addEngine(engine));
node.setTreeId(treeId);
switch (entryData.getType()) {
@@ -142,24 +150,35 @@ public class DocumentGraphMapper {
}
private Paragraph buildParagraph(Context context) {
private Paragraph buildParagraph(Context context, Map<String, String> properties) {
if (PropertiesMapper.isDuplicateParagraph(properties)) {
DuplicatedParagraph duplicatedParagraph = DuplicatedParagraph.builder().documentTree(context.documentTree).build();
Long[] unsortedTextblockIds = PropertiesMapper.getUnsortedTextblockIds(properties);
duplicatedParagraph.setUnsortedLeafTextBlock(toTextBlock(unsortedTextblockIds, context, duplicatedParagraph));
return duplicatedParagraph;
}
return Paragraph.builder().documentTree(context.documentTree).build();
}
private TextBlock toTextBlock(Long[] atomicTextBlockIds, Context context, SemanticNode parent) {
private TextBlock toTextBlock(Long[] atomicTextBlockIds, Context context, SemanticNode parent) {
return Arrays.stream(atomicTextBlockIds).map(atomicTextBlockId -> getAtomicTextBlock(context, parent, atomicTextBlockId)).collect(new TextBlockCollector());
return Arrays.stream(atomicTextBlockIds)
.map(atomicTextBlockId -> getAtomicTextBlock(context, parent, atomicTextBlockId))
.collect(new TextBlockCollector());
}
private AtomicTextBlock getAtomicTextBlock(Context context, SemanticNode parent, Long atomicTextBlockId) {
return AtomicTextBlock.fromAtomicTextBlockData(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)),
context.documentPositionData.get(Math.toIntExact(atomicTextBlockId)),
parent,
getPage(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)).getPage(), context));
context.documentPositionData.get(Math.toIntExact(atomicTextBlockId)),
parent,
getPage(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)).getPage(), context));
}
@@ -173,8 +192,7 @@ public class DocumentGraphMapper {
return context.pageData.stream()
.filter(page -> page.getNumber() == Math.toIntExact(pageIndex))
.findFirst()
.orElseThrow(() -> new NoSuchElementException(String.format("ClassificationPage with number %d not found", pageIndex)));
.findFirst().orElseThrow(() -> new NoSuchElementException(String.format("ClassificationPage with number %d not found", pageIndex)));
}
@@ -190,8 +208,10 @@ public class DocumentGraphMapper {
this.documentTree = documentTree;
this.pageData = new LinkedList<>();
this.documentTextData = Arrays.stream(documentData.getDocumentTextData()).toList();
this.documentPositionData = Arrays.stream(documentData.getDocumentPositionData()).toList();
this.documentTextData = Arrays.stream(documentData.getDocumentTextData())
.toList();
this.documentPositionData = Arrays.stream(documentData.getDocumentPositionData())
.toList();
}
@@ -11,6 +11,7 @@ public abstract class EntityComparators implements Comparator<Entity> {
return new FirstEntity();
}
public static class LongestEntity implements Comparator<Entity> {
@Override
@@ -27,6 +28,7 @@ public abstract class EntityComparators implements Comparator<Entity> {
return new LongestEntity();
}
public static class FirstEntity implements Comparator<Entity> {
@Override
@@ -1,14 +1,18 @@
package com.iqser.red.service.redaction.v1.server.service.document;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.*;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.addEntityToNodeEntitySets;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.addToPages;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.allEntitiesIntersectAndHaveSameTypes;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.checkIfBothStartAndEndAreEmpty;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.findIntersectingSubNodes;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.toLineAfterTextRange;
import static com.iqser.red.service.redaction.v1.server.service.document.EntityCreationUtility.truncateEndIfLineBreakIsBetween;
import static com.iqser.red.service.redaction.v1.server.utils.SeparatorUtils.boundaryIsSurroundedBySeparators;
import java.util.Collection;
import java.util.Collections;
import java.util.Comparator;
import java.util.LinkedList;
import java.util.List;
import java.util.NoSuchElementException;
import java.util.Optional;
import java.util.Set;
import java.util.stream.Collectors;
@@ -54,6 +58,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, case-sensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStrings(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -65,6 +80,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, case-insensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -76,6 +102,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the start string in the entity, case-sensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeStart(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -92,6 +129,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the start string in the entity, case-insensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeStartIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -108,6 +156,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the end string in the entity, case-sensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeEnd(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -124,6 +183,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the end string in the entity, case-insensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -140,6 +210,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the start and end string in the entity, case-sensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeStartAndEnd(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -160,6 +241,17 @@ public class EntityCreationService {
}
/**
* Creates entities found between specified start and stop strings, including the start and end string in the entity, case-insensitive.
*
* @param start The starting string to search for.
* @param stop The stopping string to search for.
* @param type The type of entity to create.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> betweenStringsIncludeStartAndEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(start, stop);
@@ -180,6 +272,17 @@ public class EntityCreationService {
}
/**
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node, case-sensitive.
*
* @param starts A list of start strings to search for.
* @param stops A list of stop strings to search for.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> shortestBetweenAnyString(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(starts, stops);
@@ -191,6 +294,17 @@ public class EntityCreationService {
}
/**
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node, case-insensitive.
*
* @param starts A list of start strings to search for.
* @param stops A list of stop strings to search for.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which to search.
* @return A stream of {@link TextEntity} identified objects.
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
*/
public Stream<TextEntity> shortestBetweenAnyStringIgnoreCase(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node) {
checkIfBothStartAndEndAreEmpty(starts, stops);
@@ -202,6 +316,18 @@ public class EntityCreationService {
}
/**
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node,
* case-insensitive, with a length limit.
*
* @param starts A list of start strings to search for, case-insensitively.
* @param stops A list of stop strings to search for, case-insensitively.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @param limit The maximum length of the entity text.
* @return A stream of {@link TextEntity} objects found between any of the start and stop strings, case-insensitively, and within the specified limit.
*/
public Stream<TextEntity> shortestBetweenAnyStringIgnoreCase(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node, int limit) {
checkIfBothStartAndEndAreEmpty(starts, stops);
@@ -213,6 +339,16 @@ public class EntityCreationService {
}
/**
* Creates entities based on the boundaries identified between start and stop regular expressions within a specified semantic node.
*
* @param regexStart The regular expression defining the start boundary.
* @param regexStop The regular expression defining the stop boundary.
* @param type The type of entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects identified between the start and stop regular expressions.
*/
public Stream<TextEntity> betweenRegexes(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -223,6 +359,17 @@ public class EntityCreationService {
}
/**
* Creates entities based on the boundaries identified between start and stop regular expressions within a specified semantic node,
* case-insensitive.
*
* @param regexStart The regular expression defining the start boundary, case-insensitive.
* @param regexStop The regular expression defining the stop boundary, case-insensitive.
* @param type The type of entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects identified between the start and stop regular expressions, case-insensitively.
*/
public Stream<TextEntity> betweenRegexesIgnoreCase(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -233,12 +380,35 @@ public class EntityCreationService {
}
/**
* Creates entities based on the boundaries identified between specified start and stop text ranges within a semantic node.
* This is a more general method that can be used directly with lists of start and stop {@link TextRange} objects.
*
* @param startBoundaries A list of start text range boundaries.
* @param stopBoundaries A list of stop text range boundaries.
* @param type The type of entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects identified between the start and stop text ranges.
*/
public Stream<TextEntity> betweenTextRanges(List<TextRange> startBoundaries, List<TextRange> stopBoundaries, String type, EntityType entityType, SemanticNode node) {
return betweenTextRanges(startBoundaries, stopBoundaries, type, entityType, node, 0);
}
/**
* Creates entities based on the boundaries identified between specified start and stop text ranges within a semantic node,
* with an optional length limit for the entities.
*
* @param startBoundaries A list of start text range boundaries.
* @param stopBoundaries A list of stop text range boundaries.
* @param type The type of entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @param limit The maximum length of the entity text; use 0 for no limit.
* @return A stream of {@link TextEntity} objects identified between the start and stop text ranges, within the specified limit.
*/
public Stream<TextEntity> betweenTextRanges(List<TextRange> startBoundaries, List<TextRange> stopBoundaries, String type, EntityType entityType, SemanticNode node, int limit) {
List<TextRange> entityBoundaries = findNonOverlappingBoundariesBetweenBoundariesWithMinimalDistances(startBoundaries, stopBoundaries);
@@ -276,12 +446,22 @@ public class EntityCreationService {
"this is some text. a here is more text" and "here is more text". We only want to keep the latter.
*/
return entityTextRanges.stream()
.filter(boundary -> entityTextRanges.stream().noneMatch(innerBoundary -> !innerBoundary.equals(boundary) && innerBoundary.containedBy(boundary)))
.filter(boundary -> entityTextRanges.stream()
.noneMatch(innerBoundary -> !innerBoundary.equals(boundary) && innerBoundary.containedBy(boundary)))
.toList();
}
/**
* Creates text entities based on boundaries identified by a search implementation within a specified semantic node.
*
* @param searchImplementation The search implementation to use for identifying boundaries.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects corresponding to the identified boundaries.
*/
public Stream<TextEntity> bySearchImplementation(SearchImplementation searchImplementation, String type, EntityType entityType, SemanticNode node) {
return searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
@@ -293,6 +473,15 @@ public class EntityCreationService {
}
/**
* Identifies text entities located immediately after the specified strings within a semantic node.
*
* @param strings A list of strings to search for. The text immediately following each string is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects found immediately after the specified strings.
*/
public Stream<TextEntity> lineAfterStrings(List<String> strings, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -307,6 +496,15 @@ public class EntityCreationService {
}
/**
* Identifies text entities located immediately after the specified strings within a semantic node, case-insensitive.
*
* @param strings A list of strings to search for, case-insensitive. The text immediately following each string is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects found immediately after the specified strings, case-insensitively.
*/
public Stream<TextEntity> lineAfterStringsIgnoreCase(List<String> strings, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -321,6 +519,15 @@ public class EntityCreationService {
}
/**
* Identifies a text entity located immediately after a specified string within a semantic node.
*
* @param string The string to search for. The text immediately following this string is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects found immediately after the specified string.
*/
public Stream<TextEntity> lineAfterString(String string, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -334,6 +541,15 @@ public class EntityCreationService {
}
/**
* Identifies a text entity located immediately after a specified string within a semantic node, case-insensitive.
*
* @param string The string to search for, case-insensitive. The text immediately following this string is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param node The semantic node within which the search is performed.
* @return A stream of {@link TextEntity} objects found immediately after the specified string, case-insensitively.
*/
public Stream<TextEntity> lineAfterStringIgnoreCase(String string, String type, EntityType entityType, SemanticNode node) {
TextBlock textBlock = node.getTextBlock();
@@ -347,25 +563,43 @@ public class EntityCreationService {
}
/**
* Identifies text entities located immediately after a specified string across table cell columns within a table node.
*
* @param string The string to search for. The text immediately following this string in subsequent table cells is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param tableNode The table node within which the search is performed.
* @return A stream of {@link TextEntity} objects found across table cell columns immediately after the specified string.
*/
public Stream<TextEntity> lineAfterStringAcrossColumns(String string, String type, EntityType entityType, Table tableNode) {
return tableNode.streamTableCells()
.flatMap(tableCell -> lineAfterBoundariesAcrossColumns(RedactionSearchUtility.findTextRangesByString(string, tableCell.getTextBlock()),
tableCell,
type,
entityType,
tableNode));
tableCell,
type,
entityType,
tableNode));
}
/**
* Identifies text entities located immediately after a specified string across table cell columns within a table node, case-insensitive.
*
* @param string The string to search for, case-insensitive. The text immediately following this string in subsequent table cells is considered for entity creation.
* @param type The type of the entity to be created.
* @param entityType The detailed classification of the entity.
* @param tableNode The table node within which the search is performed.
* @return A stream of {@link TextEntity} objects found across table cell columns immediately after the specified string, case-insensitively.
*/
public Stream<TextEntity> lineAfterStringAcrossColumnsIgnoreCase(String string, String type, EntityType entityType, Table tableNode) {
return tableNode.streamTableCells()
.flatMap(tableCell -> lineAfterBoundariesAcrossColumns(RedactionSearchUtility.findTextRangesByStringIgnoreCase(string, tableCell.getTextBlock()),
tableCell,
type,
entityType,
tableNode));
tableCell,
type,
entityType,
tableNode));
}
@@ -396,6 +630,15 @@ public class EntityCreationService {
}
/**
* Attempts to create a text entity for text within a semantic node, immediately after a specified string.
*
* @param semanticNode The semantic node within which to search for the string.
* @param string The string after which the entity should be created.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if the string is not found.
*/
public Optional<TextEntity> semanticNodeAfterString(SemanticNode semanticNode, String string, String type, EntityType entityType) {
var textBlock = semanticNode.getTextBlock();
@@ -414,30 +657,77 @@ public class EntityCreationService {
}
/**
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block,
* considering line breaks in the text.
*
* @param regexPattern The regex pattern to match.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node containing the text block to search.
* @return A stream of identified {@link TextEntity} objects.
*/
public Stream<TextEntity> byRegexWithLineBreaks(String regexPattern, String type, EntityType entityType, SemanticNode node) {
return byRegexWithLineBreaks(regexPattern, type, entityType, 0, node);
}
/**
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block, considering line breaks in the text, case-insensitive.
*
* @param regexPattern The regex pattern to match.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node containing the text block to search.
* @return A stream of identified {@link TextEntity} objects.
*/
public Stream<TextEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, SemanticNode node) {
return byRegexWithLineBreaksIgnoreCase(regexPattern, type, entityType, 0, node);
}
/**
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block.
*
* @param regexPattern The regex pattern to match.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node containing the text block to search.
* @return A stream of identified {@link TextEntity} objects.
*/
public Stream<TextEntity> byRegex(String regexPattern, String type, EntityType entityType, SemanticNode node) {
return byRegex(regexPattern, type, entityType, 0, node);
}
/**
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block, case-insensitive.
*
* @param regexPattern The regex pattern to match.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node containing the text block to search.
* @return A stream of identified {@link TextEntity} objects.
*/
public Stream<TextEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, SemanticNode node) {
return byRegexIgnoreCase(regexPattern, type, entityType, 0, node);
}
/**
* Identifies text entities within a semantic node's text block based on a regex pattern that includes line breaks.
*
* @param regexPattern Regex pattern to match, including handling for line breaks.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param group The regex group to target for entity creation.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the regex pattern.
*/
public Stream<TextEntity> byRegexWithLineBreaks(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByRegexWithLineBreaks(regexPattern, group, node.getTextBlock())
@@ -448,6 +738,16 @@ public class EntityCreationService {
}
/**
* Identifies text entities within a semantic node's text block based on a regex pattern that includes line breaks, case-insensitive.
*
* @param regexPattern Regex pattern to match, including handling for line breaks.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param group The regex group to target for entity creation.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the regex pattern.
*/
public Stream<TextEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByRegexWithLineBreaksIgnoreCase(regexPattern, group, node.getTextBlock())
@@ -458,6 +758,16 @@ public class EntityCreationService {
}
/**
* Identifies text entities based on a simple regex pattern.
*
* @param regexPattern Regex pattern to match, including handling for line breaks.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param group The regex group to target for entity creation.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the regex pattern.
*/
public Stream<TextEntity> byRegex(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByRegex(regexPattern, group, node.getTextBlock())
@@ -468,6 +778,16 @@ public class EntityCreationService {
}
/**
* Identifies text entities based on a simple regex pattern, case-insensitive.
*
* @param regexPattern Regex pattern to match, including handling for line breaks.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param group The regex group to target for entity creation.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the regex pattern.
*/
public Stream<TextEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByRegexIgnoreCase(regexPattern, group, node.getTextBlock())
@@ -478,6 +798,15 @@ public class EntityCreationService {
}
/**
* Identifies text entities based on an exact string match within a semantic node's text block.
*
* @param keyword String keyword to search for.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the exact string.
*/
public Stream<TextEntity> byString(String keyword, String type, EntityType entityType, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByString(keyword, node.getTextBlock())
@@ -488,6 +817,15 @@ public class EntityCreationService {
}
/**
* Identifies text entities based on an exact string match within a semantic node's text block, case-insensitive.
*
* @param keyword String keyword to search for.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects that match the exact string, case-insensitive.
*/
public Stream<TextEntity> byStringIgnoreCase(String keyword, String type, EntityType entityType, SemanticNode node) {
return RedactionSearchUtility.findTextRangesByStringIgnoreCase(keyword, node.getTextBlock())
@@ -498,12 +836,31 @@ public class EntityCreationService {
}
/**
* Extracts text entities from paragraphs only, within a given semantic node.
*
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node to search within.
* @return A stream of {@link TextEntity} objects extracted from paragraphs only.
*/
public Stream<TextEntity> bySemanticNodeParagraphsOnly(SemanticNode node, String type, EntityType entityType) {
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH).map(semanticNode -> bySemanticNode(semanticNode, type, entityType)).filter(Optional::isPresent).map(Optional::get);
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH)
.map(semanticNode -> bySemanticNode(semanticNode, type, entityType))
.filter(Optional::isPresent)
.map(Optional::get);
}
/**
* Merges consecutive paragraphs into a single text entity within a given semantic node.
*
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node to search within.
* @return A stream of merged {@link TextEntity} objects from consecutive paragraphs.
*/
public Stream<TextEntity> bySemanticNodeParagraphsOnlyMergeConsecutive(SemanticNode node, String type, EntityType entityType) {
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH)
@@ -516,6 +873,15 @@ public class EntityCreationService {
}
/**
* Creates a text entity immediately following a specified string within a semantic node.
*
* @param string The string after which to create the entity.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @param node The semantic node to search within.
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if not found.
*/
public Optional<TextEntity> semanticNodeAfterString(String string, String type, EntityType entityType, SemanticNode node) {
if (!node.containsString(string)) {
@@ -526,6 +892,14 @@ public class EntityCreationService {
}
/**
* Creates a text entity based on the entire text range of a semantic node.
*
* @param node The semantic node to base the text entity on.
* @param type The type of entity to create.
* @param entityType The entity's classification.
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if not valid.
*/
public Optional<TextEntity> bySemanticNode(SemanticNode node, String type, EntityType entityType) {
TextRange textRange = node.getTextBlock().getTextRange();
@@ -540,6 +914,13 @@ public class EntityCreationService {
}
/**
* Expands a text entity's start boundary based on a regex pattern match.
*
* @param entity The original text entity to expand.
* @param regexPattern The regex pattern used to find the new start boundary.
* @return An {@link Optional} containing the expanded {@link TextEntity}, or {@link Optional#empty()} if not valid.
*/
public Optional<TextEntity> byPrefixExpansionRegex(TextEntity entity, String regexPattern) {
int expandedStart = RedactionSearchUtility.getExpandedStartByRegex(entity, regexPattern);
@@ -547,6 +928,13 @@ public class EntityCreationService {
}
/**
* Expands a text entity's end boundary based on a regex pattern match.
*
* @param entity The original text entity to expand.
* @param regexPattern The regex pattern used to find the new end boundary.
* @return An {@link Optional} containing the expanded {@link TextEntity}, or {@link Optional#empty()} if not valid.
*/
public Optional<TextEntity> bySuffixExpansionRegex(TextEntity entity, String regexPattern) {
int expandedEnd = RedactionSearchUtility.getExpandedEndByRegex(entity, regexPattern);
@@ -590,11 +978,18 @@ public class EntityCreationService {
throw new IllegalArgumentException(String.format("%s is not in the %s of the provided semantic node %s", textRange, node.getTextRange(), node));
}
TextRange trimmedTextRange = textRange.trim(node.getTextBlock());
if (trimmedTextRange.length() == 0) {
return Optional.empty();
}
TextEntity entity = TextEntity.initialEntityNode(trimmedTextRange, type, entityType, node);
if (node.getEntities().contains(entity)) {
Optional<TextEntity> optionalTextEntity = node.getEntities().stream().filter(e -> e.equals(entity) && e.type().equals(type)).peek(e -> e.addEngines(engines)).findAny();
Optional<TextEntity> optionalTextEntity = node.getEntities()
.stream()
.filter(e -> e.equals(entity) && e.type().equals(type))
.peek(e -> e.addEngines(engines))
.findAny();
if (optionalTextEntity.isEmpty()) {
return optionalTextEntity; // Entity has been recategorized and should not be created at all.
return Optional.empty(); // Entity has been recategorized and should not be created at all.
}
TextEntity existingEntity = optionalTextEntity.get();
if (existingEntity.getTextRange().equals(textRange)) {
@@ -606,7 +1001,7 @@ public class EntityCreationService {
}
return Optional.empty(); // Entity has been resized, if there are duplicates they should be treated there
}
addEntityToGraph(entity, node);
addEntityToGraph(entity, node.getDocumentTree());
entity.addEngines(engines);
insertToKieSession(entity);
return Optional.of(entity);
@@ -635,6 +1030,17 @@ public class EntityCreationService {
}
/**
* @param entitiesToMerge The list of entities to merge.
* @param type The type for the merged entity.
* @param entityType The entity's classification.
* @param node The semantic node related to these entities.
* @return A single merged {@link TextEntity}.
* @throws IllegalArgumentException If entities do not intersect or have different types.
* @deprecated Do not use anymore. This might not work correctly due to duplicate textranges not being taken into account here.
* Merges a list of text entities into a single entity, assuming they intersect and are of the same type.
*/
@Deprecated(forRemoval = true)
public TextEntity mergeEntitiesOfSameType(List<TextEntity> entitiesToMerge, String type, EntityType entityType, SemanticNode node) {
if (!allEntitiesIntersectAndHaveSameTypes(entitiesToMerge)) {
@@ -647,30 +1053,65 @@ public class EntityCreationService {
return entitiesToMerge.get(0);
}
TextEntity mergedEntity = TextEntity.initialEntityNode(TextRange.merge(entitiesToMerge.stream().map(TextEntity::getTextRange).toList()), type, entityType, node);
mergedEntity.addEngines(entitiesToMerge.stream().flatMap(entityNode -> entityNode.getEngines().stream()).collect(Collectors.toSet()));
entitiesToMerge.stream().map(TextEntity::getMatchedRuleList).flatMap(Collection::stream).forEach(matchedRule -> mergedEntity.getMatchedRuleList().add(matchedRule));
TextEntity mergedEntity = TextEntity.initialEntityNode(TextRange.merge(entitiesToMerge.stream()
.map(TextEntity::getTextRange)
.toList()), type, entityType, node);
mergedEntity.addEngines(entitiesToMerge.stream()
.flatMap(entityNode -> entityNode.getEngines()
.stream())
.collect(Collectors.toSet()));
entitiesToMerge.stream()
.map(TextEntity::getMatchedRuleList)
.flatMap(Collection::stream)
.forEach(matchedRule -> mergedEntity.getMatchedRuleList().add(matchedRule));
entitiesToMerge.stream()
.map(TextEntity::getManualOverwrite)
.map(ManualChangeOverwrite::getManualChangeLog)
.flatMap(Collection::stream)
.forEach(manualChange -> mergedEntity.getManualOverwrite().addChange(manualChange));
mergedEntity.setDictionaryEntry(entitiesToMerge.stream().anyMatch(TextEntity::isDictionaryEntry));
mergedEntity.setDossierDictionaryEntry(entitiesToMerge.stream().anyMatch(TextEntity::isDossierDictionaryEntry));
mergedEntity.setDictionaryEntry(entitiesToMerge.stream()
.anyMatch(TextEntity::isDictionaryEntry));
mergedEntity.setDossierDictionaryEntry(entitiesToMerge.stream()
.anyMatch(TextEntity::isDossierDictionaryEntry));
entityEnrichmentService.enrichEntity(mergedEntity, node.getTextBlock());
addEntityToGraph(mergedEntity, node);
insertToKieSession(mergedEntity);
entitiesToMerge.stream()
.filter(e -> !e.equals(mergedEntity))
.forEach(node.getEntities()::remove);
return mergedEntity;
}
/**
* Copies a list of text entities, creating a new entity for each in the list with the same properties.
*
* @param entities The list of entities to copy.
* @param type The type for the copied entities.
* @param entityType The classification for the copied entities.
* @param node The semantic node related to these entities.
* @return A stream of copied {@link TextEntity} objects.
*/
public Stream<TextEntity> copyEntities(List<TextEntity> entities, String type, EntityType entityType, SemanticNode node) {
return entities.stream().map(entity -> copyEntity(entity, type, entityType, node));
return entities.stream()
.map(entity -> copyEntity(entity, type, entityType, node));
}
/**
* Copies a single text entity, preserving all its matched rules.
*
* @param entity The entity to copy.
* @param type The type for the copied entity.
* @param entityType The classification for the copied entity.
* @param node The semantic node related to the entity.
* @return A copied {@link TextEntity} with matched rules.
*/
public TextEntity copyEntity(TextEntity entity, String type, EntityType entityType, SemanticNode node) {
var newEntity = copyEntityWithoutRules(entity, type, entityType, node);
@@ -679,6 +1120,15 @@ public class EntityCreationService {
}
/**
* Copies a single text entity without its matched rules.
*
* @param entity The entity to copy.
* @param type The type for the copied entity.
* @param entityType The classification for the copied entity.
* @param node The semantic node related to the entity.
* @return A copied {@link TextEntity} without matched rules.
*/
public TextEntity copyEntityWithoutRules(TextEntity entity, String type, EntityType entityType, SemanticNode node) {
TextEntity newEntity = byTextRangeWithEngine(entity.getTextRange(), type, entityType, node, entity.getEngines()).orElseThrow(() -> new NotFoundException(
@@ -690,14 +1140,27 @@ public class EntityCreationService {
}
public void insertToKieSession(TextEntity mergedEntity) {
/**
* Inserts a text entity into the kieSession for further processing.
*
* @param textEntity The merged text entity to insert.
*/
public void insertToKieSession(TextEntity textEntity) {
if (kieSession != null) {
kieSession.insert(mergedEntity);
kieSession.insert(textEntity);
}
}
/**
* Creates a text entity based on a Named Entity Recognition (NER) entity.
*
* @param nerEntity The NER entity used for creating the text entity.
* @param entityType The entity's classification.
* @param semanticNode The semantic node related to the NER entity.
* @return A new {@link TextEntity} based on the NER entity.
*/
public TextEntity byNerEntity(NerEntities.NerEntity nerEntity, EntityType entityType, SemanticNode semanticNode) {
return byTextRangeWithEngine(nerEntity.textRange(), nerEntity.type(), entityType, semanticNode, Set.of(Engine.NER)).orElseThrow(() -> new NotFoundException(
@@ -705,24 +1168,59 @@ public class EntityCreationService {
}
/**
* Creates a text entity based on a Named Entity Recognition (NER) entity, with a specified type.
*
* @param nerEntity The NER entity used for creating the text entity.
* @param type Type of the entity.
* @param entityType The entity's classification.
* @param semanticNode The semantic node related to the NER entity.
* @return A new {@link TextEntity} based on the NER entity.
*/
public TextEntity byNerEntity(NerEntities.NerEntity nerEntity, String type, EntityType entityType, SemanticNode semanticNode) {
return byTextRangeWithEngine(nerEntity.textRange(), type, entityType, semanticNode, Set.of(Engine.NER)).orElseThrow(() -> new NotFoundException("No entity present!"));
}
/**
* Optionally creates a text entity based on a Named Entity Recognition (NER) entity.
*
* @param nerEntity The NER entity used for creating the text entity.
* @param entityType The entity's classification.
* @param semanticNode The semantic node related to the NER entity.
* @return An {@link Optional} containing the new {@link TextEntity} based on the NER entity, or {@link Optional#empty()} if not created.
*/
public Optional<TextEntity> optionalByNerEntity(NerEntities.NerEntity nerEntity, EntityType entityType, SemanticNode semanticNode) {
return byTextRangeWithEngine(nerEntity.textRange(), nerEntity.type(), entityType, semanticNode, Set.of(Engine.NER));
}
/**
* Optionally creates a text entity based on a Named Entity Recognition (NER) entity, with a specified type.
*
* @param nerEntity The NER entity used for creating the text entity.
* @param type Type of the entity.
* @param entityType The entity's classification.
* @param semanticNode The semantic node related to the NER entity.
* @return An {@link Optional} containing the new {@link TextEntity} based on the NER entity, or {@link Optional#empty()} if not created.
*/
public Optional<TextEntity> optionalByNerEntity(NerEntities.NerEntity nerEntity, String type, EntityType entityType, SemanticNode semanticNode) {
return byTextRangeWithEngine(nerEntity.textRange(), type, entityType, semanticNode, Set.of(Engine.NER));
}
/**
* Combines multiple NER entities into a single text entity.
*
* @param nerEntities The collection of NER entities to combine.
* @param type The type for the combined entity.
* @param entityType The classification for the combined entity.
* @param semanticNode The semantic node related to these entities.
* @return A stream of combined {@link TextEntity} objects.
*/
public Stream<TextEntity> combineNerEntitiesToCbiAddressDefaults(NerEntities nerEntities, String type, EntityType entityType, SemanticNode semanticNode) {
return NerEntitiesAdapter.combineNerEntitiesToCbiAddressDefaults(nerEntities)
@@ -732,34 +1230,41 @@ public class EntityCreationService {
}
/**
* Validates if a given text range within a text block represents a valid entity.
*
* @param textBlock The text block containing the text range.
* @param textRange The text range to validate.
* @return true if the text range represents a valid entity, false otherwise.
*/
public boolean isValidEntityTextRange(TextBlock textBlock, TextRange textRange) {
return textRange.length() > 0 && boundaryIsSurroundedBySeparators(textBlock, textRange);
}
/**
* Adds a text entity to its related semantic node and updates the document tree accordingly.
*
* @param entity The text entity to add.
* @param node The semantic node related to the entity.
*/
public void addEntityToGraph(TextEntity entity, SemanticNode node) {
DocumentTree documentTree = node.getDocumentTree();
try {
if (node.getEntities().contains(entity)) {
// If entity already exists and it has a different text range, we add the text range to the list of duplicated text ranges
node.getEntities().stream()//
.filter(e -> e.equals(entity))//
.filter(e -> !e.getTextRange().equals(entity.getTextRange()))//
.findAny()//
.ifPresent(entityToDuplicate -> addDuplicateEntityToGraph(entityToDuplicate, entity.getTextRange(), node));
} else {
entity.addIntersectingNode(documentTree.getRoot().getNode());
addEntityToGraph(entity, documentTree);
}
} catch (NoSuchElementException e) {
entity.setDeepestFullyContainingNode(documentTree.getRoot().getNode());
entityEnrichmentService.enrichEntity(entity, entity.getDeepestFullyContainingNode().getTextBlock());
entity.addIntersectingNode(documentTree.getRoot().getNode());
addToPages(entity);
addEntityToNodeEntitySets(entity);
if (node.getEntities().contains(entity)) {
// If entity already exists and it has a different text range, we add the text range to the list of duplicated text ranges
node.getEntities()
.stream()//
.filter(e -> e.equals(entity))//
.filter(e -> !e.getTextRange().equals(entity.getTextRange()))//
.findAny()
.ifPresent(e -> addDuplicateEntityToGraph(e, entity.getTextRange(), node));
} else {
addEntityToGraph(entity, documentTree);
}
}
@@ -770,8 +1275,10 @@ public class EntityCreationService {
SemanticNode deepestSharedNode = entityToDuplicate.getIntersectingNodes()
.stream()
.sorted(Comparator.comparingInt(n -> -n.getTreeId().size()))
.filter(intersectingNode -> entityToDuplicate.getDuplicateTextRanges().stream().allMatch(tr -> intersectingNode.getTextRange().contains(tr)) && //
intersectingNode.getTextRange().contains(entityToDuplicate.getTextRange()))
.filter(intersectingNode -> entityToDuplicate.getDuplicateTextRanges()
.stream()
.allMatch(tr -> intersectingNode.getTextRange().contains(tr)) && //
intersectingNode.getTextRange().contains(entityToDuplicate.getTextRange()))
.findFirst()
.orElse(node.getDocumentTree().getRoot().getNode());
@@ -784,7 +1291,8 @@ public class EntityCreationService {
return;
}
additionalIntersectingNode.getEntities().add(entityToDuplicate);
additionalIntersectingNode.getPages(newTextRange).forEach(page -> page.getEntities().add(entityToDuplicate));
additionalIntersectingNode.getPages(newTextRange)
.forEach(page -> page.getEntities().add(entityToDuplicate));
entityToDuplicate.addIntersectingNode(additionalIntersectingNode);
});
}
@@ -792,12 +1300,7 @@ public class EntityCreationService {
private void addEntityToGraph(TextEntity entity, DocumentTree documentTree) {
SemanticNode containingNode = documentTree.childNodes(Collections.emptyList())
.filter(node -> node.getTextBlock().containsTextRange(entity.getTextRange()))
.findFirst()
.orElseThrow(() -> new NoSuchElementException("No containing Node found!"));
containingNode.addThisToEntityIfIntersects(entity);
documentTree.getRoot().getNode().addThisToEntityIfIntersects(entity);
TextBlock textBlock = entity.getDeepestFullyContainingNode().getTextBlock();
entityEnrichmentService.enrichEntity(entity, textBlock);
@@ -806,5 +1309,4 @@ public class EntityCreationService {
addEntityToNodeEntitySets(entity);
}
}
@@ -11,7 +11,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBl
public class EntityCreationUtility {
public static void checkIfBothStartAndEndAreEmpty(String start, String end) {
checkIfBothStartAndEndAreEmpty(List.of(start), List.of(end));
@@ -57,7 +56,8 @@ public class EntityCreationUtility {
public static void addEntityToNodeEntitySets(TextEntity entity) {
entity.getIntersectingNodes().forEach(node -> node.getEntities().add(entity));
entity.getIntersectingNodes()
.forEach(node -> node.getEntities().add(entity));
}
@@ -59,7 +59,9 @@ public class EntityEnrichmentService {
private static List<String> splitToWordsAndRemoveEmptyWords(String textAfter) {
return Arrays.stream(textAfter.split(" ")).filter(word -> !Objects.equals("", word)).toList();
return Arrays.stream(textAfter.split(" "))
.filter(word -> !Objects.equals("", word))
.toList();
}
@@ -47,7 +47,9 @@ public class EntityFindingUtility {
}
public Optional<TextEntity> findClosestEntityAndReturnEmptyIfNotFound(PrecursorEntity precursorEntity, Map<String, List<TextEntity>> entitiesWithSameValue, double matchThreshold) {
public Optional<TextEntity> findClosestEntityAndReturnEmptyIfNotFound(PrecursorEntity precursorEntity,
Map<String, List<TextEntity>> entitiesWithSameValue,
double matchThreshold) {
if (precursorEntity.getValue() == null) {
return Optional.empty();
@@ -56,7 +58,7 @@ public class EntityFindingUtility {
List<TextEntity> possibleEntities = entitiesWithSameValue.get(precursorEntity.getValue().toLowerCase(Locale.ENGLISH));
if (entityIdentifierValueNotFound(possibleEntities)) {
log.warn("Entity could not be created with precursorEntity: {}, due to the value {} not being found anywhere.", precursorEntity, precursorEntity.getValue());
log.info("Entity could not be created with precursorEntity: {}, due to the value {} not being found anywhere.", precursorEntity, precursorEntity.getValue());
return Optional.empty();
}
@@ -66,18 +68,22 @@ public class EntityFindingUtility {
.min(Comparator.comparingDouble(ClosestEntity::getDistance));
if (optionalClosestEntity.isEmpty()) {
log.warn("No Entity with value {} found on page {}", precursorEntity.getValue(), precursorEntity.getEntityPosition());
log.info("No Entity with value {} found on page {}", precursorEntity.getValue(), precursorEntity.getEntityPosition());
return Optional.empty();
}
ClosestEntity closestEntity = optionalClosestEntity.get();
if (closestEntity.getDistance() > matchThreshold) {
log.warn("For entity {} on page {} with positions {} distance to closest found entity is {} and therefore higher than the threshold of {}",
precursorEntity.getValue(),
precursorEntity.getEntityPosition().get(0).pageNumber(),
precursorEntity.getEntityPosition().stream().map(RectangleWithPage::rectangle2D).toList(),
closestEntity.getDistance(),
matchThreshold);
log.info("For entity {} on page {} with positions {} distance to closest found entity is {} and therefore higher than the threshold of {}",
precursorEntity.getValue(),
precursorEntity.getEntityPosition()
.get(0).pageNumber(),
precursorEntity.getEntityPosition()
.stream()
.map(RectangleWithPage::rectangle2D)
.toList(),
closestEntity.getDistance(),
matchThreshold);
return Optional.empty();
}
@@ -93,8 +99,14 @@ public class EntityFindingUtility {
private static boolean pagesMatch(TextEntity entity, List<RectangleWithPage> originalPositions) {
Set<Integer> entityPageNumbers = entity.getPositionsOnPagePerPage().stream().map(PositionOnPage::getPage).map(Page::getNumber).collect(Collectors.toSet());
Set<Integer> originalPageNumbers = originalPositions.stream().map(RectangleWithPage::pageNumber).collect(Collectors.toSet());
Set<Integer> entityPageNumbers = entity.getPositionsOnPagePerPage()
.stream()
.map(PositionOnPage::getPage)
.map(Page::getNumber)
.collect(Collectors.toSet());
Set<Integer> originalPageNumbers = originalPositions.stream()
.map(RectangleWithPage::pageNumber)
.collect(Collectors.toSet());
return entityPageNumbers.containsAll(originalPageNumbers);
}
@@ -105,15 +117,16 @@ public class EntityFindingUtility {
return Double.MAX_VALUE;
}
return originalPositions.stream()
.mapToDouble(rectangleWithPage -> calculateMinDistancePerRectangle(entity, rectangleWithPage.pageNumber(), rectangleWithPage.rectangle2D()))
.average()
.mapToDouble(rectangleWithPage -> calculateMinDistancePerRectangle(entity, rectangleWithPage.pageNumber(), rectangleWithPage.rectangle2D())).average()
.orElse(Double.MAX_VALUE);
}
private static long countRectangles(TextEntity entity) {
return entity.getPositionsOnPagePerPage().stream().mapToLong(redactionPosition -> redactionPosition.getRectanglePerLine().size()).sum();
return entity.getPositionsOnPagePerPage()
.stream()
.mapToLong(redactionPosition -> redactionPosition.getRectanglePerLine().size()).sum();
}
@@ -144,24 +157,36 @@ public class EntityFindingUtility {
double maxY2 = Math.max(rectangle2.getMinY(), rectangle2.getMaxY());
return Math.abs(minX1 - minX2) //
+ Math.abs(minY1 - minY2) //
+ Math.abs(maxX1 - maxX2) //
+ Math.abs(maxY1 - maxY2);
+ Math.abs(minY1 - minY2) //
+ Math.abs(maxX1 - maxX2) //
+ Math.abs(maxY1 - maxY2);
}
public Map<String, List<TextEntity>> findAllPossibleEntitiesAndGroupByValue(SemanticNode node, List<PrecursorEntity> manualEntities) {
Set<Integer> pageNumbers = manualEntities.stream().flatMap(entry -> entry.getEntityPosition().stream().map(RectangleWithPage::pageNumber)).collect(Collectors.toSet());
Set<String> entryValues = manualEntities.stream().map(PrecursorEntity::getValue).filter(Objects::nonNull).map(String::toLowerCase).collect(Collectors.toSet());
Set<Integer> pageNumbers = manualEntities.stream()
.flatMap(entry -> entry.getEntityPosition()
.stream()
.map(RectangleWithPage::pageNumber))
.collect(Collectors.toSet());
Set<String> entryValues = manualEntities.stream()
.map(PrecursorEntity::getValue)
.filter(Objects::nonNull)
.map(String::toLowerCase)
.collect(Collectors.toSet());
if (!pageNumbers.stream().allMatch(node::onPage)) {
if (!pageNumbers.stream()
.allMatch(node::onPage)) {
throw new IllegalArgumentException(format("SemanticNode \"%s\" does not contain these pages %s, it has pages: %s",
node,
pageNumbers.stream().filter(pageNumber -> !node.onPage(pageNumber)).toList(),
node.getPages()));
node,
pageNumbers.stream()
.filter(pageNumber -> !node.onPage(pageNumber))
.toList(),
node.getPages()));
}
SearchImplementation searchImplementation = new SearchImplementation(entryValues, true);
SearchImplementation searchImplementation = new SearchImplementation(entryValues.stream().map(String::trim).collect(Collectors.toSet()), true);
return searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
.stream()
@@ -9,7 +9,6 @@ import java.util.Optional;
import java.util.Set;
import java.util.stream.Collectors;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
@@ -23,29 +22,21 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNo
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
import lombok.AccessLevel;
import lombok.RequiredArgsConstructor;
import lombok.experimental.FieldDefaults;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class EntityFromPrecursorCreationService {
static double MATCH_THRESHOLD = 10; // Is compared to the average sum of distances in pdf coordinates for each corner of the bounding box of the entities
EntityFindingUtility entityFindingUtility;
EntityCreationService entityCreationService;
DictionaryService dictionaryService;
@Autowired
public EntityFromPrecursorCreationService(EntityEnrichmentService entityEnrichmentService, DictionaryService dictionaryService, EntityFindingUtility entityFindingUtility) {
this.entityFindingUtility = entityFindingUtility;
entityCreationService = new EntityCreationService(entityEnrichmentService);
this.dictionaryService = dictionaryService;
}
public List<PrecursorEntity> createEntitiesIfFoundAndReturnNotFoundEntries(ManualRedactions manualRedactions, SemanticNode node, String dossierTemplateId) {
Set<IdRemoval> idRemovals = manualRedactions.getIdsToRemove();
@@ -54,7 +45,7 @@ public class EntityFromPrecursorCreationService {
.filter(BaseAnnotation::isLocal)
.filter(manualRedactionEntry -> idRemovals.stream()
.filter(idRemoval -> idRemoval.getAnnotationId().equals(manualRedactionEntry.getAnnotationId()))
.filter(idRemoval -> idRemoval.getRequestDate().isBefore(manualRedactionEntry.getRequestDate()))
.filter(idRemoval -> idRemoval.getRequestDate().isAfter(manualRedactionEntry.getRequestDate()))
.findAny()//
.isEmpty())
.map(manualRedactionEntry -> //
@@ -52,9 +52,13 @@ public class ImportedRedactionEntryService {
private List<BaseAnnotation> allManualChangesExceptAdd(ManualRedactions manualRedactions) {
return Stream.of(manualRedactions.getForceRedactions(),
manualRedactions.getResizeRedactions(),
manualRedactions.getRecategorizations(),
manualRedactions.getIdsToRemove(),
manualRedactions.getLegalBasisChanges()).flatMap(Collection::stream).map(baseAnnotation -> (BaseAnnotation) baseAnnotation).toList();
manualRedactions.getResizeRedactions(),
manualRedactions.getRecategorizations(),
manualRedactions.getIdsToRemove(),
manualRedactions.getLegalBasisChanges())
.flatMap(Collection::stream)
.map(baseAnnotation -> (BaseAnnotation) baseAnnotation)
.toList();
}
}
@@ -14,12 +14,14 @@ public class IntersectingNodeVisitor implements NodeVisitor {
private Set<SemanticNode> intersectingNodes;
private final TextRange textRange;
public IntersectingNodeVisitor(TextRange textRange) {
this.textRange = textRange;
this.intersectingNodes = new HashSet<>();
}
@Override
public void visit(SemanticNode node) {
@@ -31,7 +31,8 @@ public class ManualRedactionEntryService {
List<PrecursorEntity> notFoundManualRedactionEntries = Collections.emptyList();
if (analyzeRequest.getManualRedactions() != null) {
notFoundManualRedactionEntries = entityFromPrecursorCreationService.createEntitiesIfFoundAndReturnNotFoundEntries(analyzeRequest.getManualRedactions(),
document, dossierTemplateId);
document,
dossierTemplateId);
log.info("Added Manual redaction entries for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
}
@@ -51,10 +52,13 @@ public class ManualRedactionEntryService {
private List<BaseAnnotation> allManualChangesExceptAdd(ManualRedactions manualRedactions) {
return Stream.of(manualRedactions.getForceRedactions(),
manualRedactions.getResizeRedactions(),
manualRedactions.getRecategorizations(),
manualRedactions.getIdsToRemove(),
manualRedactions.getLegalBasisChanges()).flatMap(Collection::stream).map(baseAnnotation -> (BaseAnnotation) baseAnnotation).toList();
manualRedactions.getResizeRedactions(),
manualRedactions.getRecategorizations(),
manualRedactions.getIdsToRemove(),
manualRedactions.getLegalBasisChanges())
.flatMap(Collection::stream)
.map(baseAnnotation -> (BaseAnnotation) baseAnnotation)
.toList();
}
}
@@ -12,7 +12,8 @@ import com.iqser.red.service.redaction.v1.server.client.model.NerEntitiesModel;
import com.iqser.red.service.redaction.v1.server.model.NerEntities;
import com.iqser.red.service.redaction.v1.server.model.document.TextRange;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Document;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Section;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.NodeType;
import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNode;
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
import lombok.AccessLevel;
@@ -43,10 +44,12 @@ public class NerEntitiesAdapter {
*/
public NerEntities toNerEntities(NerEntitiesModel nerEntitiesModel, Document document) {
return new NerEntities(addOffsetsAndFlatten(getStringStartOffsetsForMainSections(document),
nerEntitiesModel).map(nerEntityModel -> new NerEntities.NerEntity(nerEntityModel.getValue(),
new TextRange(nerEntityModel.getStartOffset(), nerEntityModel.getEndOffset()),
nerEntityModel.getType())).toList());
return new NerEntities(addOffsetsAndFlatten(getStringStartOffsetsForMainSectionsHeadersFooters(document),
nerEntitiesModel).map(nerEntityModel -> new NerEntities.NerEntity(nerEntityModel.getValue(),
new TextRange(nerEntityModel.getStartOffset(),
nerEntityModel.getEndOffset()),
nerEntityModel.getType()))
.toList());
}
@@ -83,7 +86,9 @@ public class NerEntitiesAdapter {
List<List<NerEntities.NerEntity>> entityClusters = new LinkedList<>();
List<NerEntities.NerEntity> startEntitiesOfEssentialType = sortedEntities.stream().filter(e -> essentialTypes.contains(e.type())).toList();
List<NerEntities.NerEntity> startEntitiesOfEssentialType = sortedEntities.stream()
.filter(e -> essentialTypes.contains(e.type()))
.toList();
for (NerEntities.NerEntity startEntity : startEntitiesOfEssentialType) {
List<NerEntities.NerEntity> currentCluster = new LinkedList<>();
entityClusters.add(currentCluster);
@@ -105,7 +110,10 @@ public class NerEntitiesAdapter {
}
}
return entityClusters.stream().filter(cluster -> cluster.size() >= minPartsToCombine).map(NerEntitiesAdapter::toContainingBoundary).distinct();
return entityClusters.stream()
.filter(cluster -> cluster.size() >= minPartsToCombine)
.map(NerEntitiesAdapter::toContainingBoundary)
.distinct();
}
@@ -124,17 +132,18 @@ public class NerEntitiesAdapter {
public Stream<TextRange> combineNerEntitiesToCbiAddressDefaults(NerEntities entityRecognitionEntities) {
return combineNerEntities(entityRecognitionEntities,
CBI_ADDRESS_ESSENTIAL_TYPES,
CBI_ADDRESS_TYPES_TO_COMBINE,
MAX_DISTANCE_BETWEEN_PARTS,
MIN_PARTS_TO_COMBINE,
ALLOW_DUPLICATES);
CBI_ADDRESS_ESSENTIAL_TYPES,
CBI_ADDRESS_TYPES_TO_COMBINE,
MAX_DISTANCE_BETWEEN_PARTS,
MIN_PARTS_TO_COMBINE,
ALLOW_DUPLICATES);
}
private static boolean isDuplicate(List<NerEntities.NerEntity> currentCluster, NerEntities.NerEntity entity, boolean allowDuplicates) {
return allowDuplicates || currentCluster.stream().anyMatch(e -> e.type().equals(entity.type()));
return allowDuplicates || currentCluster.stream()
.anyMatch(e -> e.type().equals(entity.type()));
}
@@ -146,24 +155,34 @@ public class NerEntitiesAdapter {
private static TextRange toContainingBoundary(List<NerEntities.NerEntity> nerEntities) {
return TextRange.merge(nerEntities.stream().map(NerEntities.NerEntity::textRange).toList());
return TextRange.merge(nerEntities.stream()
.map(NerEntities.NerEntity::textRange)
.toList());
}
private static Stream<EntityRecognitionEntity> addOffsetsAndFlatten(List<Integer> stringOffsetsForMainSections, NerEntitiesModel nerEntitiesModel) {
private static Stream<EntityRecognitionEntity> addOffsetsAndFlatten(List<Integer> stringOffsetsForMainSectionsHeadersFooters, NerEntitiesModel nerEntitiesModel) {
nerEntitiesModel.getData().forEach((sectionNumber, listOfNerEntities) -> listOfNerEntities.forEach(entityRecognitionEntity -> {
int newStartOffset = entityRecognitionEntity.getStartOffset() + stringOffsetsForMainSections.get(sectionNumber);
entityRecognitionEntity.setStartOffset(newStartOffset);
entityRecognitionEntity.setEndOffset(newStartOffset + entityRecognitionEntity.getValue().length());
}));
return nerEntitiesModel.getData().values().stream().flatMap(Collection::stream);
nerEntitiesModel.getData()
.forEach((sectionNumber, listOfNerEntities) -> listOfNerEntities.forEach(entityRecognitionEntity -> {
int newStartOffset = entityRecognitionEntity.getStartOffset() + stringOffsetsForMainSectionsHeadersFooters.get(sectionNumber);
entityRecognitionEntity.setStartOffset(newStartOffset);
entityRecognitionEntity.setEndOffset(newStartOffset + entityRecognitionEntity.getValue().length());
}));
return nerEntitiesModel.getData().values()
.stream()
.flatMap(Collection::stream);
}
private static List<Integer> getStringStartOffsetsForMainSections(Document document) {
private static List<Integer> getStringStartOffsetsForMainSectionsHeadersFooters(Document document) {
return document.getMainSections().stream().map(Section::getTextBlock).map(TextBlock::getTextRange).map(TextRange::start).toList();
return document.streamChildren()
.filter(child -> (child.getType().equals(NodeType.FOOTER) ||child.getType().equals(NodeType.HEADER) ||child.getType().equals(NodeType.SECTION)))
.map(SemanticNode::getTextBlock)
.map(TextBlock::getTextRange)
.map(TextRange::start)
.toList();
}
}
@@ -5,4 +5,5 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNo
public interface NodeVisitor {
void visit(SemanticNode node);
}
@@ -43,8 +43,29 @@ public class PropertiesMapper {
private Rectangle2D parseRectangle2D(String bBox) {
List<Float> floats = Arrays.stream(bBox.split(DocumentStructure.RECTANGLE_DELIMITER)).map(Float::parseFloat).toList();
List<Float> floats = Arrays.stream(bBox.split(DocumentStructure.RECTANGLE_DELIMITER))
.map(Float::parseFloat)
.toList();
return new Rectangle2D.Float(floats.get(0), floats.get(1), floats.get(2), floats.get(3));
}
public static boolean isDuplicateParagraph(Map<String, String> properties) {
return properties.containsKey(DocumentStructure.DuplicateParagraphProperties.UNSORTED_TEXTBLOCK_ID);
}
public static Long[] getUnsortedTextblockIds(Map<String, String> properties) {
return toLongArray(properties.get(DocumentStructure.DuplicateParagraphProperties.UNSORTED_TEXTBLOCK_ID));
}
public static Long[] toLongArray(String ids) {
return Arrays.stream(ids.substring(1, ids.length() - 1).trim().split(",")).map(Long::valueOf).toArray(Long[]::new);
}
}
@@ -11,7 +11,6 @@ import java.util.stream.Stream;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeRequest;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Position;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedaction;
@@ -44,23 +43,14 @@ public class SectionFinderService {
@Timed("redactmanager_findSectionsToReanalyse")
public Set<Integer> findSectionsToReanalyse(DictionaryIncrement dictionaryIncrement,
EntityLog entityLog,
Document document,
AnalyzeRequest analyzeRequest,
ImportedRedactions importedRedactions) {
ImportedRedactions importedRedactions,
Set<String> relevantManuallyModifiedAnnotationIds) {
long start = System.currentTimeMillis();
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
Set<Integer> sectionsToReanalyse = new HashSet<>();
for (EntityLogEntry entry : entityLog.getEntityLogEntry()) {
if (relevantManuallyModifiedAnnotationIds.contains(entry.getId())) {
if (entry.getContainingNodeId().isEmpty()) {
continue; // Empty list means either Entity has not been found or it is between main sections. Thus, this might lead to wrong reanalysis.
}
sectionsToReanalyse.add(entry.getContainingNodeId()
.get(0));
}
}
var dictionaryIncrementsSearch = new SearchImplementation(dictionaryIncrement.getValues()
.stream()
@@ -133,7 +123,7 @@ public class SectionFinderService {
}
private static Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
public static Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
if (manualRedactions == null) {
return new HashSet<>();
@@ -12,6 +12,7 @@ public abstract class SemanticNodeComparators implements Comparator<SemanticNode
return new FirstSemanticNode();
}
public static class FirstSemanticNode extends SemanticNodeComparators {
@Override
@@ -50,7 +50,9 @@ public class ComponentDroolsExecutionService {
.filter(entityLogEntry -> entityLogEntry.getState().equals(EntryState.APPLIED))
.map(entry -> Entity.fromEntityLogEntry(entry, document))
.forEach(kieSession::insert);
fileAttributes.stream().filter(f -> f.getValue() != null).forEach(kieSession::insert);
fileAttributes.stream()
.filter(f -> f.getValue() != null)
.forEach(kieSession::insert);
CompletableFuture<Void> completableFuture = CompletableFuture.supplyAsync(() -> {
kieSession.fireAllRules();
@@ -58,7 +60,8 @@ public class ComponentDroolsExecutionService {
});
try {
completableFuture.orTimeout(settings.getDroolsExecutionTimeoutSecs(), TimeUnit.SECONDS).get();
completableFuture.orTimeout(settings.getDroolsExecutionTimeoutSecs(), TimeUnit.SECONDS)
.get();
} catch (ExecutionException e) {
kieSession.dispose();
if (e.getCause() instanceof TimeoutException) {
@@ -71,7 +74,9 @@ public class ComponentDroolsExecutionService {
}
List<FileAttribute> resultingFileAttributes = getFileAttributes(kieSession);
List<Component> components = getComponents(kieSession).stream().sorted(ComponentComparator.first()).toList();
List<Component> components = getComponents(kieSession).stream()
.sorted(ComponentComparator.first())
.toList();
kieSession.dispose();
return components;
}
@@ -7,6 +7,7 @@ import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import org.drools.drl.parser.DroolsParserException;
import org.kie.api.builder.KieBuilder;
@@ -15,11 +16,13 @@ import org.springframework.stereotype.Service;
import com.google.common.collect.Sets;
import com.iqser.red.service.persistence.service.v1.api.shared.model.RuleFileType;
import com.iqser.red.service.redaction.v1.model.DroolsBlacklistErrorMessage;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxDeprecatedWarnings;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxErrorMessage;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
import com.iqser.red.service.redaction.v1.server.DeprecatedElementsFinder;
import com.iqser.red.service.redaction.v1.server.RedactionServiceSettings;
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
import com.iqser.red.service.redaction.v1.server.model.drools.BasicQuery;
import com.iqser.red.service.redaction.v1.server.model.drools.BasicRule;
@@ -35,31 +38,61 @@ import lombok.extern.slf4j.Slf4j;
@Service
@RequiredArgsConstructor
@Slf4j
public class DroolsSyntaxValidationService {
private static final Pattern allowedImportsPattern = Pattern.compile("^(?:import\\s+static\\s+)?(?:import\\s+)?(?:com\\.knecon\\.fforesight|com\\.iqser\\.red)\\..*;$");
public class DroolsValidationService {
private final RedactionServiceSettings redactionServiceSettings;
private final KieContainerCreationService kieContainerCreationService;
private final DeprecatedElementsFinder deprecatedElementsFinder;
private static final Pattern allowedImportsPattern = Pattern.compile("^(?:import\\s+static\\s+)?(?:import\\s+)?(?:com\\.knecon\\.fforesight|com\\.iqser\\.red)\\..*;$");
public static final String LINEBREAK_MATCHER = "\\R";
@SneakyThrows
public DroolsSyntaxValidation testRules(RuleValidationModel rules) {
public DroolsValidation testRules(RuleValidationModel rules) {
DroolsSyntaxValidation customDroolsSyntaxValidation;
DroolsValidation customDroolsValidation;
try {
customDroolsSyntaxValidation = buildCustomDroolsSyntaxValidation(rules.getRulesString(), RuleFileType.valueOf(rules.getRuleFileType()));
customDroolsValidation = buildCustomDroolsValidation(rules.getRulesString(), RuleFileType.valueOf(rules.getRuleFileType()));
} catch (DroolsParserException e) {
// this means the parser could not parse the file at all. In this case use drools compiler only as it will return useful error messages.
customDroolsSyntaxValidation = new DroolsSyntaxValidation();
customDroolsValidation = new DroolsValidation();
}
DroolsSyntaxValidation droolsCompilerSyntaxValidation = buildDroolsCompilerSyntaxValidation(rules);
droolsCompilerSyntaxValidation.getDroolsSyntaxErrorMessages().addAll(customDroolsSyntaxValidation.getDroolsSyntaxErrorMessages());
droolsCompilerSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().addAll(customDroolsSyntaxValidation.getDroolsSyntaxDeprecatedWarnings());
return droolsCompilerSyntaxValidation;
DroolsValidation droolsCompilerValidation = buildDroolsCompilerValidation(rules);
droolsCompilerValidation.getSyntaxErrorMessages().addAll(customDroolsValidation.getSyntaxErrorMessages());
droolsCompilerValidation.getDeprecatedWarnings().addAll(customDroolsValidation.getDeprecatedWarnings());
droolsCompilerValidation.getBlacklistErrorMessages().addAll(customDroolsValidation.getBlacklistErrorMessages());
return droolsCompilerValidation;
}
private DroolsValidation buildCustomDroolsValidation(String ruleString, RuleFileType ruleFileType) throws DroolsParserException {
RuleFileBluePrint ruleFileBluePrint = RuleFileParser.buildBluePrintFromRulesString(ruleString);
DroolsValidation customValidation = ruleFileBluePrint.getDroolsValidation();
addSyntaxDeprecatedWarnings(ruleFileBluePrint, customValidation);
addSyntaxErrorMessages(ruleFileType, ruleFileBluePrint, customValidation);
if (redactionServiceSettings.isRuleExecutionSecured()) {
addBlacklistErrorMessages(ruleFileBluePrint, customValidation);
}
return customValidation;
}
private void addSyntaxDeprecatedWarnings(RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
// find deprecated elements in the ruleFileBluePrint
DroolsSyntaxDeprecatedWarnings warningMessageForImports = getWarningsForDeprecatedImports(ruleFileBluePrint);
if (warningMessageForImports != null) {
customValidation.getDeprecatedWarnings().add(warningMessageForImports);
}
customValidation.getDeprecatedWarnings().addAll(getWarningsForDeprecatedRules(ruleFileBluePrint));
}
private DroolsSyntaxDeprecatedWarnings getWarningsForDeprecatedImports(RuleFileBluePrint ruleFileBluePrint) {
if (!deprecatedElementsFinder.getDeprecatedClasses().isEmpty()) {
@@ -70,13 +103,13 @@ public class DroolsSyntaxValidationService {
String sb = "Following imports are deprecated: \n" + matches.stream()
.map(m -> imports.substring(m.startIndex(), m.endIndex()))
.collect(Collectors.joining("\n"));
return DroolsSyntaxDeprecatedWarnings.builder().line(ruleFileBluePrint.getImportLine()).column(0).message(sb)
.build();
return DroolsSyntaxDeprecatedWarnings.builder().line(ruleFileBluePrint.getImportLine()).column(0).message(sb).build();
}
}
return null;
}
private List<DroolsSyntaxDeprecatedWarnings> getWarningsForDeprecatedRules(RuleFileBluePrint ruleFileBluePrint) {
List<DroolsSyntaxDeprecatedWarnings> warningMessages = new ArrayList<>();
@@ -96,8 +129,7 @@ public class DroolsSyntaxValidationService {
.distinct()
.map(dm -> String.format("Method %s might be deprecated because of \n %s \n", dm, deprecatedMethodsSignatureMap.get(dm)))
.collect(Collectors.joining("\n"));
warningMessages.add(DroolsSyntaxDeprecatedWarnings.builder().line(basicRule.getLine()).column(0).message(warningMessage)
.build());
warningMessages.add(DroolsSyntaxDeprecatedWarnings.builder().line(basicRule.getLine()).column(0).message(warningMessage).build());
}
}
}
@@ -111,18 +143,7 @@ public class DroolsSyntaxValidationService {
}
private DroolsSyntaxValidation buildCustomDroolsSyntaxValidation(String ruleString, RuleFileType ruleFileType) throws DroolsParserException {
RuleFileBluePrint ruleFileBluePrint = RuleFileParser.buildBluePrintFromRulesString(ruleString);
DroolsSyntaxValidation customSyntaxValidation = ruleFileBluePrint.getDroolsSyntaxValidation();
// find deprecated elements in the ruleFileBluePrint
DroolsSyntaxDeprecatedWarnings warningMessageForImports = getWarningsForDeprecatedImports(ruleFileBluePrint);
if (warningMessageForImports != null) {
customSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().add(warningMessageForImports);
}
customSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().addAll(getWarningsForDeprecatedRules(ruleFileBluePrint));
private void addSyntaxErrorMessages(RuleFileType ruleFileType, RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
RuleFileBluePrint baseRuleFileBluePrint = switch (ruleFileType) {
case ENTITY -> RuleFileParser.buildBluePrintFromRulesString(RuleManagementResources.getBaseRuleFileString());
@@ -130,7 +151,7 @@ public class DroolsSyntaxValidationService {
};
if (!importsAreValid(baseRuleFileBluePrint, ruleFileBluePrint)) {
customSyntaxValidation.getDroolsSyntaxErrorMessages()
customValidation.getSyntaxErrorMessages()
.add(DroolsSyntaxErrorMessage.builder()
.line(ruleFileBluePrint.getImportLine())
.column(0)
@@ -138,7 +159,7 @@ public class DroolsSyntaxValidationService {
.build());
}
if (!ruleFileBluePrint.getGlobals().equals(baseRuleFileBluePrint.getGlobals())) {
customSyntaxValidation.getDroolsSyntaxErrorMessages()
customValidation.getSyntaxErrorMessages()
.add(DroolsSyntaxErrorMessage.builder()
.line(ruleFileBluePrint.getGlobalsLine())
.column(0)
@@ -148,7 +169,7 @@ public class DroolsSyntaxValidationService {
baseRuleFileBluePrint.getQueries()
.forEach(basicQuery -> {
if (!validateQueryIsPresent(basicQuery, ruleFileBluePrint)) {
customSyntaxValidation.getDroolsSyntaxErrorMessages()
customValidation.getSyntaxErrorMessages()
.add(DroolsSyntaxErrorMessage.builder()
.line(basicQuery.getLine())
.column(0)
@@ -159,12 +180,14 @@ public class DroolsSyntaxValidationService {
if (ruleFileType.equals(RuleFileType.ENTITY)) {
String requiredAgendaGroup = "LOCAL_DICTIONARY_ADDS";
if (!validateAgendaGroupIsPresent(ruleFileBluePrint, requiredAgendaGroup)) {
customSyntaxValidation.getDroolsSyntaxErrorMessages()
.add(DroolsSyntaxErrorMessage.builder().line(0).column(0).message(String.format("At least one rule with Agenda-Group '%s' required!", requiredAgendaGroup))
customValidation.getSyntaxErrorMessages()
.add(DroolsSyntaxErrorMessage.builder()
.line(0)
.column(0)
.message(String.format("At least one rule with Agenda-Group '%s' required!", requiredAgendaGroup))
.build());
}
}
return customSyntaxValidation;
}
@@ -196,7 +219,54 @@ public class DroolsSyntaxValidationService {
}
private DroolsSyntaxValidation buildDroolsCompilerSyntaxValidation(RuleValidationModel rules) {
private void addBlacklistErrorMessages(RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
List<DroolsBlacklistErrorMessage> blacklistErrorMessages = new ArrayList<>();
List<String> blacklistedKeywords = parseBlacklistFile(RuleManagementResources.getBlacklistFileString());
// checks the rules for occurrence of blacklisted keyword
if (!blacklistedKeywords.isEmpty()) {
SearchImplementation blacklistedKeywordSearchImplementation = new SearchImplementation(blacklistedKeywords, false);
for (RuleClass ruleClass : ruleFileBluePrint.getRuleClasses()) {
for (RuleUnit ruleUnit : ruleClass.ruleUnits()) {
for (BasicRule basicRule : ruleUnit.rules()) {
List<SearchImplementation.MatchPosition> matches = blacklistedKeywordSearchImplementation.getMatches(basicRule.getCode());
if (!matches.isEmpty()) {
List<String> foundBlacklistedKeywords = matches.stream()
.map(m -> basicRule.getCode().substring(m.startIndex(), m.endIndex()))
.distinct()
.toList();
blacklistErrorMessages.add(DroolsBlacklistErrorMessage.builder()
.line(basicRule.getLine())
.column(0)
.blacklistedKeywords(foundBlacklistedKeywords)
.build());
}
}
}
}
}
customValidation.getBlacklistErrorMessages()
.addAll(blacklistErrorMessages.stream()
.sorted(Comparator.comparingInt(DroolsBlacklistErrorMessage::getLine))
.toList());
}
private List<String> parseBlacklistFile(String blacklistFileString) {
return Stream.of(blacklistFileString.split(LINEBREAK_MATCHER))
.distinct()
.filter(s -> !s.isBlank())
.toList();
}
private DroolsValidation buildDroolsCompilerValidation(RuleValidationModel rules) {
var versionId = System.currentTimeMillis();
var testRules = "test-rules";
@@ -204,25 +274,23 @@ public class DroolsSyntaxValidationService {
versionId,
rules.getRulesString(),
RuleFileType.valueOf(rules.getRuleFileType()));
return buildDroolsCompilerSyntaxValidation(kieBuilder);
return buildDroolsCompilerValidation(kieBuilder);
}
private DroolsSyntaxValidation buildDroolsCompilerSyntaxValidation(KieBuilder kieBuilder) {
private DroolsValidation buildDroolsCompilerValidation(KieBuilder kieBuilder) {
List<Message> errorMessages = kieBuilder.getResults().getMessages(Message.Level.ERROR);
List<DroolsSyntaxErrorMessage> droolsSyntaxErrorMessages = errorMessages.stream()
.map(this::buildDroolsSyntaxErrorMessage)
.collect(Collectors.toList());
return DroolsSyntaxValidation.builder().droolsSyntaxErrorMessages(droolsSyntaxErrorMessages)
.build();
return DroolsValidation.builder().syntaxErrorMessages(droolsSyntaxErrorMessages).build();
}
private DroolsSyntaxErrorMessage buildDroolsSyntaxErrorMessage(Message message) {
return DroolsSyntaxErrorMessage.builder().line(message.getLine()).column(message.getColumn()).message(message.getText())
.build();
return DroolsSyntaxErrorMessage.builder().line(message.getLine()).column(message.getColumn()).message(message.getText()).build();
}
}
@@ -30,8 +30,7 @@ public class KieContainerCreationService {
private final RulesClient rulesClient;
@Observed(name = "KieContainerCreationService",
contextualName = "get-kie-container")
@Observed(name = "KieContainerCreationService", contextualName = "get-kie-container")
public KieWrapper getLatestKieContainer(String dossierTemplateId, RuleFileType ruleFileType) {
try {
@@ -65,7 +64,6 @@ public class KieContainerCreationService {
try {
return kieServices.newKieContainer(getReleaseId(dossierTemplateId, version, ruleFileType));
} catch (Exception e) {
registerNewKieContainerVersion(dossierTemplateId, version, ruleFileType);
return kieServices.newKieContainer(getReleaseId(dossierTemplateId, version, ruleFileType));
}
@@ -16,7 +16,7 @@ import org.drools.drl.ast.descr.RuleDescr;
import org.drools.drl.parser.DrlParser;
import org.kie.internal.builder.conf.LanguageLevelOption;
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
import com.iqser.red.service.redaction.v1.server.model.drools.BasicQuery;
import com.iqser.red.service.redaction.v1.server.model.drools.BasicRule;
import com.iqser.red.service.redaction.v1.server.model.drools.RuleClass;
@@ -38,7 +38,7 @@ public class RuleFileParser {
@SneakyThrows
public RuleFileBluePrint buildBluePrintFromRulesString(String ruleString) {
DroolsSyntaxValidation customDroolsSyntaxValidation = DroolsSyntaxValidation.builder().build();
DroolsValidation customDroolsValidation = DroolsValidation.builder().build();
DrlParser parser = new DrlParser(LanguageLevelOption.DRL6);
PackageDescr packageDescr = parser.parse(false, ruleString);
List<BasicRule> allRules = new LinkedList<>();
@@ -48,11 +48,16 @@ public class RuleFileParser {
if (rule.isQuery()) {
allQueries.add(new BasicQuery(rule.getName(), rule.getLine(), ruleString.substring(rule.getStartCharacter(), rule.getEndCharacter())));
} else {
validateRule(ruleString, rule, customDroolsSyntaxValidation, allRules);
validateRule(ruleString, rule, customDroolsValidation, allRules);
}
}
String imports = ruleString.substring(0, packageDescr.getImports().stream().mapToInt(ImportDescr::getEndCharacter).max().orElseThrow() + 1);
String imports = ruleString.substring(0,
packageDescr.getImports()
.stream()
.mapToInt(ImportDescr::getEndCharacter)
.max()
.orElseThrow() + 1);
String globals = packageDescr.getGlobals()
.stream()
.map(globalDescr -> ruleString.substring(globalDescr.getStartCharacter(), globalDescr.getEndCharacter()))
@@ -61,66 +66,87 @@ public class RuleFileParser {
List<RuleClass> ruleClasses = buildRuleClasses(allRules);
return new RuleFileBluePrint(imports.trim(),
packageDescr.getImports().stream().findFirst().map(ImportDescr::getLine).orElse(0),
globals.trim(),
packageDescr.getGlobals().stream().findFirst().map(GlobalDescr::getLine).orElse(0), allQueries,
ruleClasses,
customDroolsSyntaxValidation);
packageDescr.getImports()
.stream()
.findFirst()
.map(ImportDescr::getLine)
.orElse(0),
globals.trim(),
packageDescr.getGlobals()
.stream()
.findFirst()
.map(GlobalDescr::getLine)
.orElse(0),
allQueries,
ruleClasses, customDroolsValidation);
}
private static void validateRule(String ruleString, RuleDescr rule, DroolsSyntaxValidation customDroolsSyntaxValidation, List<BasicRule> allRules) {
private static void validateRule(String ruleString, RuleDescr rule, DroolsValidation customDroolsValidation, List<BasicRule> allRules) {
BasicRule basicRule;
try {
basicRule = BasicRule.fromRuleDescr(rule, ruleString);
} catch (Exception e) {
customDroolsSyntaxValidation.addErrorMessage(rule.getLine(), rule.getColumn(), "Malformed rule name, correct format is \"\\w+.\\d+.\\d+: <rule description>\"");
customDroolsValidation.addErrorMessage(rule.getLine(), rule.getColumn(), "Malformed rule name, correct format is \"\\w+.\\d+.\\d+: <rule description>\"");
return;
}
if (allRules.contains(basicRule)) {
addDuplicateRuleIdentifierErrorMessage(rule, basicRule, customDroolsSyntaxValidation);
addDuplicateRuleIdentifierErrorMessage(rule, basicRule, customDroolsValidation);
}
validateRuleIdentifierInCodeIsSame(basicRule.getCode(), basicRule.getIdentifier().toString(), rule.getLine(), customDroolsSyntaxValidation);
validateRuleIdentifierInCodeIsSame(basicRule.getCode(), basicRule.getIdentifier().toString(), rule.getLine(), customDroolsValidation);
allRules.add(BasicRule.fromRuleDescr(rule, ruleString));
}
private static void validateRuleIdentifierInCodeIsSame(String code, String identifier, int lineOffset, DroolsSyntaxValidation customDroolsSyntaxValidation) {
private static void validateRuleIdentifierInCodeIsSame(String code, String identifier, int lineOffset, DroolsValidation customDroolsValidation) {
Matcher matcher = ruleIdentifierInCodeFinder.matcher(code);
while (matcher.find()) {
String identifierInCode = code.substring(matcher.start(1), matcher.end(1));
long line = code.substring(0, matcher.start(1)).lines().count() + lineOffset - 1;
long line = code.substring(0, matcher.start(1)).lines()
.count() + lineOffset - 1;
if (!identifier.equals(identifierInCode)) {
customDroolsSyntaxValidation.addErrorMessage((int) line,
0,
String.format("Rule identifier %s is not equal to rule identifier %s in rule name!", identifierInCode, identifier));
customDroolsValidation.addErrorMessage((int) line,
0,
String.format("Rule identifier %s is not equal to rule identifier %s in rule name!", identifierInCode, identifier));
}
}
}
private void addDuplicateRuleIdentifierErrorMessage(RuleDescr rule, BasicRule basicRule, DroolsSyntaxValidation customDroolsSyntaxValidation) {
private void addDuplicateRuleIdentifierErrorMessage(RuleDescr rule, BasicRule basicRule, DroolsValidation customDroolsValidation) {
customDroolsSyntaxValidation.addErrorMessage(rule.getLine(),
rule.getColumn(),
String.format("RuleIdentifier: %s is a duplicate, duplicates are not allowed!", basicRule.getIdentifier()));
customDroolsValidation.addErrorMessage(rule.getLine(),
rule.getColumn(),
String.format("RuleIdentifier: %s is a duplicate, duplicates are not allowed!", basicRule.getIdentifier()));
}
private List<RuleClass> buildRuleClasses(List<BasicRule> allRules) {
List<RuleType> ruleTypeOrder = allRules.stream().map(BasicRule::getIdentifier).map(RuleIdentifier::type).distinct().toList();
Map<RuleType, List<BasicRule>> rulesPerType = allRules.stream().collect(groupingBy(rule -> rule.getIdentifier().type()));
return ruleTypeOrder.stream().map(type -> new RuleClass(type, groupingByGroup(rulesPerType.get(type)))).collect(Collectors.toList());
List<RuleType> ruleTypeOrder = allRules.stream()
.map(BasicRule::getIdentifier)
.map(RuleIdentifier::type)
.distinct()
.toList();
Map<RuleType, List<BasicRule>> rulesPerType = allRules.stream()
.collect(groupingBy(rule -> rule.getIdentifier().type()));
return ruleTypeOrder.stream()
.map(type -> new RuleClass(type, groupingByGroup(rulesPerType.get(type))))
.collect(Collectors.toList());
}
private List<RuleUnit> groupingByGroup(List<BasicRule> rules) {
Map<Integer, List<BasicRule>> rulesPerUnit = rules.stream().collect(groupingBy(rule -> rule.getIdentifier().unit()));
return rulesPerUnit.keySet().stream().sorted().map(unit -> new RuleUnit(unit, rulesPerUnit.get(unit))).collect(Collectors.toList());
Map<Integer, List<BasicRule>> rulesPerUnit = rules.stream()
.collect(groupingBy(rule -> rule.getIdentifier().unit()));
return rulesPerUnit.keySet()
.stream()
.sorted()
.map(unit -> new RuleUnit(unit, rulesPerUnit.get(unit)))
.collect(Collectors.toList());
}
}
@@ -16,6 +16,7 @@ public class ObservedStorageService {
@Observed(name = "RedactionStorageService", contextualName = "get-document-data")
public DocumentData getDocumentData(String dossierId, String fileId) {
return redactionStorageService.getDocumentData(dossierId, fileId);
}
@@ -3,20 +3,24 @@ package com.iqser.red.service.redaction.v1.server.storage;
import java.io.File;
import java.io.FileInputStream;
import java.io.InputStream;
import java.util.Collection;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import org.springframework.cache.annotation.Cacheable;
import org.springframework.stereotype.Service;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.componentlog.ComponentLog;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactionsPerPage;
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.dossier.file.FileType;
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
import com.iqser.red.service.persistence.service.v1.api.shared.mongo.service.EntityLogMongoService;
import com.iqser.red.service.redaction.v1.server.client.model.NerEntitiesModel;
import com.iqser.red.service.redaction.v1.server.model.document.DocumentData;
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
import com.iqser.red.service.redaction.v1.server.utils.exception.NotFoundException;
import com.iqser.red.storage.commons.exception.StorageObjectDoesNotExist;
import com.iqser.red.storage.commons.service.StorageService;
@@ -39,6 +43,8 @@ public class RedactionStorageService {
private final StorageService storageService;
private final EntityLogMongoService entityLogMongoService;
@SneakyThrows
public InputStream getStoredObject(String storageId) {
@@ -75,14 +81,45 @@ public class RedactionStorageService {
}
@SneakyThrows
public void updateEntityLogWithoutEntries(String dossierId, String fileId, EntityLog entityLog) {
entityLogMongoService.saveEntityLogWithoutEntries(dossierId, fileId, entityLog);
}
@SneakyThrows
public void saveEntityLog(String dossierId, String fileId, EntityLog entityLog) {
entityLogMongoService.saveEntityLog(dossierId, fileId, entityLog);
}
@SneakyThrows
public void saveEntityLogEntries(String dossierId, String fileId, List<EntityLogEntry> entityLogEntries) {
entityLogMongoService.saveEntityLogEntries(dossierId, fileId, entityLogEntries);
}
@SneakyThrows
public void updateEntityLogEntries(String dossierId, String fileId, List<EntityLogEntry> entityLogEntries) {
entityLogMongoService.updateEntityLogEntries(dossierId, fileId, entityLogEntries);
}
@Timed("redactmanager_getImportedRedactions")
public ImportedRedactions getImportedRedactions(String dossierId, String fileId) {
try {
ImportedRedactionsPerPage importedRedactionsPerPage = storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
ImportedRedactionsPerPage.class);
return new ImportedRedactions(importedRedactionsPerPage.getImportedRedactions().values().stream().flatMap(List::stream).collect(Collectors.toList()));
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
ImportedRedactionsPerPage.class);
return new ImportedRedactions(importedRedactionsPerPage.getImportedRedactions().values()
.stream()
.flatMap(List::stream)
.collect(Collectors.toList()));
} catch (StorageObjectDoesNotExist e) {
log.debug("Imported redactions not available.");
return new ImportedRedactions();
@@ -90,14 +127,13 @@ public class RedactionStorageService {
}
@Timed("redactmanager_getImportedRedactions")
public ImportedRedactionsPerPage getImportedRedactionsPerPage(String dossierId, String fileId) {
try {
return storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
ImportedRedactionsPerPage.class);
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
ImportedRedactionsPerPage.class);
} catch (StorageObjectDoesNotExist e) {
log.debug("Imported redactions not available.");
return null;
@@ -111,12 +147,12 @@ public class RedactionStorageService {
try {
RedactionLog redactionLog = storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.REDACTION_LOG),
RedactionLog.class);
StorageIdUtils.getStorageId(dossierId, fileId, FileType.REDACTION_LOG),
RedactionLog.class);
redactionLog.setRedactionLogEntry(redactionLog.getRedactionLogEntry()
.stream()
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
.collect(Collectors.toList()));
.stream()
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
.collect(Collectors.toList()));
return redactionLog;
} catch (StorageObjectDoesNotExist e) {
log.debug("RedactionLog not available.");
@@ -130,11 +166,12 @@ public class RedactionStorageService {
public EntityLog getEntityLog(String dossierId, String fileId) {
try {
EntityLog entityLog = storageService.readJSONObject(TenantContext.getTenantId(), StorageIdUtils.getStorageId(dossierId, fileId, FileType.ENTITY_LOG), EntityLog.class);
EntityLog entityLog = entityLogMongoService.findEntityLogByDossierIdAndFileId(dossierId, fileId)
.orElseThrow(() -> new StorageObjectDoesNotExist(""));
entityLog.setEntityLogEntry(entityLog.getEntityLogEntry()
.stream()
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
.collect(Collectors.toList()));
.stream()
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
.collect(Collectors.toList()));
return entityLog;
} catch (StorageObjectDoesNotExist e) {
log.debug("EntityLog not available.");
@@ -144,6 +181,33 @@ public class RedactionStorageService {
}
@Timed("redactmanager_getRedactionLog")
public EntityLog getEntityLogWithoutEntries(String dossierId, String fileId) {
try {
return entityLogMongoService.findEntityLogWithoutEntries(dossierId, fileId)
.orElseThrow(() -> new StorageObjectDoesNotExist(""));
} catch (StorageObjectDoesNotExist e) {
log.debug("EntityLog not available.");
return null;
}
}
public Set<Integer> findIdsOfSectionsToReanalyse(String dossierId, String fileId, Collection<String> entryIds) {
return entityLogMongoService.findFirstContainingNodeIdForEachEntry(dossierId, fileId, entryIds);
}
public List<EntityLogEntry> findEntriesContainedBySectionsOrNotContained(String dossierId, String fileId, Collection<Integer> sectionIds) {
return entityLogMongoService.findEntityLogEntriesNotContainedOrFirstContainedByElementInList(dossierId, fileId, sectionIds);
}
// !Warning! before activating redis cache you need to set
// -Dio.netty.noPreferDirect=true -XX:MaxDirectMemorySize=1000M
// Jvm args to the largest document data size we want to process. for 4443 pages file that was 500mb.
@@ -156,17 +220,17 @@ public class RedactionStorageService {
try {
return DocumentData.builder()
.documentStructure(storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_STRUCTURE),
DocumentStructure.class))
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_STRUCTURE),
DocumentStructure.class))
.documentTextData(storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_TEXT),
DocumentTextData[].class))
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_TEXT),
DocumentTextData[].class))
.documentPositionData(storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_POSITION),
DocumentPositionData[].class))
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_POSITION),
DocumentPositionData[].class))
.documentPages(storageService.readJSONObject(TenantContext.getTenantId(),
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_PAGES),
DocumentPage[].class))
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_PAGES),
DocumentPage[].class))
.build();
} catch (StorageObjectDoesNotExist e) {
log.debug("DocumentData not available.");
@@ -198,7 +262,7 @@ public class RedactionStorageService {
public boolean entityLogExists(String dossierId, String fileId) {
return storageService.objectExists(TenantContext.getTenantId(), StorageIdUtils.getStorageId(dossierId, fileId, FileType.ENTITY_LOG));
return entityLogMongoService.entityLogDocumentExists(dossierId, fileId);
}
@@ -11,6 +11,7 @@ public class RuleManagementResources {
private static final String folderPrefix = "drools";
@SneakyThrows
public static InputStream getBaseRuleFileInputStream() {
@@ -26,6 +27,7 @@ public class RuleManagementResources {
}
}
@SneakyThrows
public static InputStream getBaseComponentRuleFileInputStream() {
@@ -41,4 +43,20 @@ public class RuleManagementResources {
}
}
@SneakyThrows
public static InputStream getBlacklistFileInputStream() {
return new ClassPathResource(Path.of(folderPrefix, "blacklist.txt").toString()).getInputStream();
}
@SneakyThrows
public static String getBlacklistFileString() {
try (var in = getBlacklistFileInputStream()) {
return new String(in.readAllBytes());
}
}
}
@@ -1,10 +1,19 @@
package com.iqser.red.service.redaction.v1.server.utils;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStreamReader;
import java.text.DateFormat;
import java.text.SimpleDateFormat;
import java.time.LocalDate;
import java.time.ZoneId;
import java.time.format.DateTimeFormatter;
import java.time.format.DateTimeFormatterBuilder;
import java.time.format.DateTimeParseException;
import java.time.format.ResolverStyle;
import java.util.Date;
import java.util.List;
import java.util.Locale;
import java.util.Objects;
import java.util.Optional;
import lombok.AccessLevel;
@@ -17,39 +26,65 @@ import lombok.extern.slf4j.Slf4j;
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
public class DateConverter {
static List<SimpleDateFormat> formats = List.of(new SimpleDateFormat("dd MMM yy", Locale.ENGLISH),
new SimpleDateFormat("dd MM yyyy", Locale.ENGLISH),
new SimpleDateFormat("dd MM yyyy.", Locale.ENGLISH),
new SimpleDateFormat("dd MMMM yyyy", Locale.ENGLISH),
new SimpleDateFormat("MMMM dd, yyyy", Locale.ENGLISH),
new SimpleDateFormat("dd-MMM-yyyy", Locale.ENGLISH));
private static DateTimeFormatter DATE_TIME_FORMATTER;
public Optional<Date> parseDate(String dateAsString) {
Date date = null;
for (SimpleDateFormat format : formats) {
try {
date = format.parse(dateAsString);
break;
} catch (Exception e) {
log.warn("Failed to parse date from string {}. \n{}", dateAsString, e.getMessage());
// ignore, try next...
}
}
if (date == null) {
DateTimeFormatter formatter = getDateTimeFormatter();
String cleanDate = dateAsString.trim();
cleanDate = removeTrailingDot(cleanDate);
try {
LocalDate localDate = LocalDate.parse(cleanDate, formatter);
Date date = Date.from(localDate.atStartOfDay(ZoneId.systemDefault()).toInstant());
return Optional.of(date);
} catch (DateTimeParseException e) {
log.warn("Failed to parse date: {}", cleanDate);
return Optional.empty();
}
return Optional.of(date);
}
public String convertDate(Date date, String resultFormat) {
DateFormat resultDateFormat = new SimpleDateFormat(resultFormat, Locale.ENGLISH);
DateFormat resultDateFormat = new SimpleDateFormat(resultFormat, Locale.UK);
return resultDateFormat.format(date);
}
private DateTimeFormatter getDateTimeFormatter() {
if (DATE_TIME_FORMATTER == null) {
DATE_TIME_FORMATTER = createFormatterFromResource();
}
return DATE_TIME_FORMATTER;
}
private DateTimeFormatter createFormatterFromResource() {
DateTimeFormatterBuilder builder = new DateTimeFormatterBuilder();
try (BufferedReader reader = new BufferedReader(new InputStreamReader(Objects.requireNonNull(DateConverter.class.getResourceAsStream("/date_formats.txt"))))) {
String line;
while ((line = reader.readLine()) != null) {
builder.appendOptional(DateTimeFormatter.ofPattern(line.trim(), Locale.UK));
}
} catch (IOException e) {
throw new RuntimeException("Error reading date format file: " + e.getMessage());
}
return builder.toFormatter().withResolverStyle(ResolverStyle.SMART).withLocale(Locale.UK);
}
private String removeTrailingDot(String dateAsString) {
String str = dateAsString;
if (str != null && !str.isEmpty() && str.charAt(str.length() - 1) == '.') {
str = str.substring(0, str.length() - 1);
}
return str;
}
}
@@ -21,7 +21,9 @@ public final class IdBuilder {
public String buildId(Set<Page> pages, List<Rectangle2D> rectanglesPerLine, String type, String entityType) {
return buildId(pages.stream().map(Page::getNumber).collect(Collectors.toList()), rectanglesPerLine, type, entityType);
return buildId(pages.stream()
.map(Page::getNumber)
.collect(Collectors.toList()), rectanglesPerLine, type, entityType);
}
@@ -29,7 +31,9 @@ public final class IdBuilder {
StringBuilder sb = new StringBuilder();
sb.append(type).append(entityType);
List<Integer> sortedPageNumbers = pageNumbers.stream().sorted(Comparator.comparingInt(Integer::intValue)).toList();
List<Integer> sortedPageNumbers = pageNumbers.stream()
.sorted(Comparator.comparingInt(Integer::intValue))
.toList();
sortedPageNumbers.forEach(sb::append);
rectanglesPerLine.forEach(rectangle2D -> sb.append(Math.round(rectangle2D.getX()))
.append(Math.round(rectangle2D.getY()))
@@ -22,19 +22,25 @@ public class RectangleTransformations {
public static Rectangle2D atomicTextBlockBBox(List<AtomicTextBlock> atomicTextBlocks) {
return atomicTextBlocks.stream().flatMap(atomicTextBlock -> atomicTextBlock.getPositions().stream()).collect(new Rectangle2DBBoxCollector());
return atomicTextBlocks.stream()
.flatMap(atomicTextBlock -> atomicTextBlock.getPositions()
.stream())
.collect(new Rectangle2DBBoxCollector());
}
public static Rectangle2D rectangleBBox(List<Position> positions) {
return positions.stream().map(Position::toRectangle2D).collect(new Rectangle2DBBoxCollector());
return positions.stream()
.map(Position::toRectangle2D)
.collect(new Rectangle2DBBoxCollector());
}
public static Rectangle2D rectangle2DBBox(List<Rectangle2D> rectangle2DList) {
return rectangle2DList.stream().collect(new Rectangle2DBBoxCollector());
return rectangle2DList.stream()
.collect(new Rectangle2DBBoxCollector());
}
@@ -49,7 +55,9 @@ public class RectangleTransformations {
if (rectangle2DList.isEmpty()) {
return Collections.emptyList();
}
double splitThreshold = rectangle2DList.stream().mapToDouble(RectangularShape::getWidth).average().orElse(5) * 5.0;
double splitThreshold = rectangle2DList.stream()
.mapToDouble(RectangularShape::getWidth).average()
.orElse(5) * 5.0;
List<List<Rectangle2D>> rectangleListsWithGaps = new LinkedList<>();
List<Rectangle2D> rectangleListWithoutGaps = new LinkedList<>();
@@ -66,7 +74,9 @@ public class RectangleTransformations {
previousRectangle = currentRectangle;
}
}
return rectangleListsWithGaps.stream().map(RectangleTransformations::rectangle2DBBox).toList();
return rectangleListsWithGaps.stream()
.map(RectangleTransformations::rectangle2DBBox)
.toList();
}
@@ -96,9 +106,9 @@ public class RectangleTransformations {
public BinaryOperator<BBox> combiner() {
return (b1, b2) -> new BBox(Math.min(b1.lowerLeftX, b2.lowerLeftX),
Math.min(b1.lowerLeftY, b2.lowerLeftY),
Math.max(b1.upperRightX, b2.upperRightX),
Math.max(b1.upperRightY, b2.upperRightY));
Math.min(b1.lowerLeftY, b2.lowerLeftY),
Math.max(b1.upperRightX, b2.upperRightX),
Math.max(b1.upperRightY, b2.upperRightY));
}

Some files were not shown because too many files have changed in this diff Show More