Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d4e728350d | ||
|
|
b410067b8c | ||
|
|
73d3ed625a | ||
|
|
e9176db88f | ||
|
|
ef72cc6861 | ||
|
|
66b2d52d40 | ||
|
|
dd1b838a5c |
@@ -12,8 +12,7 @@ mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
echo "dependency-check:aggregate"
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
org.owasp:dependency-check-maven:aggregate \
|
||||
-DknownExploitedEnabled=false
|
||||
org.owasp:dependency-check-maven:aggregate
|
||||
|
||||
if [[ -z "${bamboo_repository_pr_key}" ]]
|
||||
then
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
<artifactId>redaction-service</artifactId>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<version>4.0-SNAPSHOT</version>
|
||||
<version>3.0-SNAPSHOT</version>
|
||||
|
||||
|
||||
<packaging>pom</packaging>
|
||||
|
||||
@@ -32,7 +32,7 @@
|
||||
<dependency>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<artifactId>platform-commons-dependency</artifactId>
|
||||
<version>1.22.0</version>
|
||||
<version>1.21.0</version>
|
||||
<scope>import</scope>
|
||||
<type>pom</type>
|
||||
</dependency>
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
<artifactId>redaction-service-api-v1</artifactId>
|
||||
|
||||
<properties>
|
||||
<persistence-service.version>2.1.0</persistence-service.version>
|
||||
<persistence-service.version>1.299.0</persistence-service.version>
|
||||
</properties>
|
||||
|
||||
<dependencies>
|
||||
@@ -32,18 +32,13 @@
|
||||
|
||||
<dependency>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>persistence-service-internal-api-v1</artifactId>
|
||||
<artifactId>persistence-service-api-v1</artifactId>
|
||||
<version>${persistence-service.version}</version>
|
||||
<exclusions>
|
||||
|
||||
<exclusion>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>redaction-service-api-v1</artifactId>
|
||||
</exclusion>
|
||||
<exclusion>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>persistence-service-api-v1</artifactId>
|
||||
</exclusion>
|
||||
</exclusions>
|
||||
</dependency>
|
||||
|
||||
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class AnalyzeRequest {
|
||||
|
||||
private MessageType messageType;
|
||||
|
||||
private String dossierId;
|
||||
private String fileId;
|
||||
private String dossierTemplateId;
|
||||
private ManualRedactions manualRedactions;
|
||||
private OffsetDateTime lastProcessed;
|
||||
private int analysisNumber;
|
||||
|
||||
@Builder.Default
|
||||
private Set<Integer> excludedPages = new HashSet<>();
|
||||
@Builder.Default
|
||||
private Set<Integer> sectionsToReanalyse = new HashSet<>();
|
||||
|
||||
@Builder.Default
|
||||
private List<FileAttribute> fileAttributes = new ArrayList<>();
|
||||
|
||||
}
|
||||
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class AnalyzeResult {
|
||||
|
||||
private MessageType messageType;
|
||||
|
||||
private String dossierId;
|
||||
private String fileId;
|
||||
private long duration;
|
||||
private int numberOfPages;
|
||||
private boolean hasUpdates;
|
||||
private long dictionaryVersion;
|
||||
private long dossierDictionaryVersion;
|
||||
private long rulesVersion;
|
||||
private long legalBasisVersion;
|
||||
|
||||
private boolean wasReanalyzed;
|
||||
|
||||
private int analysisVersion;
|
||||
private int analysisNumber;
|
||||
|
||||
private ManualRedactions manualRedactions;
|
||||
|
||||
private Set<FileAttribute> addedFileAttributes;
|
||||
|
||||
}
|
||||
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class CellRectangle {
|
||||
|
||||
private Point topLeft;
|
||||
private float width;
|
||||
private float height;
|
||||
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class Change {
|
||||
|
||||
private int analysisNumber;
|
||||
private ChangeType type;
|
||||
private OffsetDateTime dateTime;
|
||||
|
||||
}
|
||||
+7
@@ -0,0 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum ChangeType {
|
||||
ADDED,
|
||||
REMOVED,
|
||||
CHANGED
|
||||
}
|
||||
+7
@@ -0,0 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum Engine {
|
||||
DICTIONARY,
|
||||
NER,
|
||||
RULE
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class FileAttribute {
|
||||
|
||||
private String id;
|
||||
private String label;
|
||||
private String placeholder;
|
||||
private String value;
|
||||
|
||||
}
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class ImportedRedaction {
|
||||
|
||||
private String id;
|
||||
|
||||
@Builder.Default
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
|
||||
}
|
||||
+24
@@ -0,0 +1,24 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class ImportedRedactions {
|
||||
|
||||
@Builder.Default
|
||||
private Map<Integer, List<ImportedRedaction>> importedRedactions = new HashMap<>();
|
||||
|
||||
}
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.BaseAnnotation;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@Builder
|
||||
public class ManualChange {
|
||||
|
||||
private AnnotationStatus annotationStatus;
|
||||
private ManualRedactionType manualRedactionType;
|
||||
private OffsetDateTime processedDate;
|
||||
private OffsetDateTime requestedDate;
|
||||
private String userId;
|
||||
private Map<String, String> propertyChanges = new HashMap<>();
|
||||
|
||||
|
||||
public static ManualChange from(BaseAnnotation baseAnnotation) {
|
||||
|
||||
ManualChange manualChange = new ManualChange();
|
||||
manualChange.annotationStatus = baseAnnotation.getStatus();
|
||||
manualChange.processedDate = baseAnnotation.getProcessedDate();
|
||||
manualChange.requestedDate = baseAnnotation.getRequestDate();
|
||||
manualChange.userId = baseAnnotation.getUser();
|
||||
return manualChange;
|
||||
}
|
||||
|
||||
|
||||
public boolean isProcessed() {
|
||||
|
||||
return processedDate != null;
|
||||
}
|
||||
|
||||
|
||||
public ManualChange withManualRedactionType(ManualRedactionType manualRedactionType) {
|
||||
|
||||
this.manualRedactionType = manualRedactionType;
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
public ManualChange withChange(String property, String value) {
|
||||
|
||||
this.propertyChanges.put(property, value);
|
||||
return this;
|
||||
}
|
||||
|
||||
}
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum ManualRedactionType {
|
||||
ADD_LOCALLY,
|
||||
ADD_TO_DICTIONARY,
|
||||
REMOVE_LOCALLY,
|
||||
REMOVE_FROM_DICTIONARY,
|
||||
FORCE_REDACT,
|
||||
FORCE_HINT,
|
||||
RECATEGORIZE,
|
||||
LEGAL_BASIS_CHANGE,
|
||||
RESIZE
|
||||
}
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum MessageType {
|
||||
|
||||
ANALYSE,
|
||||
REANALYSE,
|
||||
STRUCTURE_ANALYSE,
|
||||
SURROUNDING_TEXT
|
||||
|
||||
}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class Point {
|
||||
|
||||
private float x;
|
||||
private float y;
|
||||
|
||||
}
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class ReanalyzeResult {
|
||||
|
||||
private RedactionLog redactionLog;
|
||||
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class Rectangle {
|
||||
|
||||
private Point topLeft;
|
||||
private float width;
|
||||
private float height;
|
||||
|
||||
private int page;
|
||||
|
||||
}
|
||||
+37
@@ -0,0 +1,37 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@CompiledJson
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class RedactionLog {
|
||||
|
||||
/**
|
||||
* Version 0 Redaction Logs have manual redactions merged inside them
|
||||
* Version 1 Redaction Logs only contain system ( rule/dictionary ) redactions. Manual Redactions are merged in at runtime.
|
||||
*/
|
||||
private long analysisVersion;
|
||||
|
||||
/**
|
||||
* Which analysis created this redactionLog.
|
||||
*/
|
||||
private int analysisNumber;
|
||||
|
||||
private List<RedactionLogEntry> redactionLogEntry = new ArrayList<>();
|
||||
private List<RedactionLogLegalBasis> legalBasis = new ArrayList<>();
|
||||
|
||||
private long dictionaryVersion = -1;
|
||||
private long dossierDictionaryVersion = -1;
|
||||
private long rulesVersion = -1;
|
||||
private long legalBasisVersion = -1;
|
||||
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class RedactionLogChanges {
|
||||
|
||||
private RedactionLog redactionLog;
|
||||
private boolean hasChanges;
|
||||
|
||||
}
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionLogComment {
|
||||
|
||||
private long id;
|
||||
private String user;
|
||||
private String text;
|
||||
private String annotationId;
|
||||
private String fileId;
|
||||
private OffsetDateTime date;
|
||||
private OffsetDateTime softDeletedTime;
|
||||
|
||||
}
|
||||
+96
@@ -0,0 +1,96 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
|
||||
import lombok.*;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
@EqualsAndHashCode
|
||||
public class RedactionLogEntry {
|
||||
|
||||
private String id;
|
||||
private String type;
|
||||
private String value;
|
||||
private String reason;
|
||||
private int matchedRule;
|
||||
private boolean rectangle;
|
||||
private String legalBasis;
|
||||
|
||||
private boolean imported;
|
||||
|
||||
private boolean redacted;
|
||||
private boolean isHint;
|
||||
private boolean isRecommendation;
|
||||
private boolean isFalsePositive;
|
||||
|
||||
private String section;
|
||||
private float[] color;
|
||||
|
||||
@Builder.Default
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
private int sectionNumber;
|
||||
|
||||
private String textBefore;
|
||||
private String textAfter;
|
||||
|
||||
@Builder.Default
|
||||
private List<RedactionLogComment> comments = new ArrayList<>();
|
||||
|
||||
private int startOffset;
|
||||
private int endOffset;
|
||||
|
||||
private boolean isImage;
|
||||
private boolean imageHasTransparency;
|
||||
|
||||
private boolean isDictionaryEntry;
|
||||
private boolean isDossierDictionaryEntry;
|
||||
|
||||
private boolean excluded;
|
||||
|
||||
private String sourceId;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
@Builder.Default
|
||||
private List<Change> changes = new ArrayList<>();
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
@Builder.Default
|
||||
private List<ManualChange> manualChanges = new ArrayList<>();
|
||||
|
||||
private Set<Engine> engines = new HashSet<>();
|
||||
|
||||
private Set<String> reference = new HashSet<>();
|
||||
|
||||
@Builder.Default
|
||||
private Set<String> importedRedactionIntersections = new HashSet<>();
|
||||
|
||||
|
||||
public boolean lastChangeIsRemoved() {
|
||||
|
||||
return last(changes).map(c -> c.getType() == ChangeType.REMOVED).orElse(false);
|
||||
}
|
||||
|
||||
|
||||
public boolean isLocalManualRedaction() {
|
||||
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
}
|
||||
|
||||
|
||||
public boolean isManuallyRemoved() {
|
||||
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
}
|
||||
|
||||
|
||||
private <T> Optional<T> last(List<T> list) {
|
||||
|
||||
return list.isEmpty() ? Optional.empty() : Optional.of(list.get(list.size() - 1));
|
||||
}
|
||||
|
||||
}
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionLogLegalBasis {
|
||||
|
||||
private String name;
|
||||
private String description;
|
||||
private String reason;
|
||||
|
||||
}
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.Type;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionRequest {
|
||||
|
||||
private String dossierId;
|
||||
private String fileId;
|
||||
private String dossierTemplateId;
|
||||
private ManualRedactions manualRedactions;
|
||||
@Builder.Default
|
||||
private Set<Integer> excludedPages = new HashSet<>();
|
||||
|
||||
private Colors colors;
|
||||
private List<Type> types;
|
||||
|
||||
private boolean includeFalsePositives;
|
||||
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionResult {
|
||||
|
||||
private byte[] document;
|
||||
private int numberOfPages;
|
||||
|
||||
}
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class SectionArea {
|
||||
|
||||
private Point topLeft;
|
||||
private float width;
|
||||
private float height;
|
||||
private int page;
|
||||
private String header;
|
||||
|
||||
|
||||
public boolean contains(Rectangle other) {
|
||||
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft()
|
||||
.getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft()
|
||||
.getY() + other.getHeight();
|
||||
}
|
||||
|
||||
|
||||
// TODO we should only use one rectangle class.
|
||||
public boolean contains(com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle other) {
|
||||
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft()
|
||||
.getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
|
||||
}
|
||||
|
||||
}
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
@Data
|
||||
@CompiledJson
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class SectionGrid {
|
||||
|
||||
private Map<Integer, List<SectionRectangle>> rectanglesPerPage = new HashMap<>();
|
||||
|
||||
private List<SectionGridSection> sections = new ArrayList<>();
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public static class SectionGridSection {
|
||||
|
||||
private int sectionNumber;
|
||||
private String headline;
|
||||
private Set<Integer> pages;
|
||||
private List<SectionArea> sectionAreas;
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class SectionRectangle {
|
||||
|
||||
private Point topLeft;
|
||||
private float width;
|
||||
private float height;
|
||||
private int part;
|
||||
private int numberOfParts;
|
||||
|
||||
private List<CellRectangle> tableCells;
|
||||
|
||||
}
|
||||
+26
@@ -1,12 +1,38 @@
|
||||
package com.iqser.red.service.redaction.v1.resources;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLog;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionResult;
|
||||
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.web.bind.annotation.PathVariable;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
|
||||
public interface RedactionResource {
|
||||
|
||||
@PostMapping(value = "/debug/classifications", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
RedactionResult classify(@RequestBody RedactionRequest redactionRequest);
|
||||
|
||||
|
||||
@PostMapping(value = "/debug/sections", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
RedactionResult sections(@RequestBody RedactionRequest redactionRequest);
|
||||
|
||||
|
||||
@PostMapping(value = "/debug/htmlTables", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest);
|
||||
|
||||
|
||||
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
void testRules(@RequestBody String rules);
|
||||
|
||||
|
||||
@PostMapping(value = "/redaction-log/preview", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
RedactionLog getRedactionLog(@RequestBody RedactionRequest redactionRequest);
|
||||
|
||||
|
||||
@PostMapping(value = "/manual/surrounding-text/{dossierId}/{fileId}", consumes = MediaType.APPLICATION_JSON_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
|
||||
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions);
|
||||
|
||||
}
|
||||
|
||||
+5
-8
@@ -1,5 +1,9 @@
|
||||
package com.iqser.red.service.redaction.v1.server;
|
||||
|
||||
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
|
||||
import org.springframework.boot.SpringApplication;
|
||||
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.SpringBootApplication;
|
||||
@@ -9,17 +13,10 @@ import org.springframework.cloud.openfeign.EnableFeignClients;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Import;
|
||||
|
||||
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.multitenancy.AsyncConfig;
|
||||
import com.iqser.red.service.redaction.v1.server.multitenancy.MultiTenancyMessagingConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.multitenancy.MultiTenancyWebConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
|
||||
import io.micrometer.core.aop.TimedAspect;
|
||||
import io.micrometer.core.instrument.MeterRegistry;
|
||||
|
||||
@Import({MultiTenancyWebConfiguration.class, AsyncConfig.class, MultiTenancyMessagingConfiguration.class})
|
||||
@Import({DefaultWebMvcConfiguration.class})
|
||||
@EnableFeignClients(basePackageClasses = RulesClient.class)
|
||||
@EnableConfigurationProperties(RedactionServiceSettings.class)
|
||||
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
|
||||
|
||||
+2
-2
@@ -3,7 +3,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionGrid;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionGrid;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryVersion;
|
||||
|
||||
import lombok.Data;
|
||||
@@ -14,7 +14,7 @@ import lombok.NoArgsConstructor;
|
||||
public class Document {
|
||||
|
||||
private List<Page> pages = new ArrayList<>();
|
||||
private List<Paragraph> paragraphs = new ArrayList<>();
|
||||
private List<Section> sections = new ArrayList<>();
|
||||
private List<Header> headers = new ArrayList<>();
|
||||
private List<Footer> footers = new ArrayList<>();
|
||||
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
|
||||
|
||||
+3
-1
@@ -3,7 +3,9 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
|
||||
|
||||
+13
-1
@@ -1,7 +1,13 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
@@ -10,10 +16,11 @@ import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
public class Paragraph implements Comparable {
|
||||
public class Section implements Comparable {
|
||||
|
||||
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
|
||||
private List<PdfImage> images = new ArrayList<>();
|
||||
@@ -62,4 +69,9 @@ public class Paragraph implements Comparable {
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
}
|
||||
+10
-3
@@ -1,11 +1,19 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
@@ -13,8 +21,6 @@ import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@@ -27,6 +33,7 @@ public class SectionText {
|
||||
|
||||
private boolean isTable;
|
||||
private String headline;
|
||||
List<Paragraph> paragraphs;
|
||||
|
||||
@Builder.Default
|
||||
private List<SectionArea> sectionAreas = new ArrayList<>();
|
||||
|
||||
+5
-3
@@ -29,6 +29,8 @@ public class TextBlock extends AbstractTextContainer {
|
||||
@JsonIgnore
|
||||
private int rotation;
|
||||
|
||||
private int indexOnPage;
|
||||
|
||||
@JsonIgnore
|
||||
private String mostPopularWordFont;
|
||||
|
||||
@@ -184,8 +186,8 @@ public class TextBlock extends AbstractTextContainer {
|
||||
}
|
||||
|
||||
|
||||
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation) {
|
||||
|
||||
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation, int indexOnPage) {
|
||||
this.indexOnPage = indexOnPage;
|
||||
this.minX = minX;
|
||||
this.maxX = maxX;
|
||||
this.minY = minY;
|
||||
@@ -248,7 +250,7 @@ public class TextBlock extends AbstractTextContainer {
|
||||
|
||||
public TextBlock copy() {
|
||||
|
||||
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation);
|
||||
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation, indexOnPage);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+48
-10
@@ -30,13 +30,15 @@ public class BlockificationService {
|
||||
* This method is building blocks by expanding the minX/maxX and minY/maxY value on each word that is not split by the conditions.
|
||||
* This method must use text direction adjusted postions (DirAdj). Where {0,0} is on the upper left. Never try to change this!
|
||||
* Rulings (Table lines) must be adjusted to the text directions as well, when checking if a block is split by a ruling.
|
||||
* @param textPositions The words of a page.
|
||||
*
|
||||
* @param textPositions The words of a page.
|
||||
* @param horizontalRulingLines Horizontal table lines.
|
||||
* @param verticalRulingLines Vertical table lines.
|
||||
* @param verticalRulingLines Vertical table lines.
|
||||
* @return Page object that contains the Textblock and text statistics.
|
||||
*/
|
||||
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
|
||||
|
||||
int indexOnPage = 0;
|
||||
List<TextPositionSequence> chunkWords = new ArrayList<>();
|
||||
List<AbstractTextContainer> chunkBlockList1 = new ArrayList<>();
|
||||
|
||||
@@ -62,7 +64,9 @@ public class BlockificationService {
|
||||
prevOrientation = chunkBlockList1.get(chunkBlockList1.size() - 1).getOrientation();
|
||||
}
|
||||
|
||||
TextBlock cb1 = buildTextBlock(chunkWords);
|
||||
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
|
||||
indexOnPage++;
|
||||
|
||||
chunkBlockList1.add(cb1);
|
||||
chunkWords = new ArrayList<>();
|
||||
|
||||
@@ -102,7 +106,7 @@ public class BlockificationService {
|
||||
}
|
||||
}
|
||||
|
||||
TextBlock cb1 = buildTextBlock(chunkWords);
|
||||
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
|
||||
if (cb1 != null) {
|
||||
chunkBlockList1.add(cb1);
|
||||
}
|
||||
@@ -163,7 +167,7 @@ public class BlockificationService {
|
||||
}
|
||||
|
||||
|
||||
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList) {
|
||||
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList, int indexOnPage) {
|
||||
|
||||
TextBlock textBlock = null;
|
||||
|
||||
@@ -182,7 +186,13 @@ public class BlockificationService {
|
||||
styleFrequencyCounter.add(wordBlock.getFontStyle());
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = new TextBlock(wordBlock.getMinXDirAdj(), wordBlock.getMaxXDirAdj(), wordBlock.getMinYDirAdj(), wordBlock.getMaxYDirAdj(), wordBlockList, wordBlock.getRotation());
|
||||
textBlock = new TextBlock(wordBlock.getMinXDirAdj(),
|
||||
wordBlock.getMaxXDirAdj(),
|
||||
wordBlock.getMinYDirAdj(),
|
||||
wordBlock.getMaxYDirAdj(),
|
||||
wordBlockList,
|
||||
wordBlock.getRotation(),
|
||||
indexOnPage);
|
||||
} else {
|
||||
TextBlock spatialEntity = textBlock.union(wordBlock);
|
||||
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity.getHeight());
|
||||
@@ -213,10 +223,38 @@ public class BlockificationService {
|
||||
List<Ruling> horizontalRulingLines,
|
||||
List<Ruling> verticalRulingLines) {
|
||||
|
||||
return isSplitByRuling(maxX, minY, word.getMinXDirAdj(), word.getMinYDirAdj(), verticalRulingLines, word.getDir().getDegrees(), word.getPageWidth(), word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX, minY, word.getMinXDirAdj(), word.getMaxYDirAdj(), horizontalRulingLines, word.getDir().getDegrees(), word.getPageWidth(), word.getPageHeight()) //
|
||||
|| isSplitByRuling(maxX, minY, word.getMinXDirAdj(), word.getMinYDirAdj(), horizontalRulingLines, word.getDir().getDegrees(), word.getPageWidth(), word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX, minY, word.getMinXDirAdj(), word.getMaxYDirAdj(), verticalRulingLines, word.getDir().getDegrees(), word.getPageWidth(), word.getPageHeight()); //
|
||||
return isSplitByRuling(maxX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMinYDirAdj(),
|
||||
verticalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMaxYDirAdj(),
|
||||
horizontalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(maxX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMinYDirAdj(),
|
||||
horizontalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMaxYDirAdj(),
|
||||
verticalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()); //
|
||||
}
|
||||
|
||||
|
||||
|
||||
+2
-2
@@ -4,8 +4,8 @@ import java.util.List;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Point;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ import java.util.regex.Pattern;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.utils;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
|
||||
import lombok.experimental.UtilityClass;
|
||||
|
||||
+6
-3
@@ -25,13 +25,16 @@ public final class RulingTextDirAdjustUtil {
|
||||
private Point2D convertPoint(float x, float y, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
var xAdj = getXRot(x, y, dir, pageWidth, pageHeight);
|
||||
var yLowerLeftRot = getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
|
||||
var yAdj = dir == 0 || dir == 180 ? pageHeight - yLowerLeftRot : pageWidth - yLowerLeftRot;
|
||||
var yAdj = 0f;
|
||||
if (dir == 0 || dir == 180) {
|
||||
yAdj = pageHeight - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
|
||||
} else {
|
||||
yAdj = pageWidth - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
|
||||
}
|
||||
return new Point2D.Float(xAdj, yAdj);
|
||||
}
|
||||
|
||||
|
||||
@SuppressWarnings("SuspiciousNameCombination")
|
||||
private float getXRot(float x, float y, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
if (dir == 0) {
|
||||
|
||||
+2
-2
@@ -2,9 +2,9 @@ package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.internal.resources.DictionaryResource;
|
||||
import com.iqser.red.service.persistence.service.v1.api.resources.DictionaryResource;
|
||||
|
||||
@FeignClient(name = "DictionaryResource", url = "${persistence-service.url}")
|
||||
public interface DictionaryClient extends DictionaryResource {
|
||||
|
||||
}
|
||||
}
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.EntityRecognitionRequest;
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
|
||||
|
||||
@FeignClient(name = "EntityRecognitionClient", url = "${entity-recognition-service.url}")
|
||||
public interface EntityRecognitionClient {
|
||||
|
||||
@PostMapping(value = "/find_authors", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
NerEntities findAuthors(EntityRecognitionRequest entityRecognitionRequest);
|
||||
|
||||
}
|
||||
+1
-1
@@ -2,7 +2,7 @@ package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.internal.resources.FileStatusProcessingUpdateResource;
|
||||
import com.iqser.red.service.persistence.service.v1.api.resources.FileStatusProcessingUpdateResource;
|
||||
|
||||
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${persistence-service.url}")
|
||||
public interface FileStatusProcessingUpdateClient extends FileStatusProcessingUpdateResource {
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@ package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.internal.resources.LegalBasisMappingResource;
|
||||
import com.iqser.red.service.persistence.service.v1.api.resources.LegalBasisMappingResource;
|
||||
|
||||
@FeignClient(name = "LegalBasisMappingResource", url = "${persistence-service.url}")
|
||||
public interface LegalBasisClient extends LegalBasisMappingResource {
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@ package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.internal.resources.RulesResource;
|
||||
import com.iqser.red.service.persistence.service.v1.api.resources.RulesResource;
|
||||
|
||||
@FeignClient(name = "RulesResource", url = "${persistence-service.url}")
|
||||
public interface RulesClient extends RulesResource {
|
||||
|
||||
-10
@@ -1,10 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.internal.resources.TenantsResource;
|
||||
|
||||
@FeignClient(name = "TenantsResource", url = "${persistence-service.url}")
|
||||
public interface TenantsClient extends TenantsResource {
|
||||
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client.model;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class EntityRecognitionRequest {
|
||||
|
||||
private List<EntityRecognitionSection> data;
|
||||
|
||||
}
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client.model;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class EntityRecognitionResult {
|
||||
|
||||
@Builder.Default
|
||||
private Map<Integer, List<EntityRecogintionEntity>> entities = new HashMap<>();
|
||||
|
||||
}
|
||||
+139
@@ -1,11 +1,31 @@
|
||||
package com.iqser.red.service.redaction.v1.server.controller;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.web.bind.annotation.PathVariable;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLog;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionResult;
|
||||
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogMergeService;
|
||||
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
|
||||
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
@@ -15,7 +35,100 @@ import lombok.extern.slf4j.Slf4j;
|
||||
@RequiredArgsConstructor
|
||||
public class RedactionController implements RedactionResource {
|
||||
|
||||
private final PdfVisualisationService pdfVisualisationService;
|
||||
private final DroolsExecutionService droolsExecutionService;
|
||||
private final PdfSegmentationService pdfSegmentationService;
|
||||
private final RedactionStorageService redactionStorageService;
|
||||
private final RedactionLogMergeService redactionLogMergeService;
|
||||
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionResult classify(@RequestBody RedactionRequest redactionRequest) {
|
||||
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
pdfVisualisationService.visualizeClassifications(classifiedDoc, pdDocument);
|
||||
|
||||
return convert(pdDocument, classifiedDoc.getPages().size());
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
|
||||
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
pdfVisualisationService.visualizeParagraphs(classifiedDoc, pdDocument);
|
||||
return convert(pdDocument, classifiedDoc.getPages().size());
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest) {
|
||||
|
||||
Document classifiedDoc;
|
||||
|
||||
try {
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
} catch (Exception e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
for (Page page : classifiedDoc.getPages()) {
|
||||
for (AbstractTextContainer textContainer : page.getTextBlocks()) {
|
||||
if (textContainer instanceof Table) {
|
||||
Table table = (Table) textContainer;
|
||||
sb.append(table.getTextAsHtml()).append("<br />").append("<br />");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return RedactionResult.builder().document(sb.toString().getBytes()).build();
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
@@ -28,4 +141,30 @@ public class RedactionController implements RedactionResource {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionLog getRedactionLog(RedactionRequest redactionRequest) {
|
||||
|
||||
return redactionLogMergeService.provideRedactionLog(redactionRequest);
|
||||
}
|
||||
|
||||
|
||||
private RedactionResult convert(PDDocument document, int numberOfPages) throws IOException {
|
||||
|
||||
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
|
||||
document.save(byteArrayOutputStream);
|
||||
return RedactionResult.builder().document(byteArrayOutputStream.toByteArray()).numberOfPages(numberOfPages).build();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions) {
|
||||
|
||||
var result = manualRedactionSurroundingTextService.addSurroundingText(dossierId, fileId, manualRedactions);
|
||||
log.info("Added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", dossierId, fileId, result.getDuration());
|
||||
return result.getManualRedactions();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class AtomicTextBlockData {
|
||||
Long id;
|
||||
String searchText;
|
||||
int start;
|
||||
int end;
|
||||
int[] lineBreaks;
|
||||
int[] stringIdxToPositionIdx;
|
||||
float[][] positions;
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class DocumentData {
|
||||
List<PageData> pages;
|
||||
List<AtomicTextBlockData> atomicTextBlocks;
|
||||
TableOfContentsData tableOfContents;
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class PageData {
|
||||
int number;
|
||||
int height;
|
||||
int width;
|
||||
|
||||
Long header;
|
||||
Long footer;
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.List;
|
||||
|
||||
import javax.management.openmbean.InvalidKeyException;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class TableOfContentsData {
|
||||
|
||||
List<EntryData> entries;
|
||||
|
||||
|
||||
public EntryData get(String tocId) {
|
||||
|
||||
List<Integer> ids = getIds(tocId);
|
||||
if (ids.size() < 1) {
|
||||
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", tocId));
|
||||
}
|
||||
EntryData entry = entries.get(ids.get(0));
|
||||
for (int id : ids.subList(1, ids.size())) {
|
||||
entry = entry.subEntries().get(id);
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
|
||||
private static List<Integer> getIds(String idsAsString) {
|
||||
|
||||
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
|
||||
}
|
||||
|
||||
|
||||
@Builder
|
||||
public record EntryData(String tocId, List<EntryData> subEntries, NodeType type, Long atomicTextBlock, Long page, int numberOnPage) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import lombok.Setter;
|
||||
|
||||
@Setter
|
||||
public class Boundary {
|
||||
|
||||
private int start;
|
||||
private int end;
|
||||
|
||||
|
||||
public Boundary(int start, int end) {
|
||||
|
||||
assert start <= end;
|
||||
this.start = start;
|
||||
this.end = end;
|
||||
}
|
||||
|
||||
|
||||
public int length() {
|
||||
|
||||
return end - start;
|
||||
}
|
||||
|
||||
|
||||
public int start() {
|
||||
|
||||
return start;
|
||||
}
|
||||
|
||||
|
||||
public int end() {
|
||||
|
||||
return end;
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(Boundary boundary) {
|
||||
|
||||
return start <= boundary.start() && boundary.end() <= end;
|
||||
}
|
||||
|
||||
|
||||
public boolean containedBy(Boundary boundary) {
|
||||
|
||||
return boundary.start() <= start && end <= boundary.end();
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
throw new UnsupportedOperationException("start > end");
|
||||
}
|
||||
return this.start <= start && end <= this.end;
|
||||
}
|
||||
|
||||
|
||||
public boolean containedBy(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
throw new UnsupportedOperationException("start > end");
|
||||
}
|
||||
return start <= this.start && this.end <= end;
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(int index) {
|
||||
|
||||
return start <= index && index < end;
|
||||
}
|
||||
|
||||
|
||||
public boolean intersects(Boundary boundary) {
|
||||
|
||||
return contains(boundary.start()) || contains(boundary.end());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return format("Boundary [%d|%d)", start, end);
|
||||
}
|
||||
|
||||
}
|
||||
+96
@@ -0,0 +1,96 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class DocumentGraph {
|
||||
|
||||
List<SectionNode> sections;
|
||||
List<PageNode> pages;
|
||||
TableOfContents tableOfContents;
|
||||
Integer numberOfPages;
|
||||
TextBlock text;
|
||||
|
||||
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamAtomicTextBlocksInOrder().collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
public Stream<AtomicTextBlock> streamAtomicTextBlocksInOrder() {
|
||||
|
||||
return Stream.concat(//
|
||||
streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock),//
|
||||
Stream.concat(//
|
||||
pages.stream().map(PageNode::getHeader),//
|
||||
pages.stream().map(PageNode::getFooter)));
|
||||
}
|
||||
|
||||
|
||||
public EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
|
||||
addEntityToGraphAndSetFields(entity);
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
public void addEntityToGraphAndSetFields(EntityNode entity) {
|
||||
|
||||
try {
|
||||
boolean inserted = streamAllNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
|
||||
} catch (NotFoundException e) {
|
||||
enrichEntity(entity, text);
|
||||
log.warn("Entity \"{}\" with {} is in between two main sections and will be removed!", entity.getValue(), entity.getBoundary());
|
||||
entity.removeFromGraph();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public Set<EntityNode> getEntities() {
|
||||
|
||||
return streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getEntities).flatMap(List::stream).collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
|
||||
private Stream<DocumentGraphNode> streamAllNodes() {
|
||||
|
||||
return tableOfContents.streamEntriesInOrder().map(TableOfContents.Entry::node);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return text.toString();
|
||||
}
|
||||
|
||||
}
|
||||
+113
@@ -0,0 +1,113 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import javax.management.openmbean.InvalidKeyException;
|
||||
|
||||
import com.google.common.hash.Hashing;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class TableOfContents {
|
||||
|
||||
List<Entry> entries;
|
||||
|
||||
|
||||
public TableOfContents() {
|
||||
|
||||
entries = new LinkedList<>();
|
||||
}
|
||||
|
||||
|
||||
public String createNewEntryAndReturnId(NodeType nodeType, String summary, DocumentGraphNode node) {
|
||||
|
||||
String id = String.format("%d", entries.size());
|
||||
entries.add(new Entry(nodeType, id, summary, new LinkedList<>(), node));
|
||||
return id;
|
||||
}
|
||||
|
||||
|
||||
public String createNewChildEntryAndReturnId(String parentId, NodeType nodeType, String summary, DocumentGraphNode node) {
|
||||
|
||||
Entry parent = getEntryById(parentId);
|
||||
String childId = parentId + String.format(".%d", parent.children().size());
|
||||
parent.children().add(new Entry(nodeType, childId, summary, new LinkedList<>(), node));
|
||||
return childId;
|
||||
}
|
||||
|
||||
|
||||
public Entry getEntryById(String parentId) {
|
||||
|
||||
List<Integer> ids = getIds(parentId);
|
||||
if (ids.size() < 1) {
|
||||
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", parentId));
|
||||
}
|
||||
Entry entry = entries.get(ids.get(0));
|
||||
for (int id : ids.subList(1, ids.size())) {
|
||||
entry = entry.children().get(id);
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return String.join("\n", streamEntriesInOrder().map(Entry::toString).toList());
|
||||
}
|
||||
|
||||
|
||||
public String toString(String id) {
|
||||
|
||||
return String.join("\n", streamSubEntriesInOrder(id).map(Entry::toString).toList());
|
||||
}
|
||||
|
||||
|
||||
public Stream<Entry> streamEntriesInOrder() {
|
||||
|
||||
return entries.stream().flatMap(TableOfContents::flatten);
|
||||
}
|
||||
|
||||
|
||||
public Stream<Entry> streamSubEntriesInOrder(String parentId) {
|
||||
|
||||
return Stream.of(getEntryById(parentId)).flatMap(TableOfContents::flatten);
|
||||
}
|
||||
|
||||
|
||||
private static List<Integer> getIds(String idsAsString) {
|
||||
|
||||
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
|
||||
}
|
||||
|
||||
|
||||
private static Stream<Entry> flatten(Entry entry) {
|
||||
|
||||
return Stream.concat(Stream.of(entry), entry.children().stream().flatMap(TableOfContents::flatten));
|
||||
}
|
||||
|
||||
|
||||
public record Entry(NodeType type, String id, String summary, List<Entry> children, DocumentGraphNode node) {
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return id + ": " + type + ".: " + summary;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
return Hashing.murmur3_32_fixed().hashString(type + id + summary + children.hashCode(), StandardCharsets.UTF_8).hashCode();
|
||||
}
|
||||
}
|
||||
}
|
||||
+219
@@ -0,0 +1,219 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
public interface DocumentGraphNode {
|
||||
|
||||
/**
|
||||
* Searches all Nodes located underneath this Node in the TableOfContents and concatenates their AtomicTextBlocks into a single TextBlockEntity.
|
||||
* So, for a Section all AtomicTextBlocks of Subsections, Paragraphs, and Tables are concatenated into a single TextBlockEntity
|
||||
*
|
||||
* @return TextBlock containing all AtomicTextBlocks that are located under this Node.
|
||||
*/
|
||||
TextBlock buildTextBlock();
|
||||
|
||||
|
||||
/**
|
||||
* Any Node maintains its own Set of Entities.
|
||||
* This Set contains all Entities, whose first index is located in any of the AtomicTextBlocks underneath this Node.
|
||||
*
|
||||
* @return Set of all Entities associated with this Node
|
||||
*/
|
||||
Set<EntityNode> getEntities();
|
||||
|
||||
|
||||
/**
|
||||
* Returns the PageNode associated with this Node.
|
||||
* If the node has more than one PageNode associated, it returns the PageNode with the lowest number.
|
||||
* For example a section might span multiple pages, it then returns the page where the section starts.
|
||||
*
|
||||
* @return PageNode representing the first page on which the Node is located in the document
|
||||
*/
|
||||
PageNode getPage();
|
||||
|
||||
|
||||
/**
|
||||
* Any Node except the First level of Sections, Header, Footer, and Pages have a direct Parent.
|
||||
* For example a Paragraph has a parent Section, a Table Cell has a parent Table, etc...
|
||||
* hasParent() may be used to check whether a parent is present.
|
||||
*
|
||||
* @return Node that represents the Parent or null, if no parent is present.
|
||||
*/
|
||||
DocumentGraphNode getParent();
|
||||
|
||||
|
||||
Stream<DocumentGraphNode> streamAllSubNodes();
|
||||
|
||||
|
||||
/**
|
||||
* Each AtomicTextBlock has a number assigned per page, this returns the number of the first AtomicTextBlock underneath this node
|
||||
*
|
||||
* @return Integer representing the number on the page
|
||||
*/
|
||||
Integer getNumberOnPage();
|
||||
|
||||
|
||||
/**
|
||||
*
|
||||
* @return the fist headline whent traversing the tree upwards
|
||||
*/
|
||||
default CharSequence getHeadline() {
|
||||
return getParent().getHeadline();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* By default, a parent is always present, this needs to be overwritten for Headers, Footers, Pages, and Sections.
|
||||
*
|
||||
* @return boolean, indicating whether a parent is present.
|
||||
*/
|
||||
default boolean hasParent() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* by default a Node does not have direct access to an AtomicTextBlock
|
||||
*
|
||||
* @return boolean, indicating if a Node has direct access to an AtomicTextBlock
|
||||
*/
|
||||
default boolean isTerminal() {
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* by default a Node does not have direct access to an AtomicTextBlock, this method throws a UnsupportedOperationException if not overridden.
|
||||
*
|
||||
* @return AtomicTextBlock
|
||||
*/
|
||||
default AtomicTextBlock getAtomicTextBlock() {
|
||||
|
||||
throw new UnsupportedOperationException("Only terminal Nodes have access to AtomicTextBlocks!");
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* creates an EntityNode with only initial values set, inserts it into the subgraph contained by this node and sets the inferrable fields.
|
||||
* Throws NotFoundException and removes the entity if the provided boundary could not be found in the subgraph.
|
||||
*
|
||||
* @param boundary start and end indices in String coordinates of the entity to be created
|
||||
* @param type type of the entity to be created
|
||||
* @param entityType entityType of the entity to be created
|
||||
* @return the newly created and inserted EntityNode with all fields set.
|
||||
*/
|
||||
default EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
|
||||
addEntityToNodeAndSetFields(entity);
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* searches for the first terminal Node containing the start index of the entity to be inserted.
|
||||
* Catches NotFoundException to remove the EntityNode from the graph, then rethrows it
|
||||
*
|
||||
* @param entity newly created EntityNode with only initial values set
|
||||
*/
|
||||
default void addEntityToNodeAndSetFields(EntityNode entity) {
|
||||
|
||||
try {
|
||||
streamAllSubNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
|
||||
} catch (NotFoundException e) {
|
||||
entity.removeFromGraph();
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* If this Node's AtomicTextBlock contains the start index of the entity, the entity's position is read from the AtomicTextBlock.
|
||||
* If the position can not be fully read from the AtomicTextBlock, it recursively looks in the TextBlocks of the parents until all positions are found.
|
||||
* Further, the function throws NotFoundException if no parent contains all positions.
|
||||
* This occurs, when the Entity is in between Nodes that do not share a parent, e.g. main sections.
|
||||
* Finally, the function adds the Entity to its own list of Entities and to every parents' list recursively.
|
||||
*
|
||||
* @param entity The entity to be added to the graph
|
||||
* @return true, if the entity has been added successfully.
|
||||
* false, if the entity's start index is not contained or the Node doesn't have an AtomicTextBlock
|
||||
*/
|
||||
default boolean addEntityAndSetFieldsIfStartIndexContained(EntityNode entity) {
|
||||
|
||||
if (!isTerminal()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
AtomicTextBlock atomicTextBlock = getAtomicTextBlock();
|
||||
if (atomicTextBlock.containsIndex(entity.getBoundary().start())) {
|
||||
|
||||
entity.addContainingNode(this);
|
||||
|
||||
getEntities().add(entity);
|
||||
|
||||
addEntityToPage(entity);
|
||||
addEntityToParents(entity);
|
||||
setFields(entity, atomicTextBlock);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
private void addEntityToPage(EntityNode entity) {
|
||||
|
||||
getPage().getEntities().add(entity);
|
||||
entity.setPage(getPage());
|
||||
}
|
||||
|
||||
|
||||
private void setFields(EntityNode entity, AtomicTextBlock atomicTextBlock) {
|
||||
|
||||
if (atomicTextBlock.containsBoundary(entity.getBoundary())) {
|
||||
enrichEntity(entity, atomicTextBlock);
|
||||
} else {
|
||||
this.setFieldsFromParents(this, entity);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addEntityToParents(EntityNode entity) {
|
||||
|
||||
DocumentGraphNode node = this;
|
||||
while (node.hasParent()) {
|
||||
node = node.getParent();
|
||||
node.getEntities().add(entity);
|
||||
entity.addContainingNode(node);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void setFieldsFromParents(DocumentGraphNode node, EntityNode entity) {
|
||||
|
||||
if (node.hasParent()) {
|
||||
DocumentGraphNode parent = node.getParent();
|
||||
TextBlock textBlock = parent.buildTextBlock();
|
||||
if (textBlock.containsBoundary(entity.getBoundary())) {
|
||||
enrichEntity(entity, textBlock);
|
||||
return;
|
||||
} else {
|
||||
setFieldsFromParents(parent, entity);
|
||||
}
|
||||
}
|
||||
throw new NotFoundException(format("Position could not be found for Entity %s", entity.toString()));
|
||||
}
|
||||
|
||||
}
|
||||
+103
@@ -0,0 +1,103 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import com.google.common.hash.Hashing;
|
||||
import com.iqser.red.service.redaction.v1.model.Engine;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class EntityNode {
|
||||
|
||||
public static EntityNode initialEntityNode(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
return EntityNode.builder().type(type).entityType(entityType).boundary(boundary).build();
|
||||
}
|
||||
|
||||
|
||||
// initial values
|
||||
Boundary boundary;
|
||||
String type;
|
||||
EntityType entityType;
|
||||
|
||||
@Builder.Default
|
||||
boolean redaction = false;
|
||||
@Builder.Default
|
||||
boolean falsePositive = false;
|
||||
@Builder.Default
|
||||
boolean removed = false;
|
||||
@Builder.Default
|
||||
boolean ignored = false;
|
||||
@Builder.Default
|
||||
boolean resized = false;
|
||||
@Builder.Default
|
||||
boolean skipRemoveEntitiesContainedInLarger = false;
|
||||
@Builder.Default
|
||||
boolean isDictionaryEntry = false;
|
||||
@Builder.Default
|
||||
Set<Engine> engines = new HashSet<>();
|
||||
@Builder.Default
|
||||
Set<Entity> references = new HashSet<>();
|
||||
@Builder.Default
|
||||
int matchedRule = -1;
|
||||
@Builder.Default
|
||||
String redactionReason = "";
|
||||
@Builder.Default
|
||||
String legalBasis = "";
|
||||
|
||||
// inferrable from graph
|
||||
String value;
|
||||
CharSequence textBefore;
|
||||
CharSequence textAfter;
|
||||
PageNode page;
|
||||
List<Rectangle2D> positions;
|
||||
@Builder.Default
|
||||
Set<DocumentGraphNode> containingNodes = new HashSet<>();
|
||||
|
||||
|
||||
public void addContainingNode(DocumentGraphNode containingNode) {
|
||||
|
||||
containingNodes.add(containingNode);
|
||||
}
|
||||
|
||||
|
||||
public void removeFromGraph() {
|
||||
|
||||
getContainingNodes().forEach(node -> node.getEntities().remove(this));
|
||||
getPage().getEntities().remove(this);
|
||||
setRemoved(true);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
|
||||
var sb = new StringBuilder();
|
||||
sb.append(value);
|
||||
sb.append(boundary.start());
|
||||
sb.append(page.getNumber());
|
||||
positions.forEach(r -> {
|
||||
sb.append(r.getMinX());
|
||||
sb.append(r.getMinY());
|
||||
sb.append(r.getWidth());
|
||||
sb.append(r.getHeight());
|
||||
});
|
||||
return Hashing.murmur3_128().hashString(sb.toString(), StandardCharsets.UTF_8).hashCode();
|
||||
}
|
||||
|
||||
}
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
public enum NodeType {
|
||||
SECTION {
|
||||
public String toString() {
|
||||
|
||||
return "Section";
|
||||
}
|
||||
},
|
||||
PARAGRAPH {
|
||||
public String toString() {
|
||||
|
||||
return "Paragraph";
|
||||
}
|
||||
},
|
||||
TABLE {
|
||||
public String toString() {
|
||||
|
||||
return "Table";
|
||||
}
|
||||
},
|
||||
TABLE_CELL {
|
||||
public String toString() {
|
||||
|
||||
return "Cell";
|
||||
}
|
||||
}
|
||||
}
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class PageNode implements DocumentGraphNode{
|
||||
|
||||
Integer number;
|
||||
Integer height;
|
||||
Integer width;
|
||||
List<DocumentGraphNode> mainBody;
|
||||
AtomicTextBlock header;
|
||||
AtomicTextBlock footer;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return mainBody.stream().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean hasParent() {
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return mainBody.stream();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Integer getNumberOnPage() {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return header.getSearchText() + buildTextBlock().toString() + footer.getSearchText();
|
||||
}
|
||||
|
||||
}
|
||||
+70
@@ -0,0 +1,70 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class ParagraphNode implements DocumentGraphNode {
|
||||
|
||||
String tocId;
|
||||
Integer numberOnPage;
|
||||
Integer numberInSection;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
SectionNode parentSection;
|
||||
@EqualsAndHashCode.Exclude
|
||||
PageNode page;
|
||||
AtomicTextBlock atomicTextBlock;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock buildTextBlock() {
|
||||
|
||||
return atomicTextBlock;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentSection;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return tocId + ": " + atomicTextBlock.toString();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return Stream.of(this);
|
||||
}
|
||||
|
||||
}
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class SectionNode implements DocumentGraphNode {
|
||||
|
||||
String tocId;
|
||||
Integer numberOnPage;
|
||||
@EqualsAndHashCode.Exclude
|
||||
TableOfContents tableOfContents;
|
||||
@EqualsAndHashCode.Exclude
|
||||
DocumentGraphNode parentSection;
|
||||
|
||||
|
||||
AtomicTextBlock headline;
|
||||
|
||||
List<SectionNode> subSections;
|
||||
List<ParagraphNode> paragraphs;
|
||||
List<TableNode> tables;
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<PageNode> pages;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return tocId + ": " + headline.toString();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamAllSubNodes().map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
if (hasParent()) {
|
||||
return parentSection;
|
||||
} else {
|
||||
throw new UnsupportedOperationException("This section has no parent Section!");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return tableOfContents.streamSubEntriesInOrder(tocId).map(TableOfContents.Entry::node);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean hasParent() {
|
||||
|
||||
return parentSection != null;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock getAtomicTextBlock() {
|
||||
|
||||
return headline;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return pages.get(0);
|
||||
}
|
||||
|
||||
}
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class TableCellNode implements DocumentGraphNode {
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
TableNode parentTable;
|
||||
Integer numberOnPage;
|
||||
AtomicTextBlock atomicTextBlock;
|
||||
PageNode page;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<EntityNode> entities = new LinkedList<>();
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock buildTextBlock() {
|
||||
|
||||
return atomicTextBlock;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentTable;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return Stream.of(this);
|
||||
}
|
||||
|
||||
}
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class TableNode implements DocumentGraphNode {
|
||||
|
||||
Integer id;
|
||||
String tocId;
|
||||
Integer numberOfRows;
|
||||
Integer numberOfCols;
|
||||
Integer numberOnPage;
|
||||
List<TableCellNode> tableHeaders;
|
||||
List<List<TableCellNode>> tableCells;
|
||||
TableOfContents tableOfContents;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
SectionNode parentSection;
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<PageNode> pages;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<EntityNode> entities = new LinkedList<>();
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableCells() {
|
||||
|
||||
return tableCells.stream().flatMap(List::stream);
|
||||
}
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableRow(int row) {
|
||||
|
||||
return tableCells.get(row).stream();
|
||||
}
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableCol(int col) {
|
||||
|
||||
return tableCells.stream().map(row -> row.get(col));
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamTableCells().map(TableCellNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return streamTableCells().map(Function.identity());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentSection;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return pages.get(0);
|
||||
}
|
||||
|
||||
}
|
||||
+126
@@ -0,0 +1,126 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class AtomicTextBlock implements TextBlock {
|
||||
|
||||
Long id;
|
||||
|
||||
//string coordinates
|
||||
Boundary boundary;
|
||||
String searchText;
|
||||
List<Integer> lineBreaks;
|
||||
|
||||
//position coordinates
|
||||
List<Integer> stringIdxToPositionIdx;
|
||||
List<Rectangle2D> positions;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
DocumentGraphNode parent;
|
||||
|
||||
|
||||
public int indexOf(String searchTerm) {
|
||||
|
||||
int pos = searchText.indexOf(searchTerm);
|
||||
return pos == -1 ? -1 : pos + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int numberOfLines() {
|
||||
|
||||
return lineBreaks.size();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public List<AtomicTextBlock> getAtomicTextBlocks() {
|
||||
|
||||
return List.of(this);
|
||||
}
|
||||
|
||||
|
||||
public int getNextLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak > fromIndex) //
|
||||
.findFirst() //
|
||||
.orElse(searchText.length()) + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int getPreviousLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak <= fromIndex)//
|
||||
.reduce((a, b) -> b)//
|
||||
.orElse(0) + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public Rectangle2D getPosition(int stringIdx) {
|
||||
|
||||
return positions.get(stringIdxToPositionIdx.get(stringIdx - boundary.start()));
|
||||
}
|
||||
|
||||
|
||||
public List<Rectangle2D> getPositions(Boundary boundary) {
|
||||
|
||||
if (!containsBoundary(boundary)) {
|
||||
throw new IndexOutOfBoundsException(format("%s is out of bounds for %s",
|
||||
boundary,
|
||||
this.boundary));
|
||||
}
|
||||
|
||||
if (boundary.end() == this.boundary.end()) {
|
||||
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), positions.size());
|
||||
}
|
||||
|
||||
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), stringIdxToPositionIdx.get(boundary.end() - this.boundary.start()));
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int length() {
|
||||
|
||||
return searchText.length();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public char charAt(int index) {
|
||||
|
||||
return searchText.charAt(index - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public CharSequence subSequence(int start, int end) {
|
||||
|
||||
return searchText.substring(start - boundary.start(), end - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return searchText;
|
||||
}
|
||||
|
||||
}
|
||||
+155
@@ -0,0 +1,155 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class ConcatenatedTextBlock implements TextBlock, Supplier<ConcatenatedTextBlock> {
|
||||
|
||||
List<AtomicTextBlock> atomicTextBlocks;
|
||||
StringBuilder searchText;
|
||||
Boundary boundary;
|
||||
|
||||
|
||||
public ConcatenatedTextBlock(List<AtomicTextBlock> atomicTextBlocks) {
|
||||
|
||||
this.atomicTextBlocks = new LinkedList<>();
|
||||
this.searchText = new StringBuilder();
|
||||
if (atomicTextBlocks.isEmpty()) {
|
||||
boundary = new Boundary(-1, -1);
|
||||
return;
|
||||
}
|
||||
var firstTextBlock = atomicTextBlocks.get(0);
|
||||
this.atomicTextBlocks.add(firstTextBlock);
|
||||
this.searchText.append(firstTextBlock.getSearchText());
|
||||
boundary = new Boundary(firstTextBlock.getBoundary().start(), firstTextBlock.getBoundary().end());
|
||||
|
||||
atomicTextBlocks.subList(1, atomicTextBlocks.size()).forEach(this::concat);
|
||||
}
|
||||
|
||||
|
||||
public ConcatenatedTextBlock(AtomicTextBlock atomicTextBlocks) {
|
||||
|
||||
new ConcatenatedTextBlock(List.of(atomicTextBlocks));
|
||||
}
|
||||
|
||||
|
||||
public ConcatenatedTextBlock concat(TextBlock textBlock) {
|
||||
|
||||
if (this.atomicTextBlocks.isEmpty()) {
|
||||
boundary.setStart(textBlock.getBoundary().start());
|
||||
boundary.setEnd(textBlock.getBoundary().end());
|
||||
} else if (boundary.end() != textBlock.getBoundary().start()) {
|
||||
throw new UnsupportedOperationException(format("Can only concat consecutive TextBlocks, trying to concat %s to %s", textBlock.getBoundary(), boundary));
|
||||
}
|
||||
this.searchText.append(textBlock.getSearchText());
|
||||
this.atomicTextBlocks.addAll(textBlock.getAtomicTextBlocks());
|
||||
boundary.setEnd(textBlock.getBoundary().end());
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
public int indexOf(String searchTerm) {
|
||||
|
||||
int pos = this.searchText.indexOf(searchTerm);
|
||||
return pos == -1 ? -1 : pos + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int numberOfLines() {
|
||||
|
||||
return atomicTextBlocks.stream().map(AtomicTextBlock::getLineBreaks).mapToInt(List::size).sum();
|
||||
}
|
||||
|
||||
|
||||
public int getNextLinebreak(int fromIndex) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(fromIndex).getNextLinebreak(fromIndex);
|
||||
}
|
||||
|
||||
|
||||
public int getPreviousLinebreak(int fromIndex) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(fromIndex).getPreviousLinebreak(fromIndex);
|
||||
}
|
||||
|
||||
|
||||
public Rectangle2D getPosition(int stringIdx) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(stringIdx).getPosition(stringIdx);
|
||||
}
|
||||
|
||||
|
||||
public List<Rectangle2D> getPositions(Boundary boundary) {
|
||||
|
||||
List<AtomicTextBlock> textBlocks = getAllAtomicTextBlocksPartiallyInStringIdxRange(boundary);
|
||||
|
||||
if (textBlocks.size() == 1) {
|
||||
return textBlocks.get(0).getPositions(boundary);
|
||||
}
|
||||
|
||||
AtomicTextBlock firstTextBlock = textBlocks.get(0);
|
||||
List<Rectangle2D> positions = new LinkedList<>(firstTextBlock.getPositions(new Boundary(boundary.start(), firstTextBlock.getBoundary().end())));
|
||||
|
||||
for (AtomicTextBlock textBlock : textBlocks.subList(1, textBlocks.size() - 1)) {
|
||||
positions.addAll(textBlock.getPositions());
|
||||
}
|
||||
|
||||
var lastTextBlock = textBlocks.get(textBlocks.size() - 1);
|
||||
positions.addAll(lastTextBlock.getPositions(new Boundary(lastTextBlock.getBoundary().start(), boundary.end())));
|
||||
|
||||
return positions;
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock getAtomicTextBlockByStringIndex(int stringIdx) {
|
||||
|
||||
return atomicTextBlocks.stream().filter(textBlock -> (textBlock.getBoundary().end()) > stringIdx).findFirst().orElseThrow(IndexOutOfBoundsException::new);
|
||||
}
|
||||
|
||||
|
||||
private List<AtomicTextBlock> getAllAtomicTextBlocksPartiallyInStringIdxRange(Boundary boundary) {
|
||||
|
||||
return atomicTextBlocks.stream().filter(tb -> tb.getBoundary().intersects(boundary)).toList();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int length() {
|
||||
|
||||
return this.searchText.length();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public char charAt(int index) {
|
||||
|
||||
return searchText.charAt(index - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public CharSequence subSequence(int start, int end) {
|
||||
|
||||
return searchText.subSequence(start - boundary.start(), end - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock get() {
|
||||
|
||||
return this;
|
||||
}
|
||||
|
||||
}
|
||||
+65
@@ -0,0 +1,65 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
public interface TextBlock extends CharSequence {
|
||||
|
||||
CharSequence getSearchText();
|
||||
|
||||
|
||||
List<AtomicTextBlock> getAtomicTextBlocks();
|
||||
|
||||
|
||||
Boundary getBoundary();
|
||||
|
||||
|
||||
int getNextLinebreak(int fromIndex);
|
||||
|
||||
|
||||
int getPreviousLinebreak(int fromIndex);
|
||||
|
||||
|
||||
Rectangle2D getPosition(int stringIdx);
|
||||
|
||||
|
||||
List<Rectangle2D> getPositions(Boundary range);
|
||||
|
||||
|
||||
int numberOfLines();
|
||||
|
||||
|
||||
int indexOf(String searchTerm);
|
||||
|
||||
|
||||
default CharSequence getFirstLine() {
|
||||
|
||||
return subSequence(getBoundary().start(), getNextLinebreak(getBoundary().start()));
|
||||
}
|
||||
|
||||
|
||||
default boolean containsBoundary(Boundary boundary) {
|
||||
|
||||
if (boundary.end() < boundary.start()) {
|
||||
throw new IllegalArgumentException(format("Invalid %s, StartIndex must be smaller than EndIndex", boundary));
|
||||
}
|
||||
return getBoundary().contains(boundary);
|
||||
}
|
||||
|
||||
|
||||
default boolean containsIndex(int stringIndex) {
|
||||
|
||||
return getBoundary().contains(stringIndex);
|
||||
}
|
||||
|
||||
|
||||
default CharSequence subSequence(Boundary boundary) {
|
||||
|
||||
return subSequence(boundary.start(), boundary.end());
|
||||
}
|
||||
|
||||
}
|
||||
+51
@@ -0,0 +1,51 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.Set;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.BinaryOperator;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collector;
|
||||
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@NoArgsConstructor
|
||||
public class TextBlockCollector implements Collector<AtomicTextBlock, ConcatenatedTextBlock, ConcatenatedTextBlock> {
|
||||
|
||||
|
||||
@Override
|
||||
public Supplier<ConcatenatedTextBlock> supplier() {
|
||||
|
||||
return new ConcatenatedTextBlock(Collections.emptyList());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public BiConsumer<ConcatenatedTextBlock, AtomicTextBlock> accumulator() {
|
||||
|
||||
return ConcatenatedTextBlock::concat;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public BinaryOperator<ConcatenatedTextBlock> combiner() {
|
||||
|
||||
return ConcatenatedTextBlock::concat;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Function<ConcatenatedTextBlock, ConcatenatedTextBlock> finisher() {
|
||||
|
||||
return Function.identity();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Set<Characteristics> characteristics() {
|
||||
|
||||
return Set.of(Characteristics.IDENTITY_FINISH, Characteristics.CONCURRENT);
|
||||
}
|
||||
|
||||
}
|
||||
+98
@@ -0,0 +1,98 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
@Service
|
||||
public class DocumentDataMapper {
|
||||
|
||||
public DocumentData toDocumentData(DocumentGraph documentGraph) {
|
||||
|
||||
List<AtomicTextBlockData> atomicTextBlockData = documentGraph.streamAtomicTextBlocksInOrder().map(this::toAtomicTextBlockData).toList();
|
||||
List<PageData> pageData = documentGraph.getPages().stream().map(this::toPageData).toList();
|
||||
TableOfContentsData tableOfContentsData = toTableOfContentsData(documentGraph.getTableOfContents());
|
||||
return DocumentData.builder().atomicTextBlocks(atomicTextBlockData).pages(pageData).tableOfContents(tableOfContentsData).build();
|
||||
}
|
||||
|
||||
|
||||
private TableOfContentsData toTableOfContentsData(TableOfContents tableOfContents) {
|
||||
|
||||
return new TableOfContentsData(tableOfContents.getEntries().stream().map(this::toEntryData).toList());
|
||||
}
|
||||
|
||||
|
||||
private TableOfContentsData.EntryData toEntryData(TableOfContents.Entry entry) {
|
||||
|
||||
return TableOfContentsData.EntryData.builder()
|
||||
.tocId(entry.id())
|
||||
.subEntries(entry.children().stream().map(this::toEntryData).toList())
|
||||
.type(entry.type())
|
||||
.atomicTextBlock(entry.node().isTerminal() ? entry.node().getAtomicTextBlock().getId() : -1L)
|
||||
.page(Long.valueOf(entry.node().getPage().getNumber()))
|
||||
.numberOnPage(entry.node().getNumberOnPage())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private PageData toPageData(PageNode pageNode) {
|
||||
|
||||
return PageData.builder()
|
||||
.height(pageNode.getHeight())
|
||||
.width(pageNode.getWidth())
|
||||
.number(pageNode.getNumber())
|
||||
.footer(pageNode.getFooter().getId())
|
||||
.header(pageNode.getHeader().getId())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlockData toAtomicTextBlockData(AtomicTextBlock atomicTextBlock) {
|
||||
|
||||
return AtomicTextBlockData.builder()
|
||||
.id(atomicTextBlock.getId())
|
||||
.searchText(atomicTextBlock.getSearchText())
|
||||
.start(atomicTextBlock.getBoundary().start())
|
||||
.end(atomicTextBlock.getBoundary().end())
|
||||
.lineBreaks(toPrimitiveIntArray(atomicTextBlock.getLineBreaks()))
|
||||
.stringIdxToPositionIdx(toPrimitiveIntArray(atomicTextBlock.getStringIdxToPositionIdx()))
|
||||
.positions(toPrimitiveFloatMatrix(atomicTextBlock.getPositions()))
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private float[][] toPrimitiveFloatMatrix(List<Rectangle2D> positions) {
|
||||
|
||||
float[][] positionMatrix = new float[positions.size()][];
|
||||
for (int i = 0; i < positions.size(); i++) {
|
||||
float[] singlePositions = new float[4];
|
||||
singlePositions[0] = (float) positions.get(i).getMinX();
|
||||
singlePositions[1] = (float) positions.get(i).getMinY();
|
||||
singlePositions[2] = (float) positions.get(i).getWidth();
|
||||
singlePositions[3] = (float) positions.get(i).getHeight();
|
||||
positionMatrix[i] = singlePositions;
|
||||
}
|
||||
return positionMatrix;
|
||||
}
|
||||
|
||||
|
||||
private int[] toPrimitiveIntArray(List<Integer> list) {
|
||||
|
||||
int[] array = new int[list.size()];
|
||||
for (int i = 0; i < list.size(); i++) {
|
||||
array[i] = list.get(i);
|
||||
}
|
||||
return array;
|
||||
}
|
||||
|
||||
}
|
||||
+304
@@ -0,0 +1,304 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Footer;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Header;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Section;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.RedRectangle2D;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchTextWithTextPositionModel;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SearchTextWithTextPositionFactory;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class DocumentGraphFactory {
|
||||
|
||||
private final SearchTextWithTextPositionFactory searchTextWithTextPositionFactory;
|
||||
|
||||
|
||||
public DocumentGraph buildDocumentGraph(Document document) {
|
||||
|
||||
Context context = new Context(new TableOfContents(), new LinkedList<>(), new LinkedList<>(), new AtomicInteger(0), new AtomicLong(0));
|
||||
|
||||
context.pages.addAll(document.getPages().stream().map(this::buildPage).toList());
|
||||
|
||||
// is tracked by Table of Contents
|
||||
addSections(document, context);
|
||||
|
||||
// not tracked by Table of Contents
|
||||
addHeaderAndFooterToEachPage(document, context);
|
||||
DocumentGraph documentGraph = DocumentGraph.builder()
|
||||
.numberOfPages(context.pages.size())
|
||||
.pages(context.pages)
|
||||
.sections(context.sections)
|
||||
.tableOfContents(context.tableOfContents)
|
||||
.build();
|
||||
documentGraph.setText(documentGraph.buildTextBlock());
|
||||
return documentGraph;
|
||||
}
|
||||
|
||||
|
||||
private void addSections(Document document, Context context) {
|
||||
|
||||
for (var section : document.getSections()) {
|
||||
addSection(section, context);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addSection(Section section, Context context) {
|
||||
|
||||
SectionNode sectionEntity = SectionNode.builder()
|
||||
.entities(new LinkedList<>())
|
||||
.pages(new LinkedList<>())
|
||||
.paragraphs(new LinkedList<>())
|
||||
.tables(new LinkedList<>())
|
||||
.subSections(new LinkedList<>())
|
||||
.tableOfContents(context.tableOfContents())
|
||||
.build();
|
||||
|
||||
context.sections().add(sectionEntity);
|
||||
List<AbstractTextContainer> pageBlocks = new ArrayList<>(section.getPageBlocks());
|
||||
PageNode page = getPage(section.getPageBlocks().get(0).getPage(), context);
|
||||
sectionEntity.getPages().add(page);
|
||||
page.getMainBody().add(sectionEntity);
|
||||
if (pageBlocks.get(0) instanceof TextBlock) {
|
||||
sectionEntity.setHeadline(buildAtomicTextBlock(((TextBlock) pageBlocks.get(0)).getSequences(), sectionEntity, context));
|
||||
sectionEntity.setNumberOnPage(((TextBlock) pageBlocks.get(0)).getIndexOnPage());
|
||||
pageBlocks.remove(0);
|
||||
} else {
|
||||
sectionEntity.setNumberOnPage(1);
|
||||
sectionEntity.setHeadline(emptyTextBlock(sectionEntity, context));
|
||||
}
|
||||
|
||||
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(sectionEntity.getHeadline()), sectionEntity);
|
||||
sectionEntity.setTocId(sectionId);
|
||||
|
||||
int paragraphIdx = 0;
|
||||
int tableIdx = 0;
|
||||
for (AbstractTextContainer abstractTextContainer : pageBlocks) {
|
||||
if (abstractTextContainer instanceof TextBlock) {
|
||||
addParagraph(sectionEntity, (TextBlock) abstractTextContainer, paragraphIdx, context);
|
||||
paragraphIdx++;
|
||||
} else if (abstractTextContainer instanceof Table) {
|
||||
//addTable(sectionEntity, (Table) abstractTextContainer, tableIdx, context);
|
||||
tableIdx++;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addTable(SectionNode sectionEntity, Table table, int tableIdx, Context context) {
|
||||
|
||||
PageNode page = getPage(table.getPage(), context);
|
||||
TableNode tableEntity = TableNode.builder().id(tableIdx).tableOfContents(context.tableOfContents()).pages(new LinkedList<>()).parentSection(sectionEntity).build();
|
||||
sectionEntity.getTables().add(tableEntity);
|
||||
|
||||
if (!page.getMainBody().contains(sectionEntity)) {
|
||||
sectionEntity.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(tableEntity);
|
||||
|
||||
}
|
||||
|
||||
|
||||
private void addParagraph(SectionNode sectionEntity, TextBlock originalTextBlock, int paragraphIdx, Context context) {
|
||||
|
||||
PageNode page = getPage(originalTextBlock.getPage(), context);
|
||||
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(originalTextBlock.getIndexOnPage()).page(page).parentSection(sectionEntity).build();
|
||||
sectionEntity.getParagraphs().add(paragraph);
|
||||
|
||||
if (!page.getMainBody().contains(sectionEntity)) {
|
||||
sectionEntity.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(paragraph);
|
||||
|
||||
var textBlock = buildAtomicTextBlock(originalTextBlock.getSequences(), paragraph, context);
|
||||
paragraph.setAtomicTextBlock(textBlock);
|
||||
|
||||
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(sectionEntity.getTocId(), NodeType.PARAGRAPH, buildSummary(textBlock), paragraph);
|
||||
paragraph.setTocId(tocId);
|
||||
}
|
||||
|
||||
|
||||
private void addHeaderAndFooterToEachPage(Document document, Context context) {
|
||||
|
||||
Map<Integer, List<TextBlock>> headers = document.getHeaders()
|
||||
.stream()
|
||||
.map(Header::getTextBlocks)
|
||||
.flatMap(List::stream)
|
||||
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
|
||||
|
||||
Map<Integer, List<TextBlock>> footers = document.getFooters()
|
||||
.stream()
|
||||
.map(Footer::getTextBlocks)
|
||||
.flatMap(List::stream)
|
||||
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
|
||||
|
||||
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
|
||||
if (headers.containsKey(pageIndex)) {
|
||||
addHeader(headers.get(pageIndex), context);
|
||||
} else {
|
||||
addEmptyHeader(pageIndex, context);
|
||||
}
|
||||
}
|
||||
|
||||
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
|
||||
if (footers.containsKey(pageIndex)) {
|
||||
addFooter(footers.get(pageIndex), context);
|
||||
} else {
|
||||
addEmptyFooter(pageIndex, context);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addFooter(List<TextBlock> textBlocks, Context context) {
|
||||
|
||||
PageNode page = getPage(textBlocks.get(0).getPage(), context);
|
||||
AtomicTextBlock footer = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
|
||||
page.setFooter(footer);
|
||||
}
|
||||
|
||||
|
||||
public void addHeader(List<TextBlock> textBlocks, Context context) {
|
||||
|
||||
PageNode page = getPage(textBlocks.get(0).getPage(), context);
|
||||
AtomicTextBlock header = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
|
||||
page.setHeader(header);
|
||||
}
|
||||
|
||||
|
||||
private void addEmptyFooter(int pageIndex, Context context) {
|
||||
|
||||
PageNode page = getPage(pageIndex, context);
|
||||
page.setFooter(emptyTextBlock(page, context));
|
||||
}
|
||||
|
||||
|
||||
private void addEmptyHeader(int pageIndex, Context context) {
|
||||
|
||||
PageNode page = getPage(pageIndex, context);
|
||||
page.setHeader(emptyTextBlock(page, context));
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock emptyTextBlock(DocumentGraphNode parent, Context context) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(context.textBlockIdx.getAndIncrement())
|
||||
.boundary(new Boundary(context.stringOffset.get(), context.stringOffset.get()))
|
||||
.searchText("")
|
||||
.lineBreaks(Collections.emptyList())
|
||||
.stringIdxToPositionIdx(Collections.emptyList())
|
||||
.positions(Collections.emptyList())
|
||||
.parent(parent)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private static String buildSummary(AtomicTextBlock textBlock) {
|
||||
|
||||
if (textBlock == null) {
|
||||
return " probably a table";
|
||||
}
|
||||
|
||||
String[] words = textBlock.getFirstLine().toString().split(" ");
|
||||
int bound = Math.min(words.length, 4);
|
||||
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
|
||||
|
||||
return String.join(" ", list);
|
||||
}
|
||||
|
||||
|
||||
private PageNode buildPage(Page p) {
|
||||
|
||||
return PageNode.builder().height((int) p.getPageHeight()).width((int) p.getPageWidth()).number(p.getPageNumber()).mainBody(new LinkedList<>()).build();
|
||||
}
|
||||
|
||||
|
||||
private List<TextPositionSequence> mergeAndSortTextPositionSequences(List<TextBlock> textBlocks) {
|
||||
|
||||
Comparator<TextPositionSequence> sortByX = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[0] - sequence2.getTextPositions()
|
||||
.get(0)
|
||||
.getPosition()[0]);
|
||||
Comparator<TextPositionSequence> sortByY = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[1] - sequence2.getTextPositions()
|
||||
.get(0)
|
||||
.getPosition()[1]);
|
||||
|
||||
return textBlocks.stream().map(TextBlock::getSequences).flatMap(List::stream).sorted(sortByX.thenComparing(sortByY)).toList();
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock buildAtomicTextBlock(List<TextPositionSequence> sequences, DocumentGraphNode parent, Context context) {
|
||||
|
||||
SearchTextWithTextPositionModel searchTextWithTextPositionModel = searchTextWithTextPositionFactory.buildSearchTextToTextPositionModel(sequences);
|
||||
int offset = context.stringOffset().getAndAdd(searchTextWithTextPositionModel.getSearchText().length());
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(context.textBlockIdx.getAndIncrement())
|
||||
.parent(parent)
|
||||
.searchText(searchTextWithTextPositionModel.getSearchText())
|
||||
.lineBreaks(searchTextWithTextPositionModel.getLineBreaks())
|
||||
.positions(toRectangle2D(searchTextWithTextPositionModel.getPositions()))
|
||||
.stringIdxToPositionIdx(searchTextWithTextPositionModel.getStringCoordsToPositionCoords())
|
||||
.boundary(new Boundary(offset, offset + searchTextWithTextPositionModel.getSearchText().length()))
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private List<Rectangle2D> toRectangle2D(List<RedRectangle2D> positions) {
|
||||
|
||||
return positions.stream().map(r -> (Rectangle2D) new Rectangle2D.Double(r.getX(), r.getY(), r.getWidth(), r.getHeight())).toList();
|
||||
}
|
||||
|
||||
|
||||
private PageNode getPage(int pageIndex, Context context) {
|
||||
|
||||
return context.pages.stream()
|
||||
.filter(page -> page.getNumber() == pageIndex)
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
|
||||
}
|
||||
|
||||
|
||||
record Context(
|
||||
TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, AtomicInteger stringOffset, AtomicLong textBlockIdx) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+171
@@ -0,0 +1,171 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
import static java.lang.Math.toIntExact;
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.commons.lang3.NotImplementedException;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.google.common.primitives.Ints;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
|
||||
@Service
|
||||
public class DocumentGraphMapper {
|
||||
|
||||
public DocumentGraph toDocumentGraph(DocumentData documentData) {
|
||||
|
||||
Context context = new Context(documentData, new TableOfContents(), new LinkedList<>(), new LinkedList<>(), documentData.getAtomicTextBlocks());
|
||||
|
||||
context.pages.addAll(documentData.getPages().stream().map(pageData -> buildPage(pageData, context)).toList());
|
||||
buildNodesFromTableOfContents("", context);
|
||||
DocumentGraph documentGraph= DocumentGraph.builder()
|
||||
.numberOfPages(documentData.getPages().size())
|
||||
.pages(context.pages)
|
||||
.sections(context.sections)
|
||||
.tableOfContents(context.tableOfContents)
|
||||
.build();
|
||||
documentGraph.setText(documentGraph.buildTextBlock());
|
||||
return documentGraph;
|
||||
}
|
||||
|
||||
|
||||
private void buildNodesFromTableOfContents(String currentTocId, Context context) {
|
||||
List <TableOfContentsData.EntryData> entries;
|
||||
if(currentTocId.equals("")) {
|
||||
entries = context.documentData().getTableOfContents().getEntries();
|
||||
} else {
|
||||
entries = context.documentData().getTableOfContents().get(currentTocId).subEntries();
|
||||
}
|
||||
for (TableOfContentsData.EntryData entryData : entries) {
|
||||
|
||||
switch (entryData.type()) {
|
||||
case SECTION -> buildSection(entryData, currentTocId, context);
|
||||
case PARAGRAPH -> buildParagraph(entryData, currentTocId, context);
|
||||
default -> throw new NotImplementedException("Not yet implemented for type " + entryData.type());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void buildSection(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
|
||||
|
||||
SectionNode section = SectionNode.builder()
|
||||
.entities(new LinkedList<>())
|
||||
.pages(new LinkedList<>())
|
||||
.paragraphs(new LinkedList<>())
|
||||
.tables(new LinkedList<>())
|
||||
.subSections(new LinkedList<>())
|
||||
.tableOfContents(context.tableOfContents())
|
||||
.numberOnPage(entryData.numberOnPage())
|
||||
.build();
|
||||
|
||||
context.sections().add(section);
|
||||
section.setHeadline(toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(entryData.atomicTextBlock())), section));
|
||||
|
||||
if (!currentTocId.equals("")) {
|
||||
SectionNode parent = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
|
||||
section.setParentSection(parent);
|
||||
parent.getSubSections().add(section);
|
||||
}
|
||||
|
||||
PageNode page = getPage(entryData.page(), context);
|
||||
page.getMainBody().add(section);
|
||||
section.getPages().add(page);
|
||||
|
||||
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(section.getHeadline()), section);
|
||||
section.setTocId(sectionId);
|
||||
buildNodesFromTableOfContents(sectionId, context);
|
||||
}
|
||||
|
||||
|
||||
private void buildParagraph(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
|
||||
|
||||
PageNode page = getPage(entryData.page(), context);
|
||||
SectionNode parentSection = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
|
||||
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(entryData.numberOnPage()).page(page).parentSection(parentSection).build();
|
||||
AtomicTextBlock atomicTextBlock = toAtomicTextBlock(context.atomicTextBlockData.get(toIntExact(entryData.atomicTextBlock())), paragraph);
|
||||
paragraph.setAtomicTextBlock(atomicTextBlock);
|
||||
|
||||
if (!page.getMainBody().contains(parentSection)) {
|
||||
parentSection.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(paragraph);
|
||||
|
||||
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(currentTocId, NodeType.PARAGRAPH, buildSummary(atomicTextBlock), paragraph);
|
||||
paragraph.setTocId(tocId);
|
||||
}
|
||||
|
||||
|
||||
private PageNode buildPage(PageData p, Context context) {
|
||||
|
||||
PageNode page = PageNode.builder().height(p.getHeight()).width(p.getWidth()).number(p.getNumber()).mainBody(new LinkedList<>()).build();
|
||||
AtomicTextBlock header = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getHeader())), page);
|
||||
AtomicTextBlock footer = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getFooter())), page);
|
||||
page.setHeader(header);
|
||||
page.setFooter(footer);
|
||||
return page;
|
||||
}
|
||||
|
||||
|
||||
private static String buildSummary(com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock textBlock) {
|
||||
|
||||
if (textBlock == null) {
|
||||
return " probably a table";
|
||||
}
|
||||
|
||||
String[] words = textBlock.getFirstLine().toString().split(" ");
|
||||
int bound = Math.min(words.length, 4);
|
||||
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
|
||||
|
||||
return String.join(" ", list);
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock toAtomicTextBlock(AtomicTextBlockData atomicTextBlockData, DocumentGraphNode parent) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(atomicTextBlockData.getId())
|
||||
.searchText(atomicTextBlockData.getSearchText())
|
||||
.boundary(new Boundary(atomicTextBlockData.getStart(), atomicTextBlockData.getEnd()))
|
||||
.lineBreaks(Ints.asList(atomicTextBlockData.getLineBreaks()))
|
||||
.positions(Arrays.stream(atomicTextBlockData.getPositions())
|
||||
.map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3]))
|
||||
.toList())
|
||||
.stringIdxToPositionIdx(Ints.asList(atomicTextBlockData.getStringIdxToPositionIdx()))
|
||||
.parent(parent)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private PageNode getPage(Long pageIndex, Context context) {
|
||||
|
||||
return context.pages.stream()
|
||||
.filter(page -> page.getNumber() == toIntExact(pageIndex))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
|
||||
}
|
||||
|
||||
|
||||
record Context(DocumentData documentData, TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, List<AtomicTextBlockData> atomicTextBlockData) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
|
||||
public class EntityEnrichmentUtility {
|
||||
|
||||
public static EntityNode enrichEntity(EntityNode entity, TextBlock textBlock) {
|
||||
|
||||
entity.setPositions(textBlock.getPositions(entity.getBoundary()));
|
||||
entity.setTextAfter(findTextAfter(entity.getBoundary().end(), textBlock));
|
||||
entity.setTextBefore(findTextBefore(entity.getBoundary().start(), textBlock));
|
||||
entity.setValue(textBlock.subSequence(entity.getBoundary()).toString());
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
private static CharSequence findTextAfter(int index, TextBlock textBlock) {
|
||||
|
||||
int nextLineBreak = textBlock.getNextLinebreak(index);
|
||||
return textBlock.subSequence(index, nextLineBreak);
|
||||
}
|
||||
|
||||
|
||||
private static CharSequence findTextBefore(int index, TextBlock textBlock) {
|
||||
|
||||
int previousLinebreak = textBlock.getPreviousLinebreak(index);
|
||||
return textBlock.subSequence(previousLinebreak, index);
|
||||
}
|
||||
|
||||
}
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.util.Comparator;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
public class RangeComparators {
|
||||
|
||||
public static Comparator<Boundary> contained() {
|
||||
|
||||
return (range1, range2) -> {
|
||||
if (contained(range1, range2)) {
|
||||
return -1;
|
||||
} else if (contained(range2, range1)) {
|
||||
return 1;
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* @param range1 A Range
|
||||
* @param range2 Also Range
|
||||
* @return true, if range1 contains range2
|
||||
* false, otherwise
|
||||
*/
|
||||
public static boolean contained(Boundary range1, Boundary range2) {
|
||||
|
||||
return range1.start() <= range2.start() && range2.end() <= range1.end();
|
||||
}
|
||||
|
||||
public static boolean contained(Boundary range, int index) {
|
||||
|
||||
return range.start() <= index && index < range.end();
|
||||
}
|
||||
}
|
||||
+37
@@ -0,0 +1,37 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.regex.Matcher;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
|
||||
|
||||
public class RegexMatcher {
|
||||
|
||||
public static boolean anyMatch(CharSequence searchText, String regexPattern) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
return pattern.matcher(searchText).find();
|
||||
}
|
||||
|
||||
|
||||
public static Boundary findFirstBoundary(String regexPattern, CharSequence searchText) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
Matcher matcher = pattern.matcher(searchText);
|
||||
return new Boundary(matcher.start(), matcher.end());
|
||||
}
|
||||
|
||||
public static List<Boundary> findBoundaries(String regexPattern, CharSequence searchText) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
Matcher matcher = pattern.matcher(searchText);
|
||||
List<Boundary> boundaries = new LinkedList<>();
|
||||
while (matcher.find()) {
|
||||
boundaries.add(new Boundary(matcher.start(), matcher.end()));
|
||||
}
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
}
|
||||
-27
@@ -1,27 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import java.util.concurrent.Executor;
|
||||
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
import org.springframework.scheduling.annotation.AsyncConfigurerSupport;
|
||||
import org.springframework.scheduling.concurrent.ThreadPoolTaskExecutor;
|
||||
|
||||
@Configuration
|
||||
public class AsyncConfig extends AsyncConfigurerSupport {
|
||||
|
||||
@Override
|
||||
public Executor getAsyncExecutor() {
|
||||
|
||||
ThreadPoolTaskExecutor executor = new ThreadPoolTaskExecutor();
|
||||
|
||||
executor.setCorePoolSize(7);
|
||||
executor.setMaxPoolSize(42);
|
||||
executor.setQueueCapacity(11);
|
||||
executor.setThreadNamePrefix("TenantAwareTaskExecutor-");
|
||||
executor.setTaskDecorator(new TenantAwareTaskDecorator());
|
||||
executor.initialize();
|
||||
|
||||
return executor;
|
||||
}
|
||||
|
||||
}
|
||||
-105
@@ -1,105 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import java.nio.ByteBuffer;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.SecureRandom;
|
||||
import java.security.spec.KeySpec;
|
||||
import java.util.Base64;
|
||||
|
||||
import javax.annotation.PostConstruct;
|
||||
import javax.crypto.Cipher;
|
||||
import javax.crypto.SecretKey;
|
||||
import javax.crypto.SecretKeyFactory;
|
||||
import javax.crypto.spec.GCMParameterSpec;
|
||||
import javax.crypto.spec.PBEKeySpec;
|
||||
import javax.crypto.spec.SecretKeySpec;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.SneakyThrows;
|
||||
|
||||
@Service
|
||||
public class EncryptionDecryptionService {
|
||||
|
||||
@Value("${redaction-service.crypto.key:redaction}")
|
||||
private String key;
|
||||
|
||||
private SecretKey secretKey;
|
||||
private byte[] iv;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
@PostConstruct
|
||||
protected void postConstruct() {
|
||||
|
||||
SecureRandom secureRandom = new SecureRandom();
|
||||
iv = new byte[12];
|
||||
secureRandom.nextBytes(iv);
|
||||
secretKey = generateSecretKey(key, iv);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public String encrypt(String strToEncrypt) {
|
||||
|
||||
return Base64.getEncoder().encodeToString(encrypt(strToEncrypt.getBytes()));
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public String decrypt(String strToDecrypt) {
|
||||
|
||||
byte[] bytes = Base64.getDecoder().decode(strToDecrypt);
|
||||
return new String(decrypt(bytes), StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public byte[] encrypt(byte[] data) {
|
||||
|
||||
Cipher cipher = Cipher.getInstance("AES/GCM/NoPadding");
|
||||
GCMParameterSpec parameterSpec = new GCMParameterSpec(128, iv);
|
||||
cipher.init(Cipher.ENCRYPT_MODE, secretKey, parameterSpec);
|
||||
byte[] encryptedData = cipher.doFinal(data);
|
||||
ByteBuffer byteBuffer = ByteBuffer.allocate(4 + iv.length + encryptedData.length);
|
||||
byteBuffer.putInt(iv.length);
|
||||
byteBuffer.put(iv);
|
||||
byteBuffer.put(encryptedData);
|
||||
return byteBuffer.array();
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public byte[] decrypt(byte[] encryptedData) {
|
||||
|
||||
ByteBuffer byteBuffer = ByteBuffer.wrap(encryptedData);
|
||||
int noonceSize = byteBuffer.getInt();
|
||||
if (noonceSize < 12 || noonceSize >= 16) {
|
||||
throw new IllegalArgumentException("Nonce size is incorrect. Make sure that the incoming data is an AES encrypted file.");
|
||||
}
|
||||
byte[] iv = new byte[noonceSize];
|
||||
byteBuffer.get(iv);
|
||||
|
||||
SecretKey secretKey = generateSecretKey(key, iv);
|
||||
|
||||
byte[] cipherBytes = new byte[byteBuffer.remaining()];
|
||||
byteBuffer.get(cipherBytes);
|
||||
|
||||
Cipher cipher = Cipher.getInstance("AES/GCM/NoPadding");
|
||||
GCMParameterSpec parameterSpec = new GCMParameterSpec(128, iv);
|
||||
cipher.init(Cipher.DECRYPT_MODE, secretKey, parameterSpec);
|
||||
return cipher.doFinal(cipherBytes);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public SecretKey generateSecretKey(String password, byte[] iv) {
|
||||
|
||||
KeySpec spec = new PBEKeySpec(password.toCharArray(), iv, 65536, 128); // AES-128
|
||||
SecretKeyFactory secretKeyFactory = SecretKeyFactory.getInstance("PBKDF2WithHmacSHA1");
|
||||
byte[] key = secretKeyFactory.generateSecret(spec).getEncoded();
|
||||
return new SecretKeySpec(key, "AES");
|
||||
}
|
||||
|
||||
}
|
||||
-17
@@ -1,17 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import feign.RequestInterceptor;
|
||||
import feign.RequestTemplate;
|
||||
|
||||
@Component
|
||||
public class ForwardTenantInterceptor implements RequestInterceptor {
|
||||
|
||||
public static final String TENANT_HEADER_NAME = "X-TENANT-ID";
|
||||
|
||||
@Override
|
||||
public void apply(RequestTemplate template) {
|
||||
template.header(TENANT_HEADER_NAME, TenantContext.getTenantId());
|
||||
}
|
||||
}
|
||||
-49
@@ -1,49 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.multitenancy.ForwardTenantInterceptor.TENANT_HEADER_NAME;
|
||||
|
||||
import org.springframework.amqp.rabbit.config.AbstractRabbitListenerContainerFactory;
|
||||
import org.springframework.amqp.rabbit.core.RabbitTemplate;
|
||||
import org.springframework.beans.BeansException;
|
||||
import org.springframework.beans.factory.config.BeanPostProcessor;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
|
||||
@Configuration
|
||||
public class MultiTenancyMessagingConfiguration {
|
||||
|
||||
@Bean
|
||||
public static BeanPostProcessor multitenancyBeanPostProcessor() {
|
||||
|
||||
return new BeanPostProcessor() {
|
||||
|
||||
@Override
|
||||
public Object postProcessAfterInitialization(Object bean, String beanName) throws BeansException {
|
||||
|
||||
if (bean instanceof RabbitTemplate) {
|
||||
|
||||
((RabbitTemplate) bean).setBeforePublishPostProcessors(m -> {
|
||||
m.getMessageProperties().setHeader(TENANT_HEADER_NAME, TenantContext.getTenantId());
|
||||
return m;
|
||||
});
|
||||
|
||||
} else if (bean instanceof AbstractRabbitListenerContainerFactory) {
|
||||
|
||||
((AbstractRabbitListenerContainerFactory<?>) bean).setAfterReceivePostProcessors(m -> {
|
||||
String tenant = m.getMessageProperties().getHeader(TENANT_HEADER_NAME);
|
||||
|
||||
if (tenant != null) {
|
||||
TenantContext.setTenantId(tenant);
|
||||
} else {
|
||||
throw new RuntimeException("No Tenant is set queue message");
|
||||
}
|
||||
return m;
|
||||
});
|
||||
}
|
||||
return bean;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
-28
@@ -1,28 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
import org.springframework.web.servlet.config.annotation.InterceptorRegistry;
|
||||
|
||||
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
|
||||
|
||||
@Configuration
|
||||
public class MultiTenancyWebConfiguration extends DefaultWebMvcConfiguration {
|
||||
|
||||
private final TenantInterceptor tenantInterceptor;
|
||||
|
||||
|
||||
@Autowired
|
||||
public MultiTenancyWebConfiguration(TenantInterceptor tenantInterceptor) {
|
||||
|
||||
this.tenantInterceptor = tenantInterceptor;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public void addInterceptors(InterceptorRegistry registry) {
|
||||
|
||||
registry.addWebRequestInterceptor(tenantInterceptor);
|
||||
}
|
||||
|
||||
}
|
||||
-45
@@ -1,45 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.client.TenantsClient;
|
||||
import com.iqser.red.storage.commons.model.AzureStorageConnection;
|
||||
import com.iqser.red.storage.commons.model.S3StorageConnection;
|
||||
import com.iqser.red.storage.commons.service.StorageConnectionProvider;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class StorageConnectionProviderImpl implements StorageConnectionProvider {
|
||||
|
||||
private final TenantsClient tenantsClient;
|
||||
private final EncryptionDecryptionService encryptionDecryptionService;
|
||||
|
||||
|
||||
@Override
|
||||
public AzureStorageConnection getAzureStorageConnection(String tenantId) {
|
||||
|
||||
var tenant = tenantsClient.getTenant(tenantId);
|
||||
return AzureStorageConnection.builder()
|
||||
.connectionString(encryptionDecryptionService.decrypt(tenant.getAzureStorageConnection().getConnectionString()))
|
||||
.containerName(tenant.getAzureStorageConnection().getContainerName())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public S3StorageConnection getS3StorageConnection(String tenantId) {
|
||||
|
||||
var tenant = tenantsClient.getTenant(tenantId);
|
||||
return S3StorageConnection.builder()
|
||||
.key(tenant.getS3StorageConnection().getKey())
|
||||
.secret(encryptionDecryptionService.decrypt(tenant.getS3StorageConnection().getSecret()))
|
||||
.signerType(tenant.getS3StorageConnection().getSignerType())
|
||||
.bucketName(tenant.getS3StorageConnection().getBucketName())
|
||||
.region(tenant.getS3StorageConnection().getRegion())
|
||||
.endpoint(tenant.getS3StorageConnection().getEndpoint())
|
||||
.build();
|
||||
}
|
||||
|
||||
}
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import org.springframework.core.task.TaskDecorator;
|
||||
import org.springframework.lang.NonNull;
|
||||
|
||||
public class TenantAwareTaskDecorator implements TaskDecorator {
|
||||
|
||||
@Override
|
||||
@NonNull
|
||||
public Runnable decorate(@NonNull Runnable runnable) {
|
||||
|
||||
String tenantId = TenantContext.getTenantId();
|
||||
return () -> {
|
||||
try {
|
||||
TenantContext.setTenantId(tenantId);
|
||||
runnable.run();
|
||||
} finally {
|
||||
TenantContext.setTenantId(null);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
-29
@@ -1,29 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
public final class TenantContext {
|
||||
|
||||
private static InheritableThreadLocal<String> currentTenant = new InheritableThreadLocal<>();
|
||||
|
||||
|
||||
public static void setTenantId(String tenantId) {
|
||||
|
||||
log.debug("Setting tenantId to " + tenantId);
|
||||
currentTenant.set(tenantId);
|
||||
}
|
||||
|
||||
|
||||
public static String getTenantId() {
|
||||
|
||||
return currentTenant.get();
|
||||
}
|
||||
|
||||
|
||||
public static void clear() {
|
||||
|
||||
currentTenant.remove();
|
||||
}
|
||||
|
||||
}
|
||||
-35
@@ -1,35 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.multitenancy;
|
||||
|
||||
import org.springframework.stereotype.Component;
|
||||
import org.springframework.ui.ModelMap;
|
||||
import org.springframework.web.context.request.WebRequest;
|
||||
import org.springframework.web.context.request.WebRequestInterceptor;
|
||||
|
||||
@Component
|
||||
public class TenantInterceptor implements WebRequestInterceptor {
|
||||
|
||||
public static final String TENANT_HEADER_NAME = "X-TENANT-ID";
|
||||
|
||||
|
||||
@Override
|
||||
public void preHandle(WebRequest request) {
|
||||
|
||||
if (request.getHeader(TENANT_HEADER_NAME) != null) {
|
||||
TenantContext.setTenantId(request.getHeader(TENANT_HEADER_NAME));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public void postHandle(WebRequest request, ModelMap model) {
|
||||
|
||||
TenantContext.clear();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public void afterCompletion(WebRequest request, Exception ex) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+2
-2
@@ -12,8 +12,8 @@ import com.dslplatform.json.CompiledJson;
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Point;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
|
||||
+3
-3
@@ -10,12 +10,12 @@ import org.springframework.stereotype.Service;
|
||||
|
||||
import com.fasterxml.jackson.core.JsonProcessingException;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeResult;
|
||||
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
|
||||
import com.iqser.red.service.redaction.v1.model.StructureAnalyzeRequest;
|
||||
import com.iqser.red.service.redaction.v1.server.client.FileStatusProcessingUpdateClient;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.AnalyzeService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.analyze.AnalyzeService;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.type.DictionaryEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.type.DictionaryEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.SearchImplementation;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
|
||||
+17
-6
@@ -1,22 +1,25 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Engine;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.*;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Engine;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
|
||||
public class Entity implements ReasonHolder {
|
||||
|
||||
private String word;
|
||||
private String type;
|
||||
private boolean redaction;
|
||||
@@ -26,6 +29,7 @@ public class Entity implements ReasonHolder {
|
||||
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
|
||||
private List<TextPositionSequence> targetSequences;
|
||||
|
||||
|
||||
@EqualsAndHashCode.Include
|
||||
private Integer start;
|
||||
@EqualsAndHashCode.Include
|
||||
@@ -38,6 +42,9 @@ public class Entity implements ReasonHolder {
|
||||
@EqualsAndHashCode.Include
|
||||
private int sectionNumber;
|
||||
|
||||
@EqualsAndHashCode.Include
|
||||
private int paragraphNumber;
|
||||
|
||||
private boolean isDictionaryEntry;
|
||||
|
||||
private String textBefore;
|
||||
@@ -66,6 +73,7 @@ public class Entity implements ReasonHolder {
|
||||
String headline,
|
||||
int matchedRule,
|
||||
int sectionNumber,
|
||||
int paragraphNumber,
|
||||
String legalBasis,
|
||||
boolean isDictionaryEntry,
|
||||
String textBefore,
|
||||
@@ -85,6 +93,7 @@ public class Entity implements ReasonHolder {
|
||||
this.headline = headline;
|
||||
this.matchedRule = matchedRule;
|
||||
this.sectionNumber = sectionNumber;
|
||||
this.paragraphNumber = paragraphNumber;
|
||||
this.legalBasis = legalBasis;
|
||||
this.isDictionaryEntry = isDictionaryEntry;
|
||||
this.textBefore = textBefore;
|
||||
@@ -104,6 +113,7 @@ public class Entity implements ReasonHolder {
|
||||
Integer end,
|
||||
String headline,
|
||||
int sectionNumber,
|
||||
int paragraphNumber,
|
||||
boolean isDictionaryEntry,
|
||||
boolean isDossierDictionaryEntry,
|
||||
Engine engine,
|
||||
@@ -115,6 +125,7 @@ public class Entity implements ReasonHolder {
|
||||
this.end = end;
|
||||
this.headline = headline;
|
||||
this.sectionNumber = sectionNumber;
|
||||
this.paragraphNumber = paragraphNumber;
|
||||
this.isDictionaryEntry = isDictionaryEntry;
|
||||
this.isDossierDictionaryEntry = isDossierDictionaryEntry;
|
||||
this.engines.add(engine);
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.FileAttribute;
|
||||
import com.iqser.red.service.redaction.v1.model.FileAttribute;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.FileAttribute;
|
||||
import com.iqser.red.service.redaction.v1.model.FileAttribute;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Builder;
|
||||
import lombok.Getter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Getter
|
||||
@Builder
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class Paragraph {
|
||||
|
||||
SearchTextWithTextPositionModel searchTextToTextPosition;
|
||||
int sectionNumber;
|
||||
int paragraphNumber;
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Builder;
|
||||
import lombok.Getter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Builder
|
||||
@Getter
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class SearchTextWithTextPositionModel {
|
||||
|
||||
String searchText;
|
||||
List<Integer> lineBreaks;
|
||||
List<Integer> stringCoordsToPositionCoords;
|
||||
List<RedRectangle2D> positions;
|
||||
}
|
||||
+7
-19
@@ -1,5 +1,11 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
@@ -7,12 +13,6 @@ import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.SeparatorUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import lombok.Getter;
|
||||
|
||||
public class SearchableText {
|
||||
@@ -186,25 +186,13 @@ public class SearchableText {
|
||||
return stringRepresentation;
|
||||
}
|
||||
|
||||
|
||||
public static String buildString(List<TextPositionSequence> sequences) {
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
|
||||
TextPositionSequence previous = null;
|
||||
for (TextPositionSequence word : sequences) {
|
||||
|
||||
if (previous != null) {
|
||||
if (Math.abs(previous.getMaxYDirAdj() - word.getMaxYDirAdj()) > word.getTextHeight()) {
|
||||
sb.append('\n');
|
||||
} else {
|
||||
sb.append(' ');
|
||||
}
|
||||
}
|
||||
sb.append(word.toString());
|
||||
previous = word;
|
||||
sb.append(' ');
|
||||
}
|
||||
|
||||
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString()).replaceAll("\n", " ").replaceAll(" {2}", " ");
|
||||
}
|
||||
|
||||
|
||||
+319
-155
@@ -18,20 +18,25 @@ import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.commons.lang3.StringUtils;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.FileAttribute;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Engine;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionArea;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
|
||||
import com.iqser.red.service.redaction.v1.model.ArgumentType;
|
||||
import com.iqser.red.service.redaction.v1.model.Engine;
|
||||
import com.iqser.red.service.redaction.v1.model.FileAttribute;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.RedTextPosition;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SurroundingWordsService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.FindEntityDetails;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.OffsetStringUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.SearchImplementation;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
@@ -44,12 +49,11 @@ public class Section {
|
||||
|
||||
private boolean isLocal;
|
||||
|
||||
private Set<String> dictionaryTypes;
|
||||
|
||||
@Builder.Default
|
||||
private Map<String, Set<String>> localDictionaryAdds = new HashMap<>();
|
||||
|
||||
private Set<Entity> entities;
|
||||
@Builder.Default
|
||||
private Set<Entity> entities = new HashSet<>();
|
||||
|
||||
private Set<Entity> nerEntities;
|
||||
|
||||
@@ -65,8 +69,6 @@ public class Section {
|
||||
|
||||
private Map<String, CellValue> tabularData;
|
||||
|
||||
private Dictionary dictionary;
|
||||
|
||||
private SearchableText searchableText;
|
||||
|
||||
@Builder.Default
|
||||
@@ -85,21 +87,24 @@ public class Section {
|
||||
|
||||
private boolean isInTable;
|
||||
|
||||
@Builder.Default
|
||||
private List<Integer> cellStarts = new ArrayList<>();
|
||||
|
||||
|
||||
@Deprecated
|
||||
@SuppressWarnings("unused")
|
||||
@ThenAction
|
||||
public void addAiEntities(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.TYPE) String asType) {
|
||||
public void addAiEntities(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.TYPE) String asType, Dictionary dictionary) {
|
||||
|
||||
redactOrRecommendAiEntities(type, asType, false, 0, null, null);
|
||||
redactOrRecommendAiEntities(type, asType, false, 0, null, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@SuppressWarnings("unused")
|
||||
@ThenAction
|
||||
public void recommendAiEntities(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.TYPE) String asType) {
|
||||
public void recommendAiEntities(@Argument(ArgumentType.TYPE) String type, @Argument(ArgumentType.TYPE) String asType, Dictionary dictionary) {
|
||||
|
||||
redactOrRecommendAiEntities(type, asType, false, 0, null, null);
|
||||
redactOrRecommendAiEntities(type, asType, false, 0, null, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -109,9 +114,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactOrRecommendAiEntities(type, asType, true, ruleNumber, reason, legalBasis);
|
||||
redactOrRecommendAiEntities(type, asType, true, ruleNumber, reason, legalBasis, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -122,7 +128,8 @@ public class Section {
|
||||
@Argument(ArgumentType.INTEGER) int maxDistanceBetween,
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.INTEGER) int minPartMatches,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean allowDuplicateTypes) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean allowDuplicateTypes,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Set<String> combineSet = Set.of(combineTypes.split(","));
|
||||
|
||||
@@ -141,7 +148,7 @@ public class Section {
|
||||
} else if (!allowDuplicateTypes && foundParts.contains(entity.getType())) {
|
||||
if (numberOfMatchParts >= minPartMatches) {
|
||||
String value = searchText.substring(start, lastEnd);
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true, dictionary));
|
||||
}
|
||||
start = -1;
|
||||
lastEnd = -1;
|
||||
@@ -156,7 +163,7 @@ public class Section {
|
||||
} else if (entity.getType().equals(startType) && start != -1) {
|
||||
if (numberOfMatchParts >= minPartMatches) {
|
||||
String value = searchText.substring(start, lastEnd);
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true, dictionary));
|
||||
}
|
||||
start = entity.getStart();
|
||||
lastEnd = entity.getEnd();
|
||||
@@ -173,7 +180,7 @@ public class Section {
|
||||
|
||||
if (numberOfMatchParts >= minPartMatches) {
|
||||
String value = searchText.substring(start, lastEnd);
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true));
|
||||
found.addAll(findEntities(value, asType, false, true, 0, null, null, Engine.NER, true, dictionary));
|
||||
}
|
||||
|
||||
if (!found.isEmpty()) {
|
||||
@@ -207,6 +214,7 @@ public class Section {
|
||||
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && value.equals(attribute.getValue()));
|
||||
}
|
||||
|
||||
|
||||
@SuppressWarnings("unused")
|
||||
@WhenCondition
|
||||
public boolean fileAttributeContainsAnyOf(@Argument(ArgumentType.FILE_ATTRIBUTE) String label, @Argument(ArgumentType.STRING) Set<String> value) {
|
||||
@@ -214,6 +222,7 @@ public class Section {
|
||||
return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && value.contains(attribute.getValue()));
|
||||
}
|
||||
|
||||
|
||||
@SuppressWarnings("unused")
|
||||
@WhenCondition
|
||||
public boolean fileAttributeByIdEqualsIgnoreCase(@Argument(ArgumentType.FILE_ATTRIBUTE) String id, @Argument(ArgumentType.STRING) String value) {
|
||||
@@ -353,9 +362,10 @@ public class Section {
|
||||
public void expandByPrefixRegEx(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.REGEX) String prefixPattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group) {
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
Dictionary dictionary) {
|
||||
|
||||
expandByPrefixRegEx(type, prefixPattern, patternCaseInsensitive, group, null);
|
||||
expandByPrefixRegEx(type, prefixPattern, patternCaseInsensitive, group, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -365,7 +375,8 @@ public class Section {
|
||||
@Argument(ArgumentType.REGEX) String prefixPattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.REGEX) String valuePattern) {
|
||||
@Argument(ArgumentType.REGEX) String valuePattern,
|
||||
Dictionary dictionary) {
|
||||
|
||||
if (StringUtils.isEmpty(prefixPattern)) {
|
||||
return;
|
||||
@@ -408,7 +419,8 @@ public class Section {
|
||||
entity.getRedactionReason(),
|
||||
entity.getLegalBasis(),
|
||||
Engine.RULE,
|
||||
false);
|
||||
false,
|
||||
dictionary);
|
||||
expanded.addAll(EntitySearchUtils.findNonOverlappingMatchEntities(entities, expandedEntities));
|
||||
}
|
||||
}
|
||||
@@ -424,9 +436,10 @@ public class Section {
|
||||
public void expandByRegEx(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.REGEX) String suffixPattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group) {
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
Dictionary dictionary) {
|
||||
|
||||
expandByRegEx(type, suffixPattern, patternCaseInsensitive, group, null);
|
||||
expandByRegEx(type, suffixPattern, patternCaseInsensitive, group, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -436,7 +449,8 @@ public class Section {
|
||||
@Argument(ArgumentType.REGEX) String suffixPattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.REGEX) String valuePattern) {
|
||||
@Argument(ArgumentType.REGEX) String valuePattern,
|
||||
Dictionary dictionary) {
|
||||
|
||||
if (StringUtils.isEmpty(suffixPattern)) {
|
||||
return;
|
||||
@@ -479,7 +493,8 @@ public class Section {
|
||||
entity.getRedactionReason(),
|
||||
entity.getLegalBasis(),
|
||||
Engine.RULE,
|
||||
false);
|
||||
false,
|
||||
dictionary);
|
||||
expanded.addAll(EntitySearchUtils.findNonOverlappingMatchEntities(entities, expandedEntities));
|
||||
}
|
||||
}
|
||||
@@ -535,9 +550,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, true);
|
||||
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, true, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -547,9 +563,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, null, false);
|
||||
redactLineAfter(start, asType, ruleNumber, redactEverywhere, reason, null, false, dictionary);
|
||||
|
||||
}
|
||||
|
||||
@@ -562,23 +579,25 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false);
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@ThenAction
|
||||
@SuppressWarnings("unused")
|
||||
public void redactByRegExWithNewlines(@Argument(ArgumentType.REGEX) String pattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactByRegExWithNewlines(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false);
|
||||
redactByRegExWithNewlines(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -591,9 +610,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, skipRemoveEntitiesContainedInLarger);
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, skipRemoveEntitiesContainedInLarger, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -604,9 +624,10 @@ public class Section {
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false, false);
|
||||
redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -619,9 +640,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, legalBasis, true, false, false, false);
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, legalBasis, true, false, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -634,9 +656,10 @@ public class Section {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean excludeHeadLine,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, legalBasis, true, false, false, false);
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, legalBasis, true, false, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -651,7 +674,8 @@ public class Section {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start,
|
||||
stop,
|
||||
@@ -665,7 +689,9 @@ public class Section {
|
||||
legalBasis,
|
||||
true,
|
||||
skipRemoveEntitiesContainedInLarger,
|
||||
sortedResult, false);
|
||||
sortedResult,
|
||||
false,
|
||||
dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -683,7 +709,8 @@ public class Section {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean ignoreTables) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean ignoreTables,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start,
|
||||
stop,
|
||||
@@ -697,11 +724,12 @@ public class Section {
|
||||
legalBasis,
|
||||
true,
|
||||
skipRemoveEntitiesContainedInLarger,
|
||||
sortedResult, ignoreTables);
|
||||
sortedResult,
|
||||
ignoreTables,
|
||||
dictionary);
|
||||
}
|
||||
|
||||
|
||||
|
||||
@ThenAction
|
||||
@SuppressWarnings("unused")
|
||||
public void redactBetween(@Argument(ArgumentType.STRING) String start,
|
||||
@@ -715,7 +743,8 @@ public class Section {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start,
|
||||
stop,
|
||||
@@ -729,7 +758,9 @@ public class Section {
|
||||
legalBasis,
|
||||
true,
|
||||
skipRemoveEntitiesContainedInLarger,
|
||||
sortedResult, false);
|
||||
sortedResult,
|
||||
false,
|
||||
dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -747,7 +778,8 @@ public class Section {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.STRING) String legalBasis,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult) {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean sortedResult,
|
||||
Dictionary dictionary) {
|
||||
|
||||
String startValue = getFirstRexExMatch(searchText, startPattern, startPatternCaseInsensitive, startGroup);
|
||||
|
||||
@@ -772,7 +804,9 @@ public class Section {
|
||||
legalBasis,
|
||||
true,
|
||||
skipRemoveEntitiesContainedInLarger,
|
||||
sortedResult, false);
|
||||
sortedResult,
|
||||
false,
|
||||
dictionary);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -785,21 +819,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start,
|
||||
stop,
|
||||
false,
|
||||
false,
|
||||
asType,
|
||||
ruleNumber,
|
||||
redactEverywhere,
|
||||
false,
|
||||
reason,
|
||||
null,
|
||||
false,
|
||||
false,
|
||||
false, false);
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, null, false, false, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -811,21 +834,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean excludeHeadLine,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween(start,
|
||||
stop,
|
||||
false,
|
||||
false,
|
||||
asType,
|
||||
ruleNumber,
|
||||
redactEverywhere,
|
||||
excludeHeadLine,
|
||||
reason,
|
||||
null,
|
||||
false,
|
||||
false,
|
||||
false, false);
|
||||
redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, null, false, false, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -837,9 +849,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, legalBasis, true);
|
||||
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, legalBasis, true, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -850,9 +863,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, null, false);
|
||||
redactLinesBetween(start, stop, asType, ruleNumber, redactEverywhere, reason, null, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -863,9 +877,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean addAsRecommendations,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis);
|
||||
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -875,9 +890,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean addAsRecommendations,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
annotateCell(cellHeader, ruleNumber, type, false, addAsRecommendations, reason, null);
|
||||
annotateCell(cellHeader, ruleNumber, type, false, addAsRecommendations, reason, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -889,9 +905,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true);
|
||||
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -902,9 +919,10 @@ public class Section {
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason) {
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false);
|
||||
redactAndRecommendByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -973,9 +991,10 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Set<Entity> found = findEntities(value.trim(), asType, true, true, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(value.trim(), asType, true, true, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
|
||||
}
|
||||
|
||||
@@ -1001,7 +1020,8 @@ public class Section {
|
||||
public void expandToFalsePositiveByRegEx(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.STRING) String pattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group) {
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
|
||||
|
||||
@@ -1017,7 +1037,7 @@ public class Section {
|
||||
while (matcher.find()) {
|
||||
String match = matcher.group(group);
|
||||
if (StringUtils.isNotBlank(match)) {
|
||||
expanded.addAll(findEntities(entity.getWord() + match, type, false, false, 0, null, null, Engine.RULE, false));
|
||||
expanded.addAll(findEntities(entity.getWord() + match, type, false, false, 0, null, null, Engine.RULE, false, dictionary));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1033,7 +1053,8 @@ public class Section {
|
||||
public void addHintAnnotationByRegEx(@Argument(ArgumentType.REGEX) String pattern,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive,
|
||||
@Argument(ArgumentType.INTEGER) int group,
|
||||
@Argument(ArgumentType.TYPE) String asType) {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
|
||||
|
||||
@@ -1042,7 +1063,7 @@ public class Section {
|
||||
while (matcher.find()) {
|
||||
String match = matcher.group(group);
|
||||
if (StringUtils.isNotBlank(match)) {
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, false, 0, null, null, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, false, 0, null, null, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
}
|
||||
}
|
||||
@@ -1051,9 +1072,9 @@ public class Section {
|
||||
|
||||
@ThenAction
|
||||
@SuppressWarnings("unused")
|
||||
public void addHintAnnotation(@Argument(ArgumentType.STRING) String value, @Argument(ArgumentType.TYPE) String asType) {
|
||||
public void addHintAnnotation(@Argument(ArgumentType.STRING) String value, @Argument(ArgumentType.TYPE) String asType, Dictionary dictionary) {
|
||||
|
||||
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
|
||||
}
|
||||
|
||||
@@ -1085,9 +1106,12 @@ public class Section {
|
||||
|
||||
@ThenAction
|
||||
@SuppressWarnings("unused")
|
||||
public void highlightCell(@Argument(ArgumentType.STRING) String cellHeader, @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.TYPE) String type) {
|
||||
public void highlightCell(@Argument(ArgumentType.STRING) String cellHeader,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.TYPE) String type,
|
||||
Dictionary dictionary) {
|
||||
|
||||
annotateCell(cellHeader, ruleNumber, type, false, false, null, null);
|
||||
annotateCell(cellHeader, ruleNumber, type, false, false, null, null, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1096,9 +1120,10 @@ public class Section {
|
||||
public void redactSectionText(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween("", "", type, ruleNumber, false, false, reason, legalBasis, true, false);
|
||||
redactBetween("", "", type, ruleNumber, false, false, reason, legalBasis, true, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1107,9 +1132,10 @@ public class Section {
|
||||
public void redactSectionTextWithoutHeadLine(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactBetween("", "", type, ruleNumber, false, true, reason, legalBasis, true, false);
|
||||
redactBetween("", "", type, ruleNumber, false, true, reason, legalBasis, true, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1117,13 +1143,14 @@ public class Section {
|
||||
public void redactHeadline(@Argument(ArgumentType.TYPE) String type,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
if (!headline.isBlank()) {
|
||||
|
||||
String cleanHeadline = headline.replaceAll("\\n", " ").replaceAll(" ", " ").trim();
|
||||
if (searchText.contains(cleanHeadline)) {
|
||||
Set<Entity> found = findEntities(cleanHeadline, type, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(cleanHeadline, type, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
}
|
||||
}
|
||||
@@ -1139,7 +1166,8 @@ public class Section {
|
||||
@Argument(ArgumentType.TYPE) String asType,
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
|
||||
|
||||
@@ -1148,7 +1176,7 @@ public class Section {
|
||||
while (findMatcher.find()) {
|
||||
String findMatch = findMatcher.group(group);
|
||||
if (StringUtils.isNotBlank(findMatch)) {
|
||||
Set<Entity> found = findEntities(findMatch.trim(), asType, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(findMatch.trim(), asType, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
|
||||
for (Entity entity : found) {
|
||||
|
||||
@@ -1235,9 +1263,10 @@ public class Section {
|
||||
@Argument(ArgumentType.RULE_NUMBER) int ruleNumber,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, false, false);
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, false, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1249,9 +1278,10 @@ public class Section {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean redactEverywhere,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, skipRemoveEntitiesContainedInLarger, false);
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, skipRemoveEntitiesContainedInLarger, false, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1264,9 +1294,10 @@ public class Section {
|
||||
@Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger,
|
||||
@Argument(ArgumentType.BOOLEAN) boolean onlyExactMatch,
|
||||
@Argument(ArgumentType.STRING) String reason,
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis) {
|
||||
@Argument(ArgumentType.LEGAL_BASIS) String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, skipRemoveEntitiesContainedInLarger, onlyExactMatch);
|
||||
redactLineAfterAcrossColumns(start, asType, ruleNumber, redactEverywhere, reason, legalBasis, skipRemoveEntitiesContainedInLarger, onlyExactMatch, dictionary);
|
||||
}
|
||||
|
||||
|
||||
@@ -1277,14 +1308,15 @@ public class Section {
|
||||
int ruleNumber,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean redaction) {
|
||||
boolean redaction,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
|
||||
Matcher matcher = compiledPattern.matcher(searchText);
|
||||
while (matcher.find()) {
|
||||
String match = matcher.group(group);
|
||||
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
localDictionaryAdds.computeIfAbsent(asType, x -> new HashSet<>()).add(match);
|
||||
}
|
||||
@@ -1292,15 +1324,16 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> findEntities(String value,
|
||||
String asType,
|
||||
boolean caseInsensitive,
|
||||
boolean redacted,
|
||||
int ruleNumber,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
Engine engine,
|
||||
boolean asRecommendation) {
|
||||
public Set<Entity> findEntities(String value,
|
||||
String asType,
|
||||
boolean caseInsensitive,
|
||||
boolean redacted,
|
||||
int ruleNumber,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
Engine engine,
|
||||
boolean asRecommendation,
|
||||
Dictionary dictionary) {
|
||||
|
||||
String text = caseInsensitive ? searchText.toLowerCase() : searchText;
|
||||
Set<Entity> found = EntitySearchUtils.findEntities(text,
|
||||
@@ -1347,7 +1380,14 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void annotateCell(String cellHeader, int ruleNumber, String type, boolean redact, boolean addAsRecommendations, String reason, String legalBasis) {
|
||||
private void annotateCell(String cellHeader,
|
||||
int ruleNumber,
|
||||
String type,
|
||||
boolean redact,
|
||||
boolean addAsRecommendations,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
Dictionary dictionary) {
|
||||
|
||||
String cleanHeaderName = cellHeader.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
|
||||
|
||||
@@ -1363,6 +1403,7 @@ public class Section {
|
||||
value.getRowSpanStart() + word.length(),
|
||||
headline,
|
||||
sectionNumber,
|
||||
-1,
|
||||
false,
|
||||
false,
|
||||
Engine.RULE,
|
||||
@@ -1410,14 +1451,21 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void redactLineAfter(String start, String asType, int ruleNumber, boolean redactEverywhere, String reason, String legalBasis, boolean redaction) {
|
||||
private void redactLineAfter(String start,
|
||||
String asType,
|
||||
int ruleNumber,
|
||||
boolean redactEverywhere,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean redaction,
|
||||
Dictionary dictionary) {
|
||||
|
||||
String[] values = StringUtils.substringsBetween(text, start, "\n");
|
||||
|
||||
if (values != null) {
|
||||
for (String value : values) {
|
||||
if (StringUtils.isNotBlank(value)) {
|
||||
Set<Entity> found = findEntities(value.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(value.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
|
||||
if (redactEverywhere && !isLocal()) {
|
||||
@@ -1436,7 +1484,8 @@ public class Section {
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean skipRemoveEntitiesContainedInLarger,
|
||||
boolean onlyExactMatch) {
|
||||
boolean onlyExactMatch,
|
||||
Dictionary dictionary) {
|
||||
|
||||
var stringOffsets = OffsetStringUtils.substringsBetween(searchableText.getAsStringWithLinebreaksSorted(), start, "\n");
|
||||
|
||||
@@ -1444,8 +1493,9 @@ public class Section {
|
||||
for (var stringOffset : stringOffsets) {
|
||||
if (StringUtils.isNotBlank(stringOffset.getValue())) {
|
||||
var trimmedOffsetString = stringOffset.trim();
|
||||
Set<Entity> found = findEntities(trimmedOffsetString.getValue(), asType, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false).stream()
|
||||
.filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()).collect(Collectors.toSet());
|
||||
Set<Entity> found = findEntities(trimmedOffsetString.getValue(), asType, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary).stream()
|
||||
.filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd())
|
||||
.collect(Collectors.toSet());
|
||||
found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger));
|
||||
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
@@ -1459,7 +1509,16 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void redactByRegExWithNewlines(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis, boolean redaction, boolean skipRemoveEntitiesContainedInLarger) {
|
||||
private void redactByRegExWithNewlines(String pattern,
|
||||
boolean patternCaseInsensitive,
|
||||
int group,
|
||||
String asType,
|
||||
int ruleNumber,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean redaction,
|
||||
boolean skipRemoveEntitiesContainedInLarger,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledMultilinePattern(pattern, patternCaseInsensitive);
|
||||
|
||||
@@ -1468,7 +1527,7 @@ public class Section {
|
||||
while (matcher.find()) {
|
||||
String match = matcher.group(group);
|
||||
if (StringUtils.isNotBlank(match)) {
|
||||
Set<Entity> found = findEntities(match.replaceAll("\\n", " ").trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(match.replaceAll("\\n", " ").trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger));
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
}
|
||||
@@ -1476,7 +1535,16 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis, boolean redaction, boolean skipRemoveEntitiesContainedInLarger) {
|
||||
private void redactByRegEx(String pattern,
|
||||
boolean patternCaseInsensitive,
|
||||
int group,
|
||||
String asType,
|
||||
int ruleNumber,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean redaction,
|
||||
boolean skipRemoveEntitiesContainedInLarger,
|
||||
Dictionary dictionary) {
|
||||
|
||||
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
|
||||
|
||||
@@ -1485,7 +1553,7 @@ public class Section {
|
||||
while (matcher.find()) {
|
||||
String match = matcher.group(group);
|
||||
if (StringUtils.isNotBlank(match)) {
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger));
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
}
|
||||
@@ -1522,10 +1590,10 @@ public class Section {
|
||||
boolean redaction,
|
||||
boolean skipRemoveEntitiesContainedInLarger,
|
||||
boolean sortedResult,
|
||||
boolean ignoreTables) {
|
||||
boolean ignoreTables,
|
||||
Dictionary dictionary) {
|
||||
|
||||
|
||||
if(isInTable && ignoreTables){
|
||||
if (isInTable && ignoreTables) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1559,7 +1627,7 @@ public class Section {
|
||||
searchString = searchString + stop;
|
||||
}
|
||||
|
||||
Set<Entity> found = findEntities(searchString.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(searchString.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
found.forEach(f -> {
|
||||
f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger);
|
||||
if (sortedResult) {
|
||||
@@ -1582,7 +1650,15 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void redactLinesBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere, String reason, String legalBasis, boolean redaction) {
|
||||
private void redactLinesBetween(String start,
|
||||
String stop,
|
||||
String asType,
|
||||
int ruleNumber,
|
||||
boolean redactEverywhere,
|
||||
String reason,
|
||||
String legalBasis,
|
||||
boolean redaction,
|
||||
Dictionary dictionary) {
|
||||
|
||||
String[] values = StringUtils.substringsBetween(text, start, stop);
|
||||
|
||||
@@ -1597,7 +1673,7 @@ public class Section {
|
||||
return;
|
||||
}
|
||||
|
||||
Set<Entity> found = findEntities(line.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false);
|
||||
Set<Entity> found = findEntities(line.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false, dictionary);
|
||||
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
|
||||
|
||||
if (redactEverywhere && !isLocal()) {
|
||||
@@ -1610,7 +1686,7 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private void redactOrRecommendAiEntities(String type, String asType, boolean redact, int ruleNumber, String reason, String legalBasis) {
|
||||
private void redactOrRecommendAiEntities(String type, String asType, boolean redact, int ruleNumber, String reason, String legalBasis, Dictionary dictionary) {
|
||||
|
||||
Set<Entity> entitiesOfType = nerEntities.stream().filter(nerEntity -> nerEntity.getType().equals(type)).collect(Collectors.toSet());
|
||||
List<String> values = entitiesOfType.stream().map(Entity::getWord).collect(Collectors.toList());
|
||||
@@ -1675,10 +1751,98 @@ public class Section {
|
||||
|
||||
}
|
||||
|
||||
|
||||
public boolean findDictionaryEntities(Dictionary dictionary, RedactionServiceSettings redactionServiceSettings) {
|
||||
|
||||
findEntities(dictionary);
|
||||
|
||||
if (cellStarts != null && !cellStarts.isEmpty()) {
|
||||
SurroundingWordsService.addSurroundingText(entities,
|
||||
searchableText,
|
||||
dictionary,
|
||||
cellStarts,
|
||||
redactionServiceSettings.getSurroundingWordsOffsetWindow(),
|
||||
redactionServiceSettings.getNumberOfSurroundingWords());
|
||||
} else {
|
||||
SurroundingWordsService.addSurroundingText(entities,
|
||||
searchableText,
|
||||
dictionary,
|
||||
redactionServiceSettings.getSurroundingWordsOffsetWindow(),
|
||||
redactionServiceSettings.getNumberOfSurroundingWords());
|
||||
}
|
||||
|
||||
if (!isLocal && manualRedactions != null) {
|
||||
|
||||
var approvedForceRedactions = manualRedactions.getForceRedactions()
|
||||
.stream()
|
||||
.filter(fr -> fr.getStatus() == AnnotationStatus.APPROVED)
|
||||
.filter(fr -> fr.getRequestDate() != null)
|
||||
.collect(Collectors.toList());
|
||||
// only approved id removals, that haven't been forced back afterwards
|
||||
var idsToRemove = manualRedactions.getIdsToRemove()
|
||||
.stream()
|
||||
.filter(idr -> idr.getStatus() == AnnotationStatus.APPROVED && !idr.isRemoveFromDictionary())
|
||||
.filter(idr -> idr.getRequestDate() != null)
|
||||
.filter(idr -> approvedForceRedactions.stream()
|
||||
.noneMatch(forceRedact -> forceRedact.getAnnotationId().equals(idr.getAnnotationId()) && forceRedact.getRequestDate().isAfter(idr.getRequestDate())))
|
||||
.map(IdRemoval::getAnnotationId)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
if (images != null && !images.isEmpty() && manualRedactions.getImageRecategorization() != null) {
|
||||
for (Image image : images) {
|
||||
String imageId = IdBuilder.buildId(image.getPosition(), image.getPage());
|
||||
for (ManualImageRecategorization imageRecategorization : manualRedactions.getImageRecategorization()) {
|
||||
if (imageRecategorization.getStatus().equals(AnnotationStatus.APPROVED) && imageRecategorization.getAnnotationId().equals(imageId)) {
|
||||
image.setType(imageRecategorization.getType());
|
||||
}
|
||||
}
|
||||
if (idsToRemove.contains(imageId)) {
|
||||
image.setIgnored(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
entities.forEach(entity -> entity.getPositionSequences().forEach(ps -> {
|
||||
if (idsToRemove.contains(ps.getId())) {
|
||||
entity.setIgnored(true);
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
private void findEntities(Dictionary dictionary) {
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
String searchableString = searchableText.asString();
|
||||
|
||||
if (StringUtils.isEmpty(searchableString)) {
|
||||
entities = new HashSet<>();
|
||||
return;
|
||||
}
|
||||
|
||||
String lowercaseInputString = searchableString.toLowerCase();
|
||||
for (DictionaryModel model : dictionary.getDictionaryModels()) {
|
||||
|
||||
var searchImplementation = isLocal ? model.getLocalSearch() : model.getEntriesSearch();
|
||||
var entities = EntitySearchUtils.findEntities(model.isCaseInsensitive() ? lowercaseInputString : searchableString,
|
||||
searchImplementation,
|
||||
model,
|
||||
new FindEntityDetails(model.getType(),
|
||||
headline,
|
||||
sectionNumber,
|
||||
!isLocal,
|
||||
model.isDossierDictionary(),
|
||||
isLocal ? Engine.RULE : Engine.DICTIONARY,
|
||||
isLocal ? EntityType.RECOMMENDATION : EntityType.ENTITY));
|
||||
|
||||
EntitySearchUtils.addOrAddEngine(found, entities);
|
||||
}
|
||||
|
||||
entities = EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary, manualRedactions);
|
||||
nerEntities = EntitySearchUtils.clearAndFindPositions(nerEntities, searchableText, dictionary, manualRedactions);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
-16
@@ -1,16 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
public class SectionSearchableTextPair {
|
||||
|
||||
private Section section;
|
||||
private SearchableText searchableText;
|
||||
private List<Integer> cellStarts;
|
||||
|
||||
}
|
||||
+122
-113
@@ -1,30 +1,38 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service.analyze;
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.redaction.service.ImportedRedactionService.IMPORTED_REDACTION_TYPE;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.kie.api.runtime.KieContainer;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeResult;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.FileAttribute;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.legalbasis.LegalBasis;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogLegalBasis;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionArea;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.section.SectionGrid;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualForceRedaction;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualLegalBasisChange;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualResizeRedaction;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.legalbasis.LegalBasis;
|
||||
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
|
||||
import com.iqser.red.service.redaction.v1.model.FileAttribute;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLog;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLogLegalBasis;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionGrid;
|
||||
import com.iqser.red.service.redaction.v1.model.StructureAnalyzeRequest;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.SectionText;
|
||||
@@ -33,97 +41,47 @@ import com.iqser.red.service.redaction.v1.server.classification.model.Simplified
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Text;
|
||||
import com.iqser.red.service.redaction.v1.server.client.LegalBasisClient;
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
|
||||
import com.iqser.red.service.redaction.v1.server.document.services.DocumentGraphFactory;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryIncrement;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryIncrementValue;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryVersion;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PageEntities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.ImportedRedactionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionChangeLogService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogCreatorService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SectionGridCreatorService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SectionTextBuilderService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.entityredaction.EntityRedactionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.RedRectangle2D;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.SearchImplementation;
|
||||
import com.iqser.red.service.redaction.v1.server.segmentation.ImageService;
|
||||
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
|
||||
|
||||
import io.micrometer.core.annotation.Timed;
|
||||
import io.micrometer.core.instrument.FunctionTimer;
|
||||
import io.micrometer.core.instrument.MeterRegistry;
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.SneakyThrows;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
@RequiredArgsConstructor
|
||||
public class AnalyzeService {
|
||||
|
||||
DictionaryService dictionaryService;
|
||||
DroolsExecutionService droolsExecutionService;
|
||||
EntityRedactionService entityRedactionService;
|
||||
RedactionLogCreatorService redactionLogCreatorService;
|
||||
RedactionStorageService redactionStorageService;
|
||||
PdfSegmentationService pdfSegmentationService;
|
||||
RedactionChangeLogService redactionChangeLogService;
|
||||
LegalBasisClient legalBasisClient;
|
||||
RedactionServiceSettings redactionServiceSettings;
|
||||
SectionTextBuilderService sectionTextBuilderService;
|
||||
SectionGridCreatorService sectionGridCreatorService;
|
||||
ImageService imageService;
|
||||
ImportedRedactionService importedRedactionService;
|
||||
SectionFinder sectionFinder;
|
||||
PagewiseCounter analyzeCounter = new PagewiseCounter();
|
||||
|
||||
|
||||
@SuppressWarnings("SpringJavaInjectionPointsAutowiringInspection")
|
||||
public AnalyzeService(DictionaryService dictionaryService,
|
||||
DroolsExecutionService droolsExecutionService,
|
||||
EntityRedactionService entityRedactionService,
|
||||
RedactionLogCreatorService redactionLogCreatorService,
|
||||
RedactionStorageService redactionStorageService,
|
||||
PdfSegmentationService pdfSegmentationService,
|
||||
RedactionChangeLogService redactionChangeLogService,
|
||||
LegalBasisClient legalBasisClient,
|
||||
RedactionServiceSettings redactionServiceSettings,
|
||||
SectionTextBuilderService sectionTextBuilderService,
|
||||
SectionGridCreatorService sectionGridCreatorService,
|
||||
ImageService imageService,
|
||||
ImportedRedactionService importedRedactionService,
|
||||
SectionFinder sectionFinder,
|
||||
MeterRegistry meterRegistry) {
|
||||
|
||||
this.dictionaryService = dictionaryService;
|
||||
this.droolsExecutionService = droolsExecutionService;
|
||||
this.entityRedactionService = entityRedactionService;
|
||||
this.redactionLogCreatorService = redactionLogCreatorService;
|
||||
this.redactionStorageService = redactionStorageService;
|
||||
this.pdfSegmentationService = pdfSegmentationService;
|
||||
this.redactionChangeLogService = redactionChangeLogService;
|
||||
this.legalBasisClient = legalBasisClient;
|
||||
this.redactionServiceSettings = redactionServiceSettings;
|
||||
this.sectionTextBuilderService = sectionTextBuilderService;
|
||||
this.sectionGridCreatorService = sectionGridCreatorService;
|
||||
this.imageService = imageService;
|
||||
this.importedRedactionService = importedRedactionService;
|
||||
this.sectionFinder = sectionFinder;
|
||||
|
||||
// This is just here as an example of how to do a timer with a counter.
|
||||
// For a full implementation this should be moved to a common helper class.
|
||||
FunctionTimer.builder("redactmanager_analyze.pagewise", //
|
||||
analyzeCounter, //
|
||||
PagewiseCounter::getTotalPageCount, //
|
||||
PagewiseCounter::getTotalTimeMillis, //
|
||||
TimeUnit.MILLISECONDS) //
|
||||
.register(meterRegistry);
|
||||
}
|
||||
private final DictionaryService dictionaryService;
|
||||
private final DroolsExecutionService droolsExecutionService;
|
||||
private final EntityRedactionService entityRedactionService;
|
||||
private final RedactionLogCreatorService redactionLogCreatorService;
|
||||
private final RedactionStorageService redactionStorageService;
|
||||
private final PdfSegmentationService pdfSegmentationService;
|
||||
private final RedactionChangeLogService redactionChangeLogService;
|
||||
private final LegalBasisClient legalBasisClient;
|
||||
private final RedactionServiceSettings redactionServiceSettings;
|
||||
private final SectionTextBuilderService sectionTextBuilderService;
|
||||
private final SectionGridCreatorService sectionGridCreatorService;
|
||||
private final ImageService imageService;
|
||||
private final ImportedRedactionService importedRedactionService;
|
||||
private final DocumentGraphFactory documentGraphFactory;
|
||||
|
||||
|
||||
@Timed("redactmanager_analyzeDocumentStructure")
|
||||
@@ -151,7 +109,6 @@ public class AnalyzeService {
|
||||
}
|
||||
|
||||
List<SectionText> sectionTexts = sectionTextBuilderService.buildSectionText(classifiedDoc);
|
||||
|
||||
sectionGridCreatorService.createSectionGrid(classifiedDoc, pageCount);
|
||||
|
||||
Text text = new Text(pageCount, sectionTexts);
|
||||
@@ -197,25 +154,32 @@ public class AnalyzeService {
|
||||
new DictionaryVersion(redactionLog.getDictionaryVersion(), redactionLog.getDossierDictionaryVersion()),
|
||||
analyzeRequest.getDossierId());
|
||||
|
||||
Set<Integer> sectionsToReanalyse = analyzeRequest.getSectionsToReanalyse().isEmpty() //
|
||||
? sectionFinder.findSectionsToReanalyse(dictionaryIncrement, redactionLog, text, analyzeRequest) //
|
||||
: analyzeRequest.getSectionsToReanalyse();
|
||||
log.info("Should reanalyze {} sections for request: {}", sectionsToReanalyse.size(), analyzeRequest);
|
||||
Set<Integer> sectionsToReanalyse = !analyzeRequest.getSectionsToReanalyse().isEmpty() ? analyzeRequest.getSectionsToReanalyse() : findSectionsToReanalyse(
|
||||
dictionaryIncrement,
|
||||
redactionLog,
|
||||
text,
|
||||
analyzeRequest);
|
||||
|
||||
if (sectionsToReanalyse.isEmpty()) {
|
||||
return finalizeAnalysis(analyzeRequest, startTime, redactionLog, text, dictionaryIncrement.getDictionaryVersion(), true, new HashSet<>());
|
||||
}
|
||||
|
||||
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId());
|
||||
NerEntities nerEntities;
|
||||
if (redactionServiceSettings.isNerServiceEnabled()) {
|
||||
nerEntities = redactionStorageService.getNerEntities(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
|
||||
} else {
|
||||
nerEntities = new NerEntities();
|
||||
}
|
||||
|
||||
List<SectionText> reanalysisSections = text.getSectionTexts()
|
||||
.stream()
|
||||
.filter(sectionText -> sectionsToReanalyse.contains(sectionText.getSectionNumber()))
|
||||
.collect(Collectors.toList());
|
||||
NerEntities nerEntities = redactionServiceSettings.isNerServiceEnabled() //
|
||||
? redactionStorageService.getNerEntities(analyzeRequest.getDossierId(), analyzeRequest.getFileId()) //
|
||||
: new NerEntities();
|
||||
|
||||
KieContainer kieContainer = droolsExecutionService.updateRules(analyzeRequest.getDossierTemplateId());
|
||||
|
||||
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(analyzeRequest.getDossierTemplateId(), analyzeRequest.getDossierId());
|
||||
|
||||
PageEntities pageEntities = entityRedactionService.findEntities(dictionary, reanalysisSections, kieContainer, analyzeRequest, nerEntities);
|
||||
|
||||
var newRedactionLogEntries = redactionLogCreatorService.createRedactionLog(pageEntities, text.getNumberOfPages(), analyzeRequest.getDossierTemplateId());
|
||||
@@ -275,13 +239,48 @@ public class AnalyzeService {
|
||||
}
|
||||
|
||||
|
||||
@Timed("redactmanager_findSectionsToReanalyse")
|
||||
private Set<Integer> findSectionsToReanalyse(DictionaryIncrement dictionaryIncrement, RedactionLog redactionLog, Text text, AnalyzeRequest analyzeRequest) {
|
||||
|
||||
long start = System.currentTimeMillis();
|
||||
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
|
||||
|
||||
Set<Integer> sectionsToReanalyse = new HashSet<>();
|
||||
Map<Integer, Set<Image>> imageEntries = new HashMap<>();
|
||||
for (RedactionLogEntry entry : redactionLog.getRedactionLogEntry()) {
|
||||
if (entry.isLocalManualRedaction() || relevantManuallyModifiedAnnotationIds.contains(entry.getId())) {
|
||||
sectionsToReanalyse.add(entry.getSectionNumber());
|
||||
}
|
||||
if (entry.isImage()) {
|
||||
imageEntries.computeIfAbsent(entry.getSectionNumber(), x -> new HashSet<>()).add(convert(entry));
|
||||
}
|
||||
}
|
||||
|
||||
var dictionaryIncrementsSearch = new SearchImplementation(dictionaryIncrement.getValues().stream().map(DictionaryIncrementValue::getValue).collect(Collectors.toList()),
|
||||
true);
|
||||
|
||||
for (SectionText sectionText : text.getSectionTexts()) {
|
||||
|
||||
if (EntitySearchUtils.sectionContainsAny(sectionText.getText(), dictionaryIncrementsSearch)) {
|
||||
sectionsToReanalyse.add(sectionText.getSectionNumber());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
log.info("Should reanalyze {} sections for request: {}, took: {}", sectionsToReanalyse.size(), analyzeRequest, System.currentTimeMillis() - start);
|
||||
|
||||
return sectionsToReanalyse;
|
||||
}
|
||||
|
||||
|
||||
private AnalyzeResult finalizeAnalysis(AnalyzeRequest analyzeRequest,
|
||||
long startTime,
|
||||
RedactionLog redactionLog,
|
||||
Text text,
|
||||
DictionaryVersion dictionaryVersion,
|
||||
boolean isReanalysis,
|
||||
Set<FileAttribute> addedFileAttributes) {
|
||||
Set<FileAttribute> addedFileAttributes
|
||||
) {
|
||||
|
||||
redactionLog.setDictionaryVersion(dictionaryVersion.getDossierTemplateVersion());
|
||||
redactionLog.setDossierDictionaryVersion(dictionaryVersion.getDossierVersion());
|
||||
@@ -296,8 +295,6 @@ public class AnalyzeService {
|
||||
|
||||
long duration = System.currentTimeMillis() - startTime;
|
||||
|
||||
analyzeCounter.increase(text.getNumberOfPages(), duration);
|
||||
|
||||
return AnalyzeResult.builder()
|
||||
.dossierId(analyzeRequest.getDossierId())
|
||||
.fileId(analyzeRequest.getFileId())
|
||||
@@ -317,12 +314,41 @@ public class AnalyzeService {
|
||||
}
|
||||
|
||||
|
||||
private Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
|
||||
|
||||
if (manualRedactions == null) {
|
||||
return new HashSet<>();
|
||||
}
|
||||
|
||||
return Stream.concat(manualRedactions.getResizeRedactions().stream().map(ManualResizeRedaction::getAnnotationId),
|
||||
Stream.concat(manualRedactions.getLegalBasisChanges().stream().map(ManualLegalBasisChange::getAnnotationId),
|
||||
Stream.concat(manualRedactions.getImageRecategorization().stream().map(ManualImageRecategorization::getAnnotationId),
|
||||
Stream.concat(manualRedactions.getIdsToRemove().stream().map(IdRemoval::getAnnotationId),
|
||||
manualRedactions.getForceRedactions().stream().map(ManualForceRedaction::getAnnotationId))))).collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
|
||||
public List<RedactionLogLegalBasis> convert(List<LegalBasis> legalBasis) {
|
||||
|
||||
return legalBasis.stream().map(l -> new RedactionLogLegalBasis(l.getName(), l.getDescription(), l.getReason())).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
public Image convert(RedactionLogEntry entry) {
|
||||
|
||||
Rectangle position = entry.getPositions().get(0);
|
||||
|
||||
return Image.builder()
|
||||
.type(entry.getType())
|
||||
.position(new RedRectangle2D(position.getTopLeft().getX(), position.getTopLeft().getY(), position.getWidth(), position.getHeight()))
|
||||
.sectionNumber(entry.getSectionNumber())
|
||||
.section(entry.getSection())
|
||||
.page(position.getPage())
|
||||
.hasTransparency(entry.isImageHasTransparency())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private void excludeExcludedPages(RedactionLog redactionLog, Set<Integer> excludedPages) {
|
||||
|
||||
if (excludedPages != null && !excludedPages.isEmpty()) {
|
||||
@@ -347,21 +373,4 @@ public class AnalyzeService {
|
||||
return SimplifiedText.builder().numberOfPages(numberOfPages).sectionTexts(sectionTexts).build();
|
||||
}
|
||||
|
||||
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@Getter
|
||||
private static final class PagewiseCounter {
|
||||
|
||||
long totalPageCount;
|
||||
long totalTimeMillis;
|
||||
|
||||
|
||||
public void increase(long pageCount, long durationMillis) {
|
||||
|
||||
totalPageCount += pageCount;
|
||||
totalTimeMillis += durationMillis;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+2
-2
@@ -1,7 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.configuration.Colors;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.type.DictionaryEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.DictionaryEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.*;
|
||||
|
||||
+33
-15
@@ -1,11 +1,12 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
|
||||
|
||||
import io.micrometer.core.annotation.Timed;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.InputStream;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.apache.commons.lang3.StringUtils;
|
||||
import org.kie.api.KieServices;
|
||||
@@ -14,13 +15,19 @@ import org.kie.api.builder.KieFileSystem;
|
||||
import org.kie.api.builder.KieModule;
|
||||
import org.kie.api.runtime.KieContainer;
|
||||
import org.kie.api.runtime.KieSession;
|
||||
import org.kie.api.runtime.rule.QueryResults;
|
||||
import org.kie.api.runtime.rule.QueryResultsRow;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.InputStream;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
|
||||
|
||||
import io.micrometer.core.annotation.Timed;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@@ -45,18 +52,29 @@ public class DroolsExecutionService {
|
||||
|
||||
|
||||
@Timed("redactmanager_executeRules")
|
||||
public Section executeRules(KieContainer kieContainer, Section section) {
|
||||
public List<Entity> executeRules(KieContainer kieContainer, List<Section> sections, List<Paragraph> paragraphs, Dictionary dictionary) {
|
||||
|
||||
KieSession kieSession = kieContainer.newKieSession();
|
||||
kieSession.setGlobal("section", section);
|
||||
kieSession.insert(section);
|
||||
kieSession.setGlobal("dictionary", dictionary);
|
||||
sections.forEach(kieSession::insert);
|
||||
paragraphs.forEach(kieSession::insert);
|
||||
kieSession.fireAllRules();
|
||||
List<Entity> entities = getEntities(kieSession);
|
||||
kieSession.dispose();
|
||||
|
||||
return section;
|
||||
return entities;
|
||||
|
||||
}
|
||||
|
||||
public List<Entity> getEntities(KieSession ks) {
|
||||
List<Entity> entities = new LinkedList<>();
|
||||
QueryResults entitiesResult = ks.getQueryResults("getEntities");
|
||||
for (QueryResultsRow resultsRow : entitiesResult) {
|
||||
entities.add((Entity) resultsRow.get("$result"));
|
||||
}
|
||||
return entities;
|
||||
}
|
||||
|
||||
|
||||
public KieContainer updateRules(String dossierTemplateId) {
|
||||
|
||||
|
||||
+172
-139
@@ -1,4 +1,4 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service.entityredaction;
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
@@ -7,46 +7,45 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.apache.commons.lang3.StringUtils;
|
||||
import org.kie.api.runtime.KieContainer;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.FileAttribute;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualImageRecategorization;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.Engine;
|
||||
import com.iqser.red.service.redaction.v1.model.FileAttribute;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.SectionText;
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.NerEntities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.FindEntitiesResult;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PageEntities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SectionSearchableTextPair;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SurroundingWordsService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.FindEntityDetails;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import io.micrometer.core.annotation.Timed;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class EntityRedactionService {
|
||||
|
||||
DroolsExecutionService droolsExecutionService;
|
||||
SurroundingWordsService surroundingWordsService;
|
||||
EntityFinder entityFinder;
|
||||
private final RedactionServiceSettings redactionServiceSettings;
|
||||
private final DroolsExecutionService droolsExecutionService;
|
||||
private final SearchTextWithTextPositionFactory searchTextWithTextPositionFactory;
|
||||
|
||||
|
||||
public PageEntities findEntities(Dictionary dictionary, List<SectionText> sectionTexts, KieContainer kieContainer, AnalyzeRequest analyzeRequest, NerEntities nerEntities) {
|
||||
@@ -93,47 +92,89 @@ public class EntityRedactionService {
|
||||
Map<Integer, Set<Image>> imagesPerPage,
|
||||
NerEntities nerEntities) {
|
||||
|
||||
List<SectionSearchableTextPair> sectionSearchableTextPairs = extractSearchableTextPairs(reanalysisSections,
|
||||
dictionary,
|
||||
analyzeRequest,
|
||||
local,
|
||||
hintsPerSectionNumber,
|
||||
nerEntities);
|
||||
List<Section> sections = new ArrayList<>(reanalysisSections.size());
|
||||
|
||||
for (SectionText reanalysisSection : reanalysisSections) {
|
||||
|
||||
Set<Entity> nerFound = new HashSet<>();
|
||||
if (!local) {
|
||||
nerFound.addAll(getNerValues(reanalysisSection.getSectionNumber(), nerEntities, reanalysisSection.getCellStarts(), reanalysisSection.getHeadline()));
|
||||
}
|
||||
|
||||
log.debug("Section {}, Images: {}", reanalysisSection.getSectionNumber(), reanalysisSection.getImages());
|
||||
sections.add(Section.builder()
|
||||
.isLocal(false)
|
||||
.dictionaryTypes(dictionary.getTypes())
|
||||
.nerEntities(nerFound)
|
||||
.text(reanalysisSection.getSearchableText().getAsStringWithLinebreaks())
|
||||
.searchText(reanalysisSection.getSearchableText().toString())
|
||||
.headline(reanalysisSection.getHeadline())
|
||||
.sectionNumber(reanalysisSection.getSectionNumber())
|
||||
.tabularData(reanalysisSection.getTabularData())
|
||||
.searchableText(reanalysisSection.getSearchableText())
|
||||
.dictionary(dictionary)
|
||||
.images(reanalysisSection.getImages())
|
||||
.sectionAreas(reanalysisSection.getSectionAreas())
|
||||
.fileAttributes(analyzeRequest.getFileAttributes())
|
||||
.manualRedactions(analyzeRequest.getManualRedactions())
|
||||
.isInTable(reanalysisSection.isTable())
|
||||
.redactionServiceSettings(redactionServiceSettings)
|
||||
.cellStarts(reanalysisSection.getCellStarts())
|
||||
.build());
|
||||
|
||||
}
|
||||
|
||||
Set<FileAttribute> addedFileAttributes = new HashSet<>();
|
||||
Set<Entity> entities = new HashSet<>();
|
||||
sectionSearchableTextPairs.forEach(sectionSearchableTextPair -> {
|
||||
sections.forEach(section -> {
|
||||
|
||||
if (!addedFileAttributes.isEmpty()) {
|
||||
//Section.Builder provides immutable list.
|
||||
List<FileAttribute> mergedFileAttributes = new ArrayList<>();
|
||||
mergedFileAttributes.addAll(sectionSearchableTextPair.getSection().getAddedFileAttributes());
|
||||
mergedFileAttributes.addAll(section.getAddedFileAttributes());
|
||||
mergedFileAttributes.addAll(addedFileAttributes);
|
||||
sectionSearchableTextPair.getSection().setFileAttributes(mergedFileAttributes);
|
||||
section.setFileAttributes(mergedFileAttributes);
|
||||
}
|
||||
});
|
||||
|
||||
Section analysedSection = droolsExecutionService.executeRules(kieContainer, sectionSearchableTextPair.getSection());
|
||||
List<Paragraph> paragraphs = new ArrayList<>();
|
||||
for (var reanalysisSection : reanalysisSections) {
|
||||
for (int i = 0; i < reanalysisSection.getTextBlocks().size(); ++i) {
|
||||
paragraphs.add(Paragraph.builder()
|
||||
.paragraphNumber(i)
|
||||
.sectionNumber(reanalysisSection.getSectionNumber())
|
||||
.searchTextToTextPosition(searchTextWithTextPositionFactory.buildSearchTextToTextPositionModel(reanalysisSection.getTextBlocks().get(i).getSequences()))
|
||||
.build());
|
||||
}
|
||||
}
|
||||
|
||||
List<Entity> entitiesList = droolsExecutionService.executeRules(kieContainer, sections, paragraphs, dictionary);
|
||||
Set<Entity> entities = new HashSet<>(entitiesList);
|
||||
|
||||
sections.forEach(analysedSection -> {
|
||||
addedFileAttributes.addAll(analysedSection.getAddedFileAttributes());
|
||||
|
||||
EntitySearchUtils.removeEntitiesContainedInLarger(analysedSection.getEntities());
|
||||
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
|
||||
|
||||
var entriesWithoutSurroundingText = analysedSection.getEntities()
|
||||
.stream()
|
||||
var entriesWithoutSurroundingText = entities.stream()
|
||||
.filter(e -> e.getSectionNumber() == analysedSection.getSectionNumber())
|
||||
.filter(e -> e.getTextAfter() == null && e.getTextBefore() == null)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
if (sectionSearchableTextPair.getCellStarts() != null && !sectionSearchableTextPair.getCellStarts().isEmpty()) {
|
||||
surroundingWordsService.addSurroundingText(entriesWithoutSurroundingText,
|
||||
sectionSearchableTextPair.getSearchableText(),
|
||||
if (analysedSection.getCellStarts() != null && !analysedSection.getCellStarts().isEmpty()) {
|
||||
SurroundingWordsService.addSurroundingText(entriesWithoutSurroundingText,
|
||||
analysedSection.getSearchableText(),
|
||||
dictionary,
|
||||
sectionSearchableTextPair.getCellStarts());
|
||||
analysedSection.getCellStarts(),
|
||||
redactionServiceSettings.getSurroundingWordsOffsetWindow(),
|
||||
redactionServiceSettings.getNumberOfSurroundingWords());
|
||||
} else {
|
||||
surroundingWordsService.addSurroundingText(entriesWithoutSurroundingText, sectionSearchableTextPair.getSearchableText(), dictionary);
|
||||
SurroundingWordsService.addSurroundingText(entriesWithoutSurroundingText,
|
||||
analysedSection.getSearchableText(),
|
||||
dictionary,
|
||||
redactionServiceSettings.getSurroundingWordsOffsetWindow(),
|
||||
redactionServiceSettings.getNumberOfSurroundingWords());
|
||||
}
|
||||
|
||||
entities.addAll(analysedSection.getEntities());
|
||||
|
||||
if (!local) {
|
||||
for (Image image : analysedSection.getImages()) {
|
||||
imagesPerPage.computeIfAbsent(image.getPage(), (a) -> new HashSet<>()).add(image);
|
||||
@@ -147,106 +188,6 @@ public class EntityRedactionService {
|
||||
}
|
||||
|
||||
|
||||
private List<SectionSearchableTextPair> extractSearchableTextPairs(List<SectionText> reanalysisSections,
|
||||
Dictionary dictionary,
|
||||
AnalyzeRequest analyzeRequest,
|
||||
boolean local,
|
||||
Map<Integer, Set<Entity>> hintsPerSectionNumber,
|
||||
NerEntities nerEntities) {
|
||||
|
||||
return reanalysisSections.stream().map(reanalysisSection -> {
|
||||
|
||||
Entities entities = entityFinder.findEntities(reanalysisSection.getSearchableText(),
|
||||
reanalysisSection.getHeadline(),
|
||||
reanalysisSection.getSectionNumber(),
|
||||
dictionary,
|
||||
local,
|
||||
nerEntities,
|
||||
reanalysisSection.getCellStarts(),
|
||||
analyzeRequest.getManualRedactions());
|
||||
|
||||
if (reanalysisSection.getCellStarts() != null && !reanalysisSection.getCellStarts().isEmpty()) {
|
||||
surroundingWordsService.addSurroundingText(entities.getEntities(), reanalysisSection.getSearchableText(), dictionary, reanalysisSection.getCellStarts());
|
||||
} else {
|
||||
surroundingWordsService.addSurroundingText(entities.getEntities(), reanalysisSection.getSearchableText(), dictionary);
|
||||
}
|
||||
|
||||
if (!local && analyzeRequest.getManualRedactions() != null) {
|
||||
|
||||
var approvedForceRedactions = analyzeRequest.getManualRedactions()
|
||||
.getForceRedactions()
|
||||
.stream()
|
||||
.filter(fr -> fr.getStatus() == AnnotationStatus.APPROVED)
|
||||
.filter(fr -> fr.getRequestDate() != null)
|
||||
.collect(Collectors.toList());
|
||||
// only approved id removals, that haven't been forced back afterwards
|
||||
var idsToRemove = analyzeRequest.getManualRedactions()
|
||||
.getIdsToRemove()
|
||||
.stream()
|
||||
.filter(idr -> idr.getStatus() == AnnotationStatus.APPROVED && !idr.isRemoveFromDictionary())
|
||||
.filter(idr -> idr.getRequestDate() != null)
|
||||
.filter(idr -> approvedForceRedactions.stream()
|
||||
.noneMatch(forceRedact -> forceRedact.getAnnotationId().equals(idr.getAnnotationId()) && forceRedact.getRequestDate()
|
||||
.isAfter(idr.getRequestDate())))
|
||||
.map(IdRemoval::getAnnotationId)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
if (reanalysisSection.getImages() != null && !reanalysisSection.getImages().isEmpty() && analyzeRequest.getManualRedactions().getImageRecategorization() != null) {
|
||||
for (Image image : reanalysisSection.getImages()) {
|
||||
String imageId = IdBuilder.buildId(image.getPosition(), image.getPage());
|
||||
for (ManualImageRecategorization imageRecategorization : analyzeRequest.getManualRedactions().getImageRecategorization()) {
|
||||
if (imageRecategorization.getStatus().equals(AnnotationStatus.APPROVED) && imageRecategorization.getAnnotationId().equals(imageId)) {
|
||||
image.setType(imageRecategorization.getType());
|
||||
}
|
||||
}
|
||||
if (idsToRemove.contains(imageId)) {
|
||||
image.setIgnored(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
entities.getEntities().forEach(entity -> entity.getPositionSequences().forEach(ps -> {
|
||||
if (idsToRemove.contains(ps.getId())) {
|
||||
entity.setIgnored(true);
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
log.debug("Section {}, Images: {}", reanalysisSection.getSectionNumber(), reanalysisSection.getImages());
|
||||
|
||||
return toSectionSearchableTextPair(dictionary, analyzeRequest, hintsPerSectionNumber, reanalysisSection, entities);
|
||||
}).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
private SectionSearchableTextPair toSectionSearchableTextPair(Dictionary dictionary,
|
||||
AnalyzeRequest analyzeRequest,
|
||||
Map<Integer, Set<Entity>> hintsPerSectionNumber,
|
||||
SectionText reanalysisSection,
|
||||
Entities entities) {
|
||||
|
||||
return new SectionSearchableTextPair(Section.builder()
|
||||
.isLocal(false)
|
||||
.dictionaryTypes(dictionary.getTypes())
|
||||
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(reanalysisSection.getSectionNumber()) ? Stream.concat(entities.getEntities().stream(),
|
||||
hintsPerSectionNumber.get(reanalysisSection.getSectionNumber()).stream()).collect(Collectors.toSet()) : entities.getEntities())
|
||||
.nerEntities(entities.getNerEntities())
|
||||
.text(reanalysisSection.getSearchableText().getAsStringWithLinebreaks())
|
||||
.searchText(reanalysisSection.getSearchableText().toString())
|
||||
.headline(reanalysisSection.getHeadline())
|
||||
.sectionNumber(reanalysisSection.getSectionNumber())
|
||||
.tabularData(reanalysisSection.getTabularData())
|
||||
.searchableText(reanalysisSection.getSearchableText())
|
||||
.dictionary(dictionary)
|
||||
.images(reanalysisSection.getImages())
|
||||
.sectionAreas(reanalysisSection.getSectionAreas())
|
||||
.fileAttributes(analyzeRequest.getFileAttributes())
|
||||
.manualRedactions(analyzeRequest.getManualRedactions())
|
||||
.isInTable(reanalysisSection.isTable())
|
||||
.build(), reanalysisSection.getSearchableText(), reanalysisSection.getCellStarts());
|
||||
}
|
||||
|
||||
|
||||
private Map<Integer, List<Entity>> convertToEntitiesPerPage(Set<Entity> entities) {
|
||||
|
||||
Map<Integer, List<Entity>> entitiesPerPage = new HashMap<>();
|
||||
@@ -266,6 +207,7 @@ public class EntityRedactionService {
|
||||
entity.getHeadline(),
|
||||
entity.getMatchedRule(),
|
||||
entity.getSectionNumber(),
|
||||
-1,
|
||||
entity.getLegalBasis(),
|
||||
entity.isDictionaryEntry(),
|
||||
entity.getTextBefore(),
|
||||
@@ -285,7 +227,7 @@ public class EntityRedactionService {
|
||||
private Map<Integer, Set<Entity>> getHintsPerSection(Set<Entity> entities, Dictionary dictionary) {
|
||||
|
||||
Map<Integer, Set<Entity>> hintsPerSectionNumber = new HashMap<>();
|
||||
entities.forEach(entity -> {
|
||||
entities.stream().forEach(entity -> {
|
||||
if (dictionary.isHint(entity.getType()) && entity.isDictionaryEntry()) {
|
||||
hintsPerSectionNumber.computeIfAbsent(entity.getSectionNumber(), (x) -> new HashSet<>()).add(entity);
|
||||
}
|
||||
@@ -310,4 +252,95 @@ public class EntityRedactionService {
|
||||
}));
|
||||
}
|
||||
|
||||
|
||||
@Timed("redactmanager_findEntities")
|
||||
private Entities findEntities(SearchableText searchableText,
|
||||
String headline,
|
||||
int sectionNumber,
|
||||
Dictionary dictionary,
|
||||
boolean local,
|
||||
NerEntities nerEntities,
|
||||
List<Integer> cellStarts,
|
||||
ManualRedactions manualRedactions) {
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
String searchableString = searchableText.asString();
|
||||
|
||||
if (StringUtils.isEmpty(searchableString)) {
|
||||
return new Entities(new HashSet<>(), new HashSet<>());
|
||||
}
|
||||
|
||||
String lowercaseInputString = searchableString.toLowerCase();
|
||||
for (DictionaryModel model : dictionary.getDictionaryModels()) {
|
||||
|
||||
var searchImplementation = local ? model.getLocalSearch() : model.getEntriesSearch();
|
||||
var entities = EntitySearchUtils.findEntities(model.isCaseInsensitive() ? lowercaseInputString : searchableString,
|
||||
searchImplementation,
|
||||
model,
|
||||
new FindEntityDetails(model.getType(),
|
||||
headline,
|
||||
sectionNumber,
|
||||
!local,
|
||||
model.isDossierDictionary(),
|
||||
local ? Engine.RULE : Engine.DICTIONARY,
|
||||
local ? EntityType.RECOMMENDATION : EntityType.ENTITY));
|
||||
|
||||
EntitySearchUtils.addOrAddEngine(found, entities);
|
||||
}
|
||||
|
||||
Set<Entity> nerFound = new HashSet<>();
|
||||
if (!local) {
|
||||
nerFound.addAll(getNerValues(sectionNumber, nerEntities, cellStarts, headline));
|
||||
}
|
||||
|
||||
var cleared = EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary, manualRedactions);
|
||||
return new Entities(cleared.stream().filter(e -> !e.isFalsePositive()).collect(Collectors.toSet()), nerFound);
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> getNerValues(int sectionNumber, NerEntities nerEntities, List<Integer> cellStarts, String headline) {
|
||||
|
||||
Set<Entity> entities = new HashSet<>();
|
||||
|
||||
if (redactionServiceSettings.isNerServiceEnabled() && nerEntities.getData().containsKey(sectionNumber)) {
|
||||
nerEntities.getData().get(sectionNumber).forEach(res -> {
|
||||
if (cellStarts == null || cellStarts.isEmpty()) {
|
||||
entities.add(new Entity(res.getValue(),
|
||||
res.getType(),
|
||||
res.getStartOffset(),
|
||||
res.getEndOffset(),
|
||||
headline,
|
||||
sectionNumber,
|
||||
-1,
|
||||
false,
|
||||
false,
|
||||
Engine.NER,
|
||||
EntityType.RECOMMENDATION));
|
||||
} else {
|
||||
boolean intersectsCellStart = false;
|
||||
for (Integer cellStart : cellStarts) {
|
||||
if (res.getStartOffset() < cellStart && cellStart < res.getEndOffset()) {
|
||||
intersectsCellStart = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!intersectsCellStart) {
|
||||
entities.add(new Entity(res.getValue(),
|
||||
res.getType(),
|
||||
res.getStartOffset(),
|
||||
res.getEndOffset(),
|
||||
headline,
|
||||
sectionNumber,
|
||||
-1,
|
||||
false,
|
||||
false,
|
||||
Engine.NER,
|
||||
EntityType.RECOMMENDATION));
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
return entities;
|
||||
}
|
||||
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user