Compare commits
28
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
97291ffcae | ||
|
|
9a9867e739 | ||
|
|
aa20a5c201 | ||
|
|
600394c366 | ||
|
|
8fd9855b2c | ||
|
|
2a8f1ecdd2 | ||
|
|
6202e95ad8 | ||
|
|
b1fee50718 | ||
|
|
3a59bff246 | ||
|
|
57683d49b6 | ||
|
|
0fd01fa1c3 | ||
|
|
6d66e5dac3 | ||
|
|
446c3e56a8 | ||
|
|
daab1f1b0a | ||
|
|
af70705250 | ||
|
|
f29f8626e9 | ||
|
|
940e6e7b42 | ||
|
|
c08ad75893 | ||
|
|
b8d9b101c9 | ||
|
|
a1ea354225 | ||
|
|
e9f516580e | ||
|
|
b006b8c1ab | ||
|
|
78912adeb6 | ||
|
|
9bbff158c0 | ||
|
|
6e0b343de3 | ||
|
|
5bc9b60623 | ||
|
|
e8b32681ce | ||
|
|
d29b6bf816 |
+11
-10
@@ -1,35 +1,36 @@
|
||||
variables:
|
||||
SONAR_PROJECT_KEY: 'RED_redaction-service'
|
||||
GIT_SUBMODULE_STRATEGY: recursive
|
||||
GIT_SUBMODULE_FORCE_HTTPS: "true"
|
||||
include:
|
||||
- project: 'gitlab/gitlab'
|
||||
ref: 'main'
|
||||
file: 'ci-templates/gradle_java.yml'
|
||||
|
||||
deploy JavaDoc:
|
||||
deploy:
|
||||
stage: deploy
|
||||
tags:
|
||||
- dind
|
||||
script:
|
||||
- echo "Building JavaDoc with gradle version ${BUILDVERSION}"
|
||||
- echo "Building with gradle version ${BUILDVERSION}"
|
||||
- gradle -Pversion=${BUILDVERSION} publish
|
||||
- gradle bootBuildImage --publishImage -PbuildbootDockerHostNetwork=true -Pversion=${BUILDVERSION}
|
||||
- echo "BUILDVERSION=$BUILDVERSION" >> version.env
|
||||
artifacts:
|
||||
reports:
|
||||
dotenv: version.env
|
||||
rules:
|
||||
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
|
||||
- if: $CI_COMMIT_BRANCH =~ /^release/
|
||||
- if: $CI_COMMIT_TAG
|
||||
|
||||
generateJavaDoc:
|
||||
pages:
|
||||
stage: build
|
||||
tags:
|
||||
- dind
|
||||
script:
|
||||
- echo "Generating Javadoc..."
|
||||
- gradle generateJavaDoc -PjavadocDestinationDir="javadoc"
|
||||
- mkdir public
|
||||
- mv redaction-service-v1/redaction-service-server-v1/javadoc/* public/
|
||||
artifacts:
|
||||
paths:
|
||||
- redaction-service-v1/redaction-service-server-v1/javadoc/*
|
||||
rules:
|
||||
- if: $CI_COMMIT_BRANCH == $CI_DEFAULT_BRANCH
|
||||
- if: $CI_COMMIT_BRANCH =~ /^release/
|
||||
- if: $CI_COMMIT_TAG
|
||||
- public
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
[submodule "redaction-service-v1/redaction-service-server-v1/src/test/resources/files/syngenta"]
|
||||
path = redaction-service-v1/redaction-service-server-v1/src/test/resources/files/syngenta
|
||||
url = ssh://git@git.knecon.com:22222/fforesight/documents/syngenta.git
|
||||
update = merge
|
||||
[submodule "redaction-service-v1/redaction-service-server-v1/src/test/resources/files/basf"]
|
||||
path = redaction-service-v1/redaction-service-server-v1/src/test/resources/files/basf
|
||||
url = ssh://git@git.knecon.com:22222/fforesight/documents/basf.git
|
||||
update = merge
|
||||
@@ -7,7 +7,7 @@ description = "redaction-service-api-v1"
|
||||
|
||||
dependencies {
|
||||
implementation("org.springframework:spring-web:6.0.12")
|
||||
implementation("com.iqser.red.service:persistence-service-internal-api-v1:2.395.0")
|
||||
implementation("com.iqser.red.service:persistence-service-internal-api-v1:2.351.0")
|
||||
}
|
||||
|
||||
publishing {
|
||||
|
||||
-1
@@ -15,5 +15,4 @@ public class AnalyzeResponse {
|
||||
|
||||
private String fileId;
|
||||
private List<UnprocessedManualEntity> unprocessedManualEntities;
|
||||
|
||||
}
|
||||
|
||||
-27
@@ -1,27 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
@Data
|
||||
@SuperBuilder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class DroolsBlacklistErrorMessage extends DroolsValidationMessage {
|
||||
|
||||
List<String> blacklistedKeywords;
|
||||
|
||||
public String getMessage() {
|
||||
return String.format("Blacklisted keywords found in this rule: %s", String.join(", ", blacklistedKeywords));
|
||||
}
|
||||
|
||||
}
|
||||
+4
-6
@@ -4,19 +4,17 @@ import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
@Data
|
||||
@SuperBuilder
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class DroolsSyntaxDeprecatedWarnings extends DroolsValidationMessage {
|
||||
public class DroolsSyntaxDeprecatedWarnings {
|
||||
|
||||
Integer line;
|
||||
Integer column;
|
||||
String message;
|
||||
|
||||
}
|
||||
|
||||
+4
-6
@@ -4,19 +4,17 @@ import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
@Data
|
||||
@SuperBuilder
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class DroolsSyntaxErrorMessage extends DroolsValidationMessage {
|
||||
public class DroolsSyntaxErrorMessage {
|
||||
|
||||
Integer line;
|
||||
Integer column;
|
||||
String message;
|
||||
|
||||
}
|
||||
|
||||
+36
@@ -0,0 +1,36 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class DroolsSyntaxValidation {
|
||||
|
||||
@Builder.Default
|
||||
List<DroolsSyntaxErrorMessage> droolsSyntaxErrorMessages = new LinkedList<>();
|
||||
@Builder.Default
|
||||
List<DroolsSyntaxDeprecatedWarnings> droolsSyntaxDeprecatedWarnings = new LinkedList<>();
|
||||
|
||||
|
||||
public void addErrorMessage(int line, int column, String message) {
|
||||
|
||||
getDroolsSyntaxErrorMessages().add(DroolsSyntaxErrorMessage.builder().line(line).column(column).message(message).build());
|
||||
}
|
||||
|
||||
public boolean isCompiled() {
|
||||
|
||||
return droolsSyntaxErrorMessages.isEmpty();
|
||||
}
|
||||
|
||||
}
|
||||
-39
@@ -1,39 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class DroolsValidation {
|
||||
|
||||
@Builder.Default
|
||||
List<DroolsSyntaxErrorMessage> syntaxErrorMessages = new ArrayList<>();
|
||||
@Builder.Default
|
||||
List<DroolsSyntaxDeprecatedWarnings> deprecatedWarnings = new ArrayList<>();
|
||||
@Builder.Default
|
||||
List<DroolsBlacklistErrorMessage> blacklistErrorMessages = new ArrayList<>();
|
||||
|
||||
|
||||
public void addErrorMessage(int line, int column, String message) {
|
||||
|
||||
getSyntaxErrorMessages().add(DroolsSyntaxErrorMessage.builder().line(line).column(column).message(message).build());
|
||||
}
|
||||
|
||||
|
||||
public boolean isCompiled() {
|
||||
|
||||
return syntaxErrorMessages.isEmpty() && blacklistErrorMessages.isEmpty();
|
||||
}
|
||||
|
||||
}
|
||||
-21
@@ -1,21 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
@Data
|
||||
@SuperBuilder
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@EqualsAndHashCode
|
||||
public class DroolsValidationMessage {
|
||||
|
||||
Integer line;
|
||||
Integer column;
|
||||
}
|
||||
-13
@@ -1,15 +1,11 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.NonNull;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@@ -17,18 +13,9 @@ import lombok.NonNull;
|
||||
@AllArgsConstructor
|
||||
public class MigrationRequest {
|
||||
|
||||
@NonNull
|
||||
String dossierTemplateId;
|
||||
@NonNull
|
||||
String dossierId;
|
||||
@NonNull
|
||||
String fileId;
|
||||
|
||||
boolean fileIsApproved;
|
||||
@NonNull
|
||||
ManualRedactions manualRedactions;
|
||||
@NonNull
|
||||
@Builder.Default
|
||||
Set<String> entitiesWithComments = Collections.emptySet();
|
||||
|
||||
}
|
||||
|
||||
-1
@@ -24,5 +24,4 @@ public class UnprocessedManualEntity {
|
||||
private String section;
|
||||
@Builder.Default
|
||||
private List<Position> positions = new ArrayList<>();
|
||||
|
||||
}
|
||||
|
||||
+2
-2
@@ -4,12 +4,12 @@ import org.springframework.http.MediaType;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
|
||||
|
||||
public interface RedactionResource {
|
||||
|
||||
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
DroolsValidation testRules(@RequestBody RuleValidationModel rulesValidationModel);
|
||||
DroolsSyntaxValidation testRules(@RequestBody RuleValidationModel rulesValidationModel);
|
||||
|
||||
}
|
||||
|
||||
@@ -12,14 +12,12 @@ plugins {
|
||||
description = "redaction-service-server-v1"
|
||||
|
||||
|
||||
val layoutParserVersion = "0.107.0"
|
||||
val layoutParserVersion = "0.94.0"
|
||||
val jacksonVersion = "2.15.2"
|
||||
val droolsVersion = "9.44.0.Final"
|
||||
val pdfBoxVersion = "3.0.0"
|
||||
val persistenceServiceVersion = "2.395.0"
|
||||
val persistenceServiceVersion = "2.360.0"
|
||||
val springBootStarterVersion = "3.1.5"
|
||||
val springCloudVersion = "4.0.4"
|
||||
val testContainersVersion = "1.19.7"
|
||||
|
||||
configurations {
|
||||
all {
|
||||
@@ -33,7 +31,6 @@ dependencies {
|
||||
|
||||
implementation(project(":redaction-service-api-v1")) { exclude(group = "com.iqser.red.service", module = "persistence-service-internal-api-v1") }
|
||||
implementation("com.iqser.red.service:persistence-service-internal-api-v1:${persistenceServiceVersion}") { exclude(group = "org.springframework.boot") }
|
||||
implementation("com.iqser.red.service:persistence-service-shared-mongo-v1:${persistenceServiceVersion}")
|
||||
implementation("com.knecon.fforesight:layoutparser-service-internal-api:${layoutParserVersion}")
|
||||
|
||||
implementation("com.iqser.red.commons:spring-commons:6.2.0")
|
||||
@@ -41,7 +38,7 @@ dependencies {
|
||||
|
||||
implementation("com.iqser.red.commons:dictionary-merge-commons:1.5.0")
|
||||
implementation("com.iqser.red.commons:storage-commons:2.45.0")
|
||||
implementation("com.knecon.fforesight:tenant-commons:0.24.0")
|
||||
implementation("com.knecon.fforesight:tenant-commons:0.21.0")
|
||||
implementation("com.knecon.fforesight:tracing-commons:0.5.0")
|
||||
|
||||
implementation("com.fasterxml.jackson.module:jackson-module-afterburner:${jacksonVersion}")
|
||||
@@ -55,7 +52,7 @@ dependencies {
|
||||
|
||||
implementation("org.locationtech.jts:jts-core:1.19.0")
|
||||
|
||||
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:${springCloudVersion}")
|
||||
implementation("org.springframework.cloud:spring-cloud-starter-openfeign:4.0.4")
|
||||
implementation("org.springframework.boot:spring-boot-starter-amqp:${springBootStarterVersion}")
|
||||
implementation("org.springframework.boot:spring-boot-starter-cache:${springBootStarterVersion}")
|
||||
implementation("org.springframework.boot:spring-boot-starter-data-redis:${springBootStarterVersion}")
|
||||
@@ -69,9 +66,6 @@ dependencies {
|
||||
testImplementation("org.apache.pdfbox:pdfbox:${pdfBoxVersion}")
|
||||
testImplementation("org.apache.pdfbox:pdfbox-tools:${pdfBoxVersion}")
|
||||
|
||||
testImplementation("org.testcontainers:testcontainers:${testContainersVersion}")
|
||||
testImplementation("org.testcontainers:junit-jupiter:${testContainersVersion}")
|
||||
|
||||
testImplementation("org.springframework.boot:spring-boot-starter-test:${springBootStarterVersion}")
|
||||
testImplementation("com.knecon.fforesight:viewer-doc-processor:${layoutParserVersion}")
|
||||
testImplementation("com.knecon.fforesight:layoutparser-service-processor:${layoutParserVersion}") {
|
||||
@@ -82,12 +76,6 @@ dependencies {
|
||||
}
|
||||
}
|
||||
|
||||
dependencyManagement {
|
||||
imports {
|
||||
mavenBom("org.testcontainers:testcontainers-bom:${testContainersVersion}")
|
||||
}
|
||||
}
|
||||
|
||||
tasks.test {
|
||||
configure<JacocoTaskExtension> {
|
||||
excludes = listOf("org/drools/**/*")
|
||||
@@ -96,9 +84,6 @@ tasks.test {
|
||||
maxHeapSize = "1024m"
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
tasks.named<BootBuildImage>("bootBuildImage") {
|
||||
|
||||
environment.put("BPE_DELIM_JAVA_TOOL_OPTIONS", " ")
|
||||
@@ -127,7 +112,6 @@ tasks.named<BootBuildImage>("bootBuildImage") {
|
||||
}
|
||||
|
||||
fun parseDroolsImports(droolsFilePath: String): List<String> {
|
||||
|
||||
val imports = mutableListOf<String>()
|
||||
val importPattern = Regex("^import\\s+(com\\.iqser\\.red\\.service\\.redaction\\.v1\\.[\\w.]+);")
|
||||
val desiredPrefix = "com.iqser.red.service.redaction.v1"
|
||||
@@ -148,7 +132,6 @@ fun parseDroolsImports(droolsFilePath: String): List<String> {
|
||||
val droolsImports = parseDroolsImports("redaction-service-v1/redaction-service-server-v1/src/main/resources/drools/all_rules_documine.drl")
|
||||
|
||||
tasks.register("generateJavaDoc", Javadoc::class) {
|
||||
|
||||
dependsOn("compileJava")
|
||||
dependsOn("delombok")
|
||||
classpath = project.sourceSets["main"].runtimeClasspath
|
||||
|
||||
+3
-16
@@ -4,24 +4,16 @@ import org.springframework.boot.SpringApplication;
|
||||
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.ImportAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.SpringBootApplication;
|
||||
import org.springframework.boot.autoconfigure.data.mongo.MongoDataAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.jdbc.DataSourceAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.liquibase.LiquibaseAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.mongo.MongoAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
|
||||
import org.springframework.boot.context.properties.EnableConfigurationProperties;
|
||||
import org.springframework.cache.annotation.EnableCaching;
|
||||
import org.springframework.cloud.openfeign.EnableFeignClients;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Import;
|
||||
import org.springframework.data.mongodb.repository.config.EnableMongoRepositories;
|
||||
|
||||
import com.iqser.red.service.dictionarymerge.commons.DictionaryMergeService;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.mongo.SharedMongoAutoConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.storage.commons.StorageAutoConfiguration;
|
||||
import com.knecon.fforesight.mongo.database.commons.MongoDatabaseCommonsAutoConfiguration;
|
||||
import com.knecon.fforesight.mongo.database.commons.liquibase.EnableMongoLiquibase;
|
||||
import com.knecon.fforesight.tenantcommons.MultiTenancyAutoConfiguration;
|
||||
|
||||
import io.micrometer.core.aop.TimedAspect;
|
||||
@@ -30,13 +22,11 @@ import io.micrometer.observation.ObservationRegistry;
|
||||
import io.micrometer.observation.aop.ObservedAspect;
|
||||
|
||||
@EnableCaching
|
||||
@ImportAutoConfiguration({MultiTenancyAutoConfiguration.class, SharedMongoAutoConfiguration.class})
|
||||
@Import({MetricsConfiguration.class, StorageAutoConfiguration.class, MongoDatabaseCommonsAutoConfiguration.class})
|
||||
@ImportAutoConfiguration({MultiTenancyAutoConfiguration.class})
|
||||
@Import({MetricsConfiguration.class, StorageAutoConfiguration.class})
|
||||
@EnableFeignClients(basePackageClasses = RulesClient.class)
|
||||
@EnableConfigurationProperties(RedactionServiceSettings.class)
|
||||
@EnableMongoRepositories(basePackages = "com.iqser.red.service.persistence")
|
||||
@EnableMongoLiquibase
|
||||
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class, DataSourceAutoConfiguration.class, LiquibaseAutoConfiguration.class, MongoAutoConfiguration.class, MongoDataAutoConfiguration.class})
|
||||
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
|
||||
public class Application {
|
||||
|
||||
public static void main(String[] args) {
|
||||
@@ -45,14 +35,11 @@ public class Application {
|
||||
SpringApplication.run(Application.class, args);
|
||||
}
|
||||
|
||||
|
||||
@Bean
|
||||
public ObservedAspect observedAspect(ObservationRegistry observationRegistry) {
|
||||
|
||||
return new ObservedAspect(observationRegistry);
|
||||
}
|
||||
|
||||
|
||||
@Bean
|
||||
public TimedAspect timedAspect(MeterRegistry registry) {
|
||||
|
||||
|
||||
-1
@@ -95,7 +95,6 @@ public class DeprecatedElementsFinder {
|
||||
return this.deprecatedClasses;
|
||||
}
|
||||
|
||||
|
||||
private String getMethodSignature(Method method) {
|
||||
|
||||
String methodName = method.getName();
|
||||
|
||||
-2
@@ -30,6 +30,4 @@ public class RedactionServiceSettings {
|
||||
|
||||
private int droolsExecutionTimeoutSecs = 300;
|
||||
|
||||
private boolean ruleExecutionSecured = true;
|
||||
|
||||
}
|
||||
|
||||
+1
-1
@@ -16,7 +16,7 @@ public class RedisCachingConfiguration {
|
||||
public RedisCacheManagerBuilderCustomizer redisCacheManagerBuilderCustomizer() {
|
||||
|
||||
return (builder) -> builder.withCacheConfiguration("documentDataCache",
|
||||
RedisCacheConfiguration.defaultCacheConfig().entryTtl(Duration.ofMinutes(30)).disableCachingNullValues());
|
||||
RedisCacheConfiguration.defaultCacheConfig().entryTtl(Duration.ofMinutes(30)).disableCachingNullValues());
|
||||
}
|
||||
|
||||
|
||||
|
||||
+1
@@ -1,5 +1,6 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client.model;
|
||||
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
+5
-5
@@ -3,10 +3,10 @@ package com.iqser.red.service.redaction.v1.server.controller;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
|
||||
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
|
||||
import com.iqser.red.service.redaction.v1.server.service.drools.DroolsValidationService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.drools.DroolsSyntaxValidationService;
|
||||
import com.iqser.red.service.redaction.v1.server.utils.exception.RulesValidationException;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
@@ -17,14 +17,14 @@ import lombok.extern.slf4j.Slf4j;
|
||||
@RequiredArgsConstructor
|
||||
public class RedactionController implements RedactionResource {
|
||||
|
||||
private final DroolsValidationService droolsValidationService;
|
||||
private final DroolsSyntaxValidationService droolsSyntaxValidationService;
|
||||
|
||||
|
||||
@Override
|
||||
public DroolsValidation testRules(@RequestBody RuleValidationModel rulesValidationModel) {
|
||||
public DroolsSyntaxValidation testRules(@RequestBody RuleValidationModel rulesValidationModel) {
|
||||
|
||||
try {
|
||||
return droolsValidationService.testRules(rulesValidationModel);
|
||||
return droolsSyntaxValidationService.testRules(rulesValidationModel);
|
||||
} catch (Exception e) {
|
||||
throw new RulesValidationException("Could not test rules: " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
+1
@@ -13,6 +13,7 @@ import lombok.RequiredArgsConstructor;
|
||||
@RequiredArgsConstructor
|
||||
public class RuleBuilderController implements RuleBuilderResource {
|
||||
|
||||
|
||||
@Override
|
||||
public RuleBuilderModel getRuleBuilderModel() {
|
||||
|
||||
|
||||
+2
-12
@@ -92,16 +92,11 @@ public class LegacyRedactionLogMergeService {
|
||||
return redactionLog;
|
||||
}
|
||||
|
||||
|
||||
public long getNumberOfAffectedAnnotations(ManualRedactions manualRedactions) {
|
||||
|
||||
return createManualRedactionWrappers(manualRedactions).stream()
|
||||
.map(ManualRedactionWrapper::getId)
|
||||
.distinct()
|
||||
.count();
|
||||
return createManualRedactionWrappers(manualRedactions).stream().map(ManualRedactionWrapper::getId).distinct().count();
|
||||
}
|
||||
|
||||
|
||||
private List<ManualRedactionWrapper> createManualRedactionWrappers(ManualRedactions manualRedactions) {
|
||||
|
||||
List<ManualRedactionWrapper> manualRedactionWrappers = new ArrayList<>();
|
||||
@@ -202,12 +197,7 @@ public class LegacyRedactionLogMergeService {
|
||||
}
|
||||
|
||||
redactionLogEntry.getManualChanges()
|
||||
.add(ManualChange.from(imageRecategorization)
|
||||
.withManualRedactionType(ManualRedactionType.RECATEGORIZE)
|
||||
.withChange("type", imageRecategorization.getType())
|
||||
.withChange("section", imageRecategorization.getSection())
|
||||
.withChange("legalBasis", imageRecategorization.getLegalBasis())
|
||||
.withChange("value", imageRecategorization.getValue()));
|
||||
.add(ManualChange.from(imageRecategorization).withManualRedactionType(ManualRedactionType.RECATEGORIZE).withChange("type", imageRecategorization.getType()));
|
||||
}
|
||||
|
||||
|
||||
|
||||
+2
-7
@@ -21,9 +21,7 @@ public class LegacyVersion0MigrationService {
|
||||
public RedactionLog mergeDuplicateAnnotationIds(RedactionLog redactionLog) {
|
||||
|
||||
List<RedactionLogEntry> mergedEntries = new LinkedList<>();
|
||||
Map<String, List<RedactionLogEntry>> entriesById = redactionLog.getRedactionLogEntry()
|
||||
.stream()
|
||||
.collect(Collectors.groupingBy(RedactionLogEntry::getId));
|
||||
Map<String, List<RedactionLogEntry>> entriesById = redactionLog.getRedactionLogEntry().stream().collect(Collectors.groupingBy(RedactionLogEntry::getId));
|
||||
for (List<RedactionLogEntry> entries : entriesById.values()) {
|
||||
|
||||
if (entries.isEmpty()) {
|
||||
@@ -35,10 +33,7 @@ public class LegacyVersion0MigrationService {
|
||||
continue;
|
||||
}
|
||||
|
||||
List<RedactionLogEntry> sortedEntries = entries.stream()
|
||||
.sorted(Comparator.comparing(entry -> entry.getChanges()
|
||||
.get(0).getDateTime()))
|
||||
.toList();
|
||||
List<RedactionLogEntry> sortedEntries = entries.stream().sorted(Comparator.comparing(entry -> entry.getChanges().get(0).getDateTime())).toList();
|
||||
|
||||
RedactionLogEntry initialEntry = sortedEntries.get(0);
|
||||
for (RedactionLogEntry entry : sortedEntries.subList(1, sortedEntries.size())) {
|
||||
|
||||
+4
-12
@@ -1,9 +1,9 @@
|
||||
package com.iqser.red.service.redaction.v1.server.migration;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ChangeType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualChange;
|
||||
@@ -70,21 +70,13 @@ public class MigrationMapper {
|
||||
|
||||
public static Set<com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine> getMigratedEngines(RedactionLogEntry entry) {
|
||||
|
||||
Set<com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine> engines = new HashSet<>();
|
||||
|
||||
if (entry.isImported()) {
|
||||
engines.add(com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine.IMPORTED);
|
||||
}
|
||||
|
||||
if (entry.getEngines() == null) {
|
||||
return engines;
|
||||
return Collections.emptySet();
|
||||
}
|
||||
entry.getEngines()
|
||||
return entry.getEngines()
|
||||
.stream()
|
||||
.map(MigrationMapper::toEntityLogEngine)
|
||||
.forEach(engines::add);
|
||||
|
||||
return engines;
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
|
||||
|
||||
+2
-5
@@ -57,7 +57,7 @@ public class MigrationMessageReceiver {
|
||||
|
||||
if (redactionLog.getAnalysisVersion() == 0) {
|
||||
redactionLog = legacyVersion0MigrationService.mergeDuplicateAnnotationIds(redactionLog);
|
||||
} else {
|
||||
} else if (migrationRequest.getManualRedactions() != null) {
|
||||
redactionLog = legacyRedactionLogMergeService.addManualAddEntriesAndRemoveSkippedImported(redactionLog,
|
||||
migrationRequest.getManualRedactions(),
|
||||
migrationRequest.getDossierTemplateId());
|
||||
@@ -67,12 +67,9 @@ public class MigrationMessageReceiver {
|
||||
document,
|
||||
migrationRequest.getDossierTemplateId(),
|
||||
migrationRequest.getManualRedactions(),
|
||||
migrationRequest.getFileId(),
|
||||
migrationRequest.getEntitiesWithComments(),
|
||||
migrationRequest.isFileIsApproved());
|
||||
migrationRequest.getFileId());
|
||||
|
||||
log.info("Storing migrated entityLog and ids to migrate in DB for file {}", migrationRequest.getFileId());
|
||||
|
||||
redactionStorageService.storeObject(migrationRequest.getDossierId(), migrationRequest.getFileId(), FileType.ENTITY_LOG, migratedEntityLog.getEntityLog());
|
||||
redactionStorageService.storeObject(migrationRequest.getDossierId(), migrationRequest.getFileId(), FileType.MIGRATED_IDS, migratedEntityLog.getMigratedIds());
|
||||
|
||||
|
||||
+21
-85
@@ -8,7 +8,6 @@ import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
@@ -20,9 +19,7 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.migration.MigratedIds;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.BaseAnnotation;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRedactionEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualResizeRedaction;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.Rectangle;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
|
||||
@@ -37,6 +34,7 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
|
||||
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.ManualChangesApplicationService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.EntityEnrichmentService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.EntityFindingUtility;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.EntityFromPrecursorCreationService;
|
||||
import com.iqser.red.service.redaction.v1.server.utils.IdBuilder;
|
||||
@@ -56,17 +54,12 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
|
||||
private static final double MATCH_THRESHOLD = 10;
|
||||
EntityFindingUtility entityFindingUtility;
|
||||
EntityEnrichmentService entityEnrichmentService;
|
||||
DictionaryService dictionaryService;
|
||||
ManualChangesApplicationService manualChangesApplicationService;
|
||||
|
||||
|
||||
public MigratedEntityLog migrate(RedactionLog redactionLog,
|
||||
Document document,
|
||||
String dossierTemplateId,
|
||||
ManualRedactions manualRedactions,
|
||||
String fileId,
|
||||
Set<String> entitiesWithComments,
|
||||
boolean fileIsApproved) {
|
||||
public MigratedEntityLog migrate(RedactionLog redactionLog, Document document, String dossierTemplateId, ManualRedactions manualRedactions, String fileId) {
|
||||
|
||||
log.info("Migrating entities for file {}", fileId);
|
||||
List<MigrationEntity> entitiesToMigrate = calculateMigrationEntitiesFromRedactionLog(redactionLog, document, dossierTemplateId, fileId);
|
||||
@@ -74,8 +67,8 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
MigratedIds migratedIds = entitiesToMigrate.stream()
|
||||
.collect(new MigratedIdsCollector());
|
||||
|
||||
applyManualChanges(entitiesToMigrate, manualRedactions);
|
||||
log.info("applying manual changes to migrated entities for file {}", fileId);
|
||||
applyLocalProcessedManualChanges(entitiesToMigrate, manualRedactions, fileIsApproved);
|
||||
|
||||
EntityLog entityLog = new EntityLog();
|
||||
entityLog.setAnalysisNumber(redactionLog.getAnalysisNumber());
|
||||
@@ -96,7 +89,7 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
.map(migrationEntity -> migrationEntity.toEntityLogEntry(oldToNewIDMapping))
|
||||
.toList());
|
||||
|
||||
if (getNumberOfApprovedEntries(redactionLog, document.getNumberOfPages()) != entityLog.getEntityLogEntry().size()) {
|
||||
if (getNumberOfApprovedEntries(redactionLog) != entityLog.getEntityLogEntry().size()) {
|
||||
String message = String.format("Not all entities have been found during the migration redactionLog has %d entries and new entityLog %d",
|
||||
redactionLog.getRedactionLogEntry().size(),
|
||||
entityLog.getEntityLogEntry().size());
|
||||
@@ -104,14 +97,8 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
throw new AssertionError(message);
|
||||
}
|
||||
|
||||
Set<String> entitiesWithUnprocessedChanges = manualRedactions.buildAll()
|
||||
.stream()
|
||||
.filter(manualRedaction -> manualRedaction.getProcessedDate() == null)
|
||||
.map(BaseAnnotation::getAnnotationId)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
MigratedIds idsToMigrateInDb = entitiesToMigrate.stream()
|
||||
.filter(migrationEntity -> migrationEntity.hasManualChangesOrComments(entitiesWithComments, entitiesWithUnprocessedChanges))
|
||||
.filter(MigrationEntity::hasManualChangesOrComments)
|
||||
.filter(m -> !m.getOldId().equals(m.getNewId()))
|
||||
.collect(new MigratedIdsCollector());
|
||||
|
||||
@@ -126,29 +113,20 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
}
|
||||
|
||||
|
||||
private void applyLocalProcessedManualChanges(List<MigrationEntity> entitiesToMigrate, ManualRedactions manualRedactions, boolean fileIsApproved) {
|
||||
private void applyManualChanges(List<MigrationEntity> entitiesToMigrate, ManualRedactions manualRedactions) {
|
||||
|
||||
if (manualRedactions == null) {
|
||||
return;
|
||||
}
|
||||
Map<String, List<BaseAnnotation>> manualChangesPerAnnotationId;
|
||||
|
||||
if (fileIsApproved) {
|
||||
manualChangesPerAnnotationId = manualRedactions.buildAll()
|
||||
.stream()
|
||||
.filter(manualChange -> (manualChange.getProcessedDate() != null && manualChange.isLocal()) //
|
||||
// unprocessed dict change of type IdRemoval or ManualResize must be applied for approved documents
|
||||
|| (manualChange.getProcessedDate() == null && !manualChange.isLocal() //
|
||||
&& (manualChange instanceof IdRemoval || manualChange instanceof ManualResizeRedaction)))
|
||||
.map(this::convertPendingDictChangesToLocal)
|
||||
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
|
||||
} else {
|
||||
manualChangesPerAnnotationId = manualRedactions.buildAll()
|
||||
.stream()
|
||||
.filter(manualChange -> manualChange.getProcessedDate() != null)
|
||||
.filter(BaseAnnotation::isLocal)
|
||||
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
|
||||
}
|
||||
Map<String, List<BaseAnnotation>> manualChangesPerAnnotationId = Stream.of(manualRedactions.getIdsToRemove(),
|
||||
manualRedactions.getEntriesToAdd(),
|
||||
manualRedactions.getForceRedactions(),
|
||||
manualRedactions.getResizeRedactions(),
|
||||
manualRedactions.getLegalBasisChanges(),
|
||||
manualRedactions.getRecategorizations())
|
||||
.flatMap(Collection::stream)
|
||||
.collect(Collectors.groupingBy(BaseAnnotation::getAnnotationId));
|
||||
|
||||
entitiesToMigrate.forEach(migrationEntity -> migrationEntity.applyManualChanges(manualChangesPerAnnotationId.getOrDefault(migrationEntity.getOldId(),
|
||||
Collections.emptyList()),
|
||||
@@ -157,40 +135,15 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
}
|
||||
|
||||
|
||||
private BaseAnnotation convertPendingDictChangesToLocal(BaseAnnotation baseAnnotation) {
|
||||
private static long getNumberOfApprovedEntries(RedactionLog redactionLog) {
|
||||
|
||||
if (baseAnnotation.getProcessedDate() != null) {
|
||||
return baseAnnotation;
|
||||
}
|
||||
|
||||
if (baseAnnotation.isLocal()) {
|
||||
return baseAnnotation;
|
||||
}
|
||||
|
||||
if (baseAnnotation instanceof ManualResizeRedaction manualResizeRedaction) {
|
||||
manualResizeRedaction.setAddToAllDossiers(false);
|
||||
manualResizeRedaction.setUpdateDictionary(false);
|
||||
} else if (baseAnnotation instanceof IdRemoval idRemoval) {
|
||||
idRemoval.setRemoveFromAllDossiers(false);
|
||||
idRemoval.setRemoveFromDictionary(false);
|
||||
}
|
||||
|
||||
return baseAnnotation;
|
||||
}
|
||||
|
||||
|
||||
private long getNumberOfApprovedEntries(RedactionLog redactionLog, int numberOfPages) {
|
||||
|
||||
return redactionLog.getRedactionLogEntry()
|
||||
.stream()
|
||||
.filter(redactionLogEntry -> isOnExistingPage(redactionLogEntry, numberOfPages))
|
||||
.count();
|
||||
return redactionLog.getRedactionLogEntry().size();
|
||||
}
|
||||
|
||||
|
||||
private List<MigrationEntity> calculateMigrationEntitiesFromRedactionLog(RedactionLog redactionLog, Document document, String dossierTemplateId, String fileId) {
|
||||
|
||||
List<MigrationEntity> images = getImageBasedMigrationEntities(redactionLog, document, fileId, dossierTemplateId);
|
||||
List<MigrationEntity> images = getImageBasedMigrationEntities(redactionLog, document, fileId);
|
||||
List<MigrationEntity> textMigrationEntities = getTextBasedMigrationEntities(redactionLog, document, dossierTemplateId, fileId);
|
||||
return Stream.of(textMigrationEntities.stream(), images.stream())
|
||||
.flatMap(Function.identity())
|
||||
@@ -204,7 +157,7 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
}
|
||||
|
||||
|
||||
private List<MigrationEntity> getImageBasedMigrationEntities(RedactionLog redactionLog, Document document, String fileId, String dossierTemplateId) {
|
||||
private List<MigrationEntity> getImageBasedMigrationEntities(RedactionLog redactionLog, Document document, String fileId) {
|
||||
|
||||
List<Image> images = document.streamAllImages()
|
||||
.collect(Collectors.toList());
|
||||
@@ -251,7 +204,7 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
} else {
|
||||
closestImage.skip(ruleIdentifier, reason);
|
||||
}
|
||||
migrationEntities.add(MigrationEntity.fromRedactionLogImage(redactionLogImage, closestImage, fileId, dictionaryService, dossierTemplateId));
|
||||
migrationEntities.add(MigrationEntity.fromRedactionLogImage(redactionLogImage, closestImage, fileId));
|
||||
}
|
||||
return migrationEntities;
|
||||
}
|
||||
@@ -297,8 +250,7 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
List<MigrationEntity> entitiesToMigrate = redactionLog.getRedactionLogEntry()
|
||||
.stream()
|
||||
.filter(redactionLogEntry -> !redactionLogEntry.isImage())
|
||||
.filter(redactionLogEntry -> isOnExistingPage(redactionLogEntry, document.getNumberOfPages()))
|
||||
.map(entry -> MigrationEntity.fromRedactionLogEntry(entry, fileId, dictionaryService, dossierTemplateId))
|
||||
.map(entry -> MigrationEntity.fromRedactionLogEntry(entry, dictionaryService.isHint(entry.getType(), dossierTemplateId), fileId))
|
||||
.toList();
|
||||
|
||||
List<PrecursorEntity> precursorEntities = entitiesToMigrate.stream()
|
||||
@@ -335,20 +287,4 @@ public class RedactionLogToEntityLogMigrationService {
|
||||
return entitiesToMigrate;
|
||||
}
|
||||
|
||||
|
||||
private boolean isOnExistingPage(RedactionLogEntry redactionLogEntry, int numberOfPages) {
|
||||
|
||||
var pages = redactionLogEntry.getPositions()
|
||||
.stream()
|
||||
.map(Rectangle::getPage)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
for (int page : pages) {
|
||||
if (page > numberOfPages) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-1
@@ -14,5 +14,4 @@ public record KieWrapper(KieContainer container, long rulesVersion) {
|
||||
|
||||
return container != null && rulesVersion >= 0;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-1
@@ -19,5 +19,4 @@ public class MigratedEntityLog {
|
||||
|
||||
MigratedIds migratedIds;
|
||||
EntityLog entityLog;
|
||||
|
||||
}
|
||||
|
||||
+41
-61
@@ -4,11 +4,9 @@ import static com.iqser.red.service.redaction.v1.server.service.EntityLogCreator
|
||||
import static com.iqser.red.service.redaction.v1.server.service.EntityLogCreatorService.buildEntryType;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
@@ -20,10 +18,8 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Position;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.Rectangle;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.BaseAnnotation;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRecategorization;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualRedactionEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.ManualResizeRedaction;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.type.DictionaryEntryType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.ManualRedactionType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.migration.MigrationMapper;
|
||||
@@ -32,8 +28,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.entity.IEntity;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.entity.ManualChangeOverwrite;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntity;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
|
||||
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.ManualChangeFactory;
|
||||
import com.iqser.red.service.redaction.v1.server.service.ManualChangesApplicationService;
|
||||
|
||||
@@ -52,8 +46,6 @@ public final class MigrationEntity {
|
||||
|
||||
private final PrecursorEntity precursorEntity;
|
||||
private final RedactionLogEntry redactionLogEntry;
|
||||
private final DictionaryService dictionaryService;
|
||||
private final String dossierTemplateId;
|
||||
private IEntity migratedEntity;
|
||||
private String oldId;
|
||||
private String newId;
|
||||
@@ -63,10 +55,10 @@ public final class MigrationEntity {
|
||||
List<BaseAnnotation> manualChanges = new LinkedList<>();
|
||||
|
||||
|
||||
public static MigrationEntity fromRedactionLogEntry(RedactionLogEntry redactionLogEntry, String fileId, DictionaryService dictionaryService, String dossierTemplateId) {
|
||||
public static MigrationEntity fromRedactionLogEntry(RedactionLogEntry redactionLogEntry, boolean hint, String fileId) {
|
||||
|
||||
boolean hint = dictionaryService.isHint(redactionLogEntry.getType(), dossierTemplateId);
|
||||
PrecursorEntity precursorEntity = createPrecursorEntity(redactionLogEntry, hint);
|
||||
|
||||
if (precursorEntity.getEntityType().equals(EntityType.HINT) && !redactionLogEntry.isHint() && !redactionLogEntry.isRedacted()) {
|
||||
precursorEntity.ignore(precursorEntity.getRuleIdentifier(), precursorEntity.getReason());
|
||||
} else if (redactionLogEntry.lastChangeIsRemoved()) {
|
||||
@@ -81,32 +73,13 @@ public final class MigrationEntity {
|
||||
precursorEntity.skip(precursorEntity.getRuleIdentifier(), precursorEntity.getReason());
|
||||
}
|
||||
|
||||
return MigrationEntity.builder()
|
||||
.precursorEntity(precursorEntity)
|
||||
.redactionLogEntry(redactionLogEntry)
|
||||
.oldId(redactionLogEntry.getId())
|
||||
.fileId(fileId)
|
||||
.dictionaryService(dictionaryService)
|
||||
.dossierTemplateId(dossierTemplateId)
|
||||
.build();
|
||||
return MigrationEntity.builder().precursorEntity(precursorEntity).redactionLogEntry(redactionLogEntry).oldId(redactionLogEntry.getId()).fileId(fileId).build();
|
||||
}
|
||||
|
||||
|
||||
public static MigrationEntity fromRedactionLogImage(RedactionLogEntry redactionLogImage,
|
||||
Image image,
|
||||
String fileId,
|
||||
DictionaryService dictionaryService,
|
||||
String dossierTemplateId) {
|
||||
public static MigrationEntity fromRedactionLogImage(RedactionLogEntry redactionLogImage, Image image, String fileId) {
|
||||
|
||||
return MigrationEntity.builder()
|
||||
.redactionLogEntry(redactionLogImage)
|
||||
.migratedEntity(image)
|
||||
.oldId(redactionLogImage.getId())
|
||||
.newId(image.getId())
|
||||
.fileId(fileId)
|
||||
.dictionaryService(dictionaryService)
|
||||
.dossierTemplateId(dossierTemplateId)
|
||||
.build();
|
||||
return MigrationEntity.builder().redactionLogEntry(redactionLogImage).migratedEntity(image).oldId(redactionLogImage.getId()).newId(image.getId()).fileId(fileId).build();
|
||||
}
|
||||
|
||||
|
||||
@@ -183,6 +156,18 @@ public final class MigrationEntity {
|
||||
}
|
||||
|
||||
|
||||
private static EntryType getEntryType(EntityType entityType) {
|
||||
|
||||
return switch (entityType) {
|
||||
case ENTITY -> EntryType.ENTITY;
|
||||
case HINT -> EntryType.HINT;
|
||||
case FALSE_POSITIVE -> EntryType.FALSE_POSITIVE;
|
||||
case RECOMMENDATION -> EntryType.RECOMMENDATION;
|
||||
case FALSE_RECOMMENDATION -> EntryType.FALSE_RECOMMENDATION;
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
public EntityLogEntry toEntityLogEntry(Map<String, String> oldToNewIdMapping) {
|
||||
|
||||
EntityLogEntry entityLogEntry;
|
||||
@@ -205,15 +190,14 @@ public final class MigrationEntity {
|
||||
entityLogEntry.setReference(migrateSetOfIds(redactionLogEntry.getReference(), oldToNewIdMapping));
|
||||
entityLogEntry.setImportedRedactionIntersections(migrateSetOfIds(redactionLogEntry.getImportedRedactionIntersections(), oldToNewIdMapping));
|
||||
entityLogEntry.setEngines(MigrationMapper.getMigratedEngines(redactionLogEntry));
|
||||
if (redactionLogEntry.getLegalBasis() != null) {
|
||||
entityLogEntry.setLegalBasis(redactionLogEntry.getLegalBasis());
|
||||
}
|
||||
|
||||
if (entityLogEntry.getEntryType().equals(EntryType.HINT) && lastManualChangeIsRemoveLocally(entityLogEntry)) {
|
||||
entityLogEntry.setState(EntryState.IGNORED);
|
||||
}
|
||||
|
||||
if (redactionLogEntry.isImported() && redactionLogEntry.getValue() == null) {
|
||||
entityLogEntry.setValue("Imported Redaction");
|
||||
}
|
||||
|
||||
return entityLogEntry;
|
||||
}
|
||||
|
||||
@@ -241,15 +225,13 @@ public final class MigrationEntity {
|
||||
|
||||
public EntityLogEntry createEntityLogEntry(Image image) {
|
||||
|
||||
String imageType = image.getImageType().equals(ImageType.OTHER) ? "image" : image.getImageType().toString().toLowerCase(Locale.ENGLISH);
|
||||
List<Position> positions = getPositionsFromOverride(image).orElse(List.of(new Position(image.getPosition(), image.getPage().getNumber())));
|
||||
return EntityLogEntry.builder()
|
||||
.id(image.getId())
|
||||
.value(image.getValue())
|
||||
.type(imageType)
|
||||
.value(image.value())
|
||||
.type(image.type())
|
||||
.reason(image.buildReasonWithManualChangeDescriptions())
|
||||
.legalBasis(image.getManualOverwrite().getLegalBasis()
|
||||
.orElse(redactionLogEntry.getLegalBasis()))
|
||||
.legalBasis(image.legalBasis())
|
||||
.matchedRule(image.getMatchedRule().getRuleIdentifier().toString())
|
||||
.dictionaryEntry(false)
|
||||
.positions(positions)
|
||||
@@ -261,7 +243,7 @@ public final class MigrationEntity {
|
||||
.textBefore(redactionLogEntry.getTextBefore())
|
||||
.imageHasTransparency(image.isTransparent())
|
||||
.state(buildEntryState(image))
|
||||
.entryType(dictionaryService.isHint(imageType, dossierTemplateId) ? EntryType.IMAGE_HINT : EntryType.IMAGE)
|
||||
.entryType(redactionLogEntry.isHint() ? EntryType.IMAGE_HINT : EntryType.IMAGE)
|
||||
.build();
|
||||
|
||||
}
|
||||
@@ -272,8 +254,7 @@ public final class MigrationEntity {
|
||||
return EntityLogEntry.builder()
|
||||
.id(precursorEntity.getId())
|
||||
.reason(precursorEntity.buildReasonWithManualChangeDescriptions())
|
||||
.legalBasis(precursorEntity.getManualOverwrite().getLegalBasis()
|
||||
.orElse(redactionLogEntry.getLegalBasis()))
|
||||
.legalBasis(precursorEntity.legalBasis())
|
||||
.value(precursorEntity.value())
|
||||
.type(precursorEntity.type())
|
||||
.state(buildEntryState(precursorEntity))
|
||||
@@ -307,8 +288,7 @@ public final class MigrationEntity {
|
||||
.id(entity.getId())
|
||||
.positions(rectanglesPerLine)
|
||||
.reason(entity.buildReasonWithManualChangeDescriptions())
|
||||
.legalBasis(entity.getManualOverwrite().getLegalBasis()
|
||||
.orElse(redactionLogEntry.getLegalBasis()))
|
||||
.legalBasis(entity.legalBasis())
|
||||
.value(entity.getManualOverwrite().getValue()
|
||||
.orElse(entity.getMatchedRule().isWriteValueWithLineBreaks() ? entity.getValueWithLineBreaks() : entity.getValue()))
|
||||
.type(entity.type())
|
||||
@@ -351,11 +331,11 @@ public final class MigrationEntity {
|
||||
}
|
||||
|
||||
|
||||
public boolean hasManualChangesOrComments(Set<String> entitiesWithComments, Set<String> entitiesWithUnprocessedChanges) {
|
||||
public boolean hasManualChangesOrComments() {
|
||||
|
||||
return !(redactionLogEntry.getManualChanges() == null || redactionLogEntry.getManualChanges().isEmpty()) || //
|
||||
!(redactionLogEntry.getComments() == null || redactionLogEntry.getComments().isEmpty()) //
|
||||
|| hasManualChanges() || entitiesWithComments.contains(oldId) || entitiesWithUnprocessedChanges.contains(oldId);
|
||||
|| hasManualChanges();
|
||||
}
|
||||
|
||||
|
||||
@@ -370,11 +350,17 @@ public final class MigrationEntity {
|
||||
manualChanges.addAll(manualChangesToApply);
|
||||
manualChangesToApply.forEach(manualChange -> {
|
||||
if (manualChange instanceof ManualResizeRedaction manualResizeRedaction && migratedEntity instanceof TextEntity textEntity) {
|
||||
manualResizeRedaction.setAnnotationId(newId);
|
||||
manualChangesApplicationService.resize(textEntity, manualResizeRedaction);
|
||||
} else if (manualChange instanceof ManualRecategorization manualRecategorization && migratedEntity instanceof Image image) {
|
||||
image.setImageType(ImageType.fromString(manualRecategorization.getType()));
|
||||
migratedEntity.getManualOverwrite().addChange(manualChange);
|
||||
// Due to the value in the old redaction log already being resized, there is no way to find the original entity ID and therefore to migrate the resize annotation correctly.
|
||||
// Instead, we add an add_locally change to the db.
|
||||
ManualResizeRedaction migratedManualResizeRedaction = ManualResizeRedaction.builder()
|
||||
.positions(manualResizeRedaction.getPositions())
|
||||
.annotationId(getNewId())
|
||||
.updateDictionary(manualResizeRedaction.getUpdateDictionary())
|
||||
.addToAllDossiers(manualResizeRedaction.isAddToAllDossiers())
|
||||
.textAfter(manualResizeRedaction.getTextAfter())
|
||||
.textBefore(manualResizeRedaction.getTextBefore())
|
||||
.build();
|
||||
manualChangesApplicationService.resize(textEntity, migratedManualResizeRedaction);
|
||||
} else {
|
||||
migratedEntity.getManualOverwrite().addChange(manualChange);
|
||||
}
|
||||
@@ -392,25 +378,19 @@ public final class MigrationEntity {
|
||||
.findFirst()
|
||||
.orElse(manualChanges.get(0)).getUser();
|
||||
|
||||
OffsetDateTime requestDate = manualChanges.get(0).getRequestDate();
|
||||
|
||||
return ManualRedactionEntry.builder()
|
||||
.annotationId(newId)
|
||||
.fileId(fileId)
|
||||
.user(user)
|
||||
.requestDate(requestDate)
|
||||
.type(redactionLogEntry.getType())
|
||||
.value(redactionLogEntry.getValue())
|
||||
.reason(redactionLogEntry.getReason())
|
||||
.legalBasis(redactionLogEntry.getLegalBasis())
|
||||
.section(redactionLogEntry.getSection())
|
||||
.rectangle(false)
|
||||
.addToDictionary(false)
|
||||
.addToDossierDictionary(false)
|
||||
.rectangle(false)
|
||||
.positions(buildPositions(migratedEntity))
|
||||
.textAfter(redactionLogEntry.getTextAfter())
|
||||
.textBefore(redactionLogEntry.getTextBefore())
|
||||
.dictionaryEntryType(DictionaryEntryType.ENTRY)
|
||||
.user(user)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
+2
-23
@@ -11,10 +11,6 @@ import lombok.AllArgsConstructor;
|
||||
import lombok.Getter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a collection of named entity recognition (NER) entities.
|
||||
* This class provides methods to manage and query NER entities.
|
||||
*/
|
||||
@Getter
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE, makeFinal = true)
|
||||
@@ -29,35 +25,18 @@ public class NerEntities {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if there are any entities of a specified type.
|
||||
*
|
||||
* @param type The type of entity to check for.
|
||||
* @return true if there is at least one entity of the specified type, false otherwise.
|
||||
*/
|
||||
public boolean hasEntitiesOfType(String type) {
|
||||
|
||||
return nerEntityList.stream()
|
||||
.anyMatch(nerEntity -> nerEntity.type.equals(type));
|
||||
return nerEntityList.stream().anyMatch(nerEntity -> nerEntity.type.equals(type));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns a stream of NER entities of a specified type.
|
||||
*
|
||||
* @param type The type of entities to return.
|
||||
* @return a stream of {@link NerEntity} objects of the specified type.
|
||||
*/
|
||||
public Stream<NerEntity> streamEntitiesOfType(String type) {
|
||||
|
||||
return nerEntityList.stream()
|
||||
.filter(nerEntity -> nerEntity.type().equals(type));
|
||||
return nerEntityList.stream().filter(nerEntity -> nerEntity.type().equals(type));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Represents a single NER entity with its value, text range, and type.
|
||||
*/
|
||||
public record NerEntity(String value, TextRange textRange, String type) {
|
||||
|
||||
}
|
||||
|
||||
+1
-2
@@ -88,8 +88,7 @@ public class Entity {
|
||||
.textAfter(e.getTextAfter())
|
||||
.startOffset(e.getStartOffset())
|
||||
.endOffset(e.getEndOffset())
|
||||
.length(Optional.ofNullable(e.getValue())
|
||||
.orElse("").length())
|
||||
.length(Optional.ofNullable(e.getValue()).orElse("").length())
|
||||
.imageHasTransparency(e.isImageHasTransparency())
|
||||
.isDictionaryEntry(e.isDictionaryEntry())
|
||||
.isDossierDictionaryEntry(e.isDossierDictionaryEntry())
|
||||
|
||||
+4
-74
@@ -23,9 +23,6 @@ import com.iqser.red.service.redaction.v1.server.utils.exception.NotFoundExcepti
|
||||
import lombok.Data;
|
||||
import lombok.Getter;
|
||||
|
||||
/**
|
||||
* A class representing a dictionary used for redaction processes, containing various dictionary models and their versions.
|
||||
*/
|
||||
@Data
|
||||
public class Dictionary {
|
||||
|
||||
@@ -54,15 +51,9 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the dictionary contains local entries.
|
||||
*
|
||||
* @return true if any dictionary model contains local entries, false otherwise.
|
||||
*/
|
||||
public boolean hasLocalEntries() {
|
||||
|
||||
return dictionaryModels.stream()
|
||||
.anyMatch(dm -> !dm.getLocalEntriesWithMatchedRules().isEmpty());
|
||||
return dictionaryModels.stream().anyMatch(dm -> !dm.getLocalEntriesWithMatchedRules().isEmpty());
|
||||
}
|
||||
|
||||
|
||||
@@ -72,13 +63,6 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the {@link DictionaryModel} of a specified type.
|
||||
*
|
||||
* @param type The type of dictionary model to retrieve.
|
||||
* @return The {@link DictionaryModel} of the specified type.
|
||||
* @throws NotFoundException If the specified type is not found in the dictionary.
|
||||
*/
|
||||
public DictionaryModel getType(String type) {
|
||||
|
||||
DictionaryModel model = localAccessMap.get(type);
|
||||
@@ -89,12 +73,6 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the dictionary of a specific type is considered a hint.
|
||||
*
|
||||
* @param type The type of dictionary to check.
|
||||
* @return true if the dictionary model is marked as a hint, false otherwise.
|
||||
*/
|
||||
public boolean isHint(String type) {
|
||||
|
||||
DictionaryModel model = localAccessMap.get(type);
|
||||
@@ -105,12 +83,6 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the dictionary of a specific type is case-insensitive.
|
||||
*
|
||||
* @param type The type of dictionary to check.
|
||||
* @return true if the dictionary is case-insensitive, false otherwise.
|
||||
*/
|
||||
public boolean isCaseInsensitiveDictionary(String type) {
|
||||
|
||||
DictionaryModel dictionaryModel = localAccessMap.get(type);
|
||||
@@ -121,18 +93,6 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Adds a local dictionary entry of a specific type.
|
||||
*
|
||||
* @param type The type of dictionary to add the entry to.
|
||||
* @param value The value of the entry.
|
||||
* @param matchedRules A collection of {@link MatchedRule} associated with the entry.
|
||||
* @param alsoAddLastname Indicates whether to also add the lastname separately as an entry.
|
||||
* @throws IllegalArgumentException If the specified type does not exist within the dictionary, if the type
|
||||
* does not have any local entries defined, or if the provided value is
|
||||
* blank. This ensures that only valid, non-empty entries
|
||||
* are added to the dictionary.
|
||||
*/
|
||||
private void addLocalDictionaryEntry(String type, String value, Collection<MatchedRule> matchedRules, boolean alsoAddLastname) {
|
||||
|
||||
if (value.isBlank()) {
|
||||
@@ -156,49 +116,28 @@ public class Dictionary {
|
||||
}
|
||||
localAccessMap.get(type)
|
||||
.getLocalEntriesWithMatchedRules()
|
||||
.merge(cleanedValue.trim(),
|
||||
matchedRulesSet,
|
||||
(set1, set2) -> Stream.concat(set1.stream(), set2.stream())
|
||||
.collect(Collectors.toSet()));
|
||||
.merge(cleanedValue.trim(), matchedRulesSet, (set1, set2) -> Stream.concat(set1.stream(), set2.stream()).collect(Collectors.toSet()));
|
||||
if (alsoAddLastname) {
|
||||
String lastname = cleanedValue.split(" ")[0];
|
||||
localAccessMap.get(type)
|
||||
.getLocalEntriesWithMatchedRules()
|
||||
.merge(lastname,
|
||||
matchedRulesSet,
|
||||
(set1, set2) -> Stream.concat(set1.stream(), set2.stream())
|
||||
.collect(Collectors.toSet()));
|
||||
.merge(lastname, matchedRulesSet, (set1, set2) -> Stream.concat(set1.stream(), set2.stream()).collect(Collectors.toSet()));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Recommends a text entity for inclusion in every dictionary model without separating the last name.
|
||||
*
|
||||
* @param textEntity The {@link TextEntity} to be recommended.
|
||||
*/
|
||||
public void recommendEverywhere(TextEntity textEntity) {
|
||||
|
||||
addLocalDictionaryEntry(textEntity.type(), textEntity.getValue(), textEntity.getMatchedRuleList(), false);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Recommends a text entity for inclusion in every dictionary model with the last name added separately.
|
||||
*
|
||||
* @param textEntity The {@link TextEntity} to be recommended.
|
||||
*/
|
||||
public void recommendEverywhereWithLastNameSeparately(TextEntity textEntity) {
|
||||
|
||||
addLocalDictionaryEntry(textEntity.type(), textEntity.getValue(), textEntity.getMatchedRuleList(), true);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Adds multiple author names contained within a text entity as recommendations in the dictionary.
|
||||
*
|
||||
* @param textEntity The {@link TextEntity} containing author names to be added.
|
||||
*/
|
||||
public void addMultipleAuthorsAsRecommendation(TextEntity textEntity) {
|
||||
|
||||
splitIntoAuthorNames(textEntity).forEach(authorName -> addLocalDictionaryEntry(textEntity.type(), authorName, textEntity.getMatchedRuleList(), true));
|
||||
@@ -206,12 +145,6 @@ public class Dictionary {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Splits a {@link TextEntity} into individual author names based on commas or new lines.
|
||||
*
|
||||
* @param textEntity The {@link TextEntity} to split.
|
||||
* @return A list of strings where each string is an author name.
|
||||
*/
|
||||
public static List<String> splitIntoAuthorNames(TextEntity textEntity) {
|
||||
|
||||
List<String> splitAuthorNames;
|
||||
@@ -220,10 +153,7 @@ public class Dictionary {
|
||||
} else {
|
||||
splitAuthorNames = Arrays.asList(textEntity.getValueWithLineBreaks().split("\n"));
|
||||
}
|
||||
return splitAuthorNames.stream()
|
||||
.map(String::trim)
|
||||
.filter(authorName -> Patterns.AUTHOR_NAME_PATTERN.matcher(authorName).matches())
|
||||
.toList();
|
||||
return splitAuthorNames.stream().map(String::trim).filter(authorName -> Patterns.AUTHOR_NAME_PATTERN.matcher(authorName).matches()).toList();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+24
-78
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.server.model.dictionary;
|
||||
|
||||
import java.io.Serializable;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
@@ -10,17 +11,11 @@ import com.iqser.red.service.dictionarymerge.commons.DictionaryEntry;
|
||||
import com.iqser.red.service.dictionarymerge.commons.DictionaryEntryModel;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.entity.MatchedRule;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Represents a model of a dictionary containing entries for redaction processes.
|
||||
* It includes various types of entries such as standard entries, false positives,
|
||||
* and false recommendations. Additionally, it manages local entries with matched
|
||||
* rules for enhanced search and matching capabilities.
|
||||
*/
|
||||
@Data
|
||||
@Slf4j
|
||||
@AllArgsConstructor
|
||||
public class DictionaryModel implements Serializable {
|
||||
|
||||
private final String type;
|
||||
@@ -34,7 +29,6 @@ public class DictionaryModel implements Serializable {
|
||||
private final Set<DictionaryEntryModel> falseRecommendations;
|
||||
|
||||
private transient SearchImplementation entriesSearch;
|
||||
private transient SearchImplementation deletionEntriesSearch;
|
||||
private transient SearchImplementation falsePositiveSearch;
|
||||
private transient SearchImplementation falseRecommendationsSearch;
|
||||
|
||||
@@ -42,19 +36,6 @@ public class DictionaryModel implements Serializable {
|
||||
private transient SearchImplementation localSearch;
|
||||
|
||||
|
||||
/**
|
||||
* Constructs a new DictionaryModel with specified parameters.
|
||||
*
|
||||
* @param type The type of the dictionary model.
|
||||
* @param rank The rank order of the dictionary model.
|
||||
* @param color An array representing the color associated with this model.
|
||||
* @param caseInsensitive Flag indicating whether the dictionary is case-insensitive.
|
||||
* @param hint Flag indicating whether this model should be used as a hint.
|
||||
* @param entries Set of dictionary entry models representing the entries.
|
||||
* @param falsePositives Set of dictionary entry models representing false positives.
|
||||
* @param falseRecommendations Set of dictionary entry models representing false recommendations.
|
||||
* @param isDossierDictionary Flag indicating whether this model is for a dossier dictionary.
|
||||
*/
|
||||
public DictionaryModel(String type,
|
||||
int rank,
|
||||
float[] color,
|
||||
@@ -71,17 +52,23 @@ public class DictionaryModel implements Serializable {
|
||||
this.caseInsensitive = caseInsensitive;
|
||||
this.hint = hint;
|
||||
this.isDossierDictionary = isDossierDictionary;
|
||||
|
||||
this.entries = entries;
|
||||
this.falsePositives = falsePositives;
|
||||
this.falseRecommendations = falseRecommendations;
|
||||
|
||||
this.entriesSearch = new SearchImplementation(this.entries.stream().filter(e -> !e.isDeleted()).map(DictionaryEntryModel::getValue).collect(Collectors.toList()),
|
||||
caseInsensitive);
|
||||
this.falsePositiveSearch = new SearchImplementation(this.falsePositives.stream().filter(e -> !e.isDeleted()).map(DictionaryEntryModel::getValue).collect(Collectors.toList()),
|
||||
caseInsensitive);
|
||||
this.falseRecommendationsSearch = new SearchImplementation(this.falseRecommendations.stream()
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the search implementation for local entries.
|
||||
*
|
||||
* @return The {@link SearchImplementation} for local entries.
|
||||
*/
|
||||
public SearchImplementation getLocalSearch() {
|
||||
|
||||
if (this.localSearch == null || this.localSearch.getValues().size() != this.localEntriesWithMatchedRules.size()) {
|
||||
@@ -91,85 +78,44 @@ public class DictionaryModel implements Serializable {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the search implementation for non-deleted dictionary entries.
|
||||
*
|
||||
* @return The {@link SearchImplementation} for non-deleted dictionary entries.
|
||||
*/
|
||||
public SearchImplementation getEntriesSearch() {
|
||||
|
||||
if (entriesSearch == null) {
|
||||
this.entriesSearch = new SearchImplementation(this.entries.stream()
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
this.entriesSearch = new SearchImplementation(this.entries.stream().filter(e -> !e.isDeleted()).map(DictionaryEntry::getValue).collect(Collectors.toList()),
|
||||
caseInsensitive);
|
||||
}
|
||||
return entriesSearch;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the search implementation for deleted dictionary entries.
|
||||
*
|
||||
* @return The {@link SearchImplementation} for deleted dictionary entries.
|
||||
*/
|
||||
public SearchImplementation getDeletionEntriesSearch() {
|
||||
|
||||
if (deletionEntriesSearch == null) {
|
||||
this.deletionEntriesSearch = new SearchImplementation(this.entries.stream()
|
||||
.filter(DictionaryEntry::isDeleted)
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
}
|
||||
return deletionEntriesSearch;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the search implementation for non-deleted false positive entries.
|
||||
*
|
||||
* @return The {@link SearchImplementation} for non-deleted false positive entries.
|
||||
*/
|
||||
public SearchImplementation getFalsePositiveSearch() {
|
||||
|
||||
if (falsePositiveSearch == null) {
|
||||
this.falsePositiveSearch = new SearchImplementation(this.falsePositives.stream()
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
}
|
||||
return falsePositiveSearch;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the search implementation for non-deleted false recommendation entries.
|
||||
*
|
||||
* @return The {@link SearchImplementation} for non-deleted false recommendation entries.
|
||||
*/
|
||||
public SearchImplementation getFalseRecommendationsSearch() {
|
||||
|
||||
if (falseRecommendationsSearch == null) {
|
||||
this.falseRecommendationsSearch = new SearchImplementation(this.falseRecommendations.stream()
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
.filter(e -> !e.isDeleted())
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toList()), caseInsensitive);
|
||||
}
|
||||
return falseRecommendationsSearch;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the matched rules for a given value from the local dictionary entries.
|
||||
* The value is processed based on the case sensitivity of the dictionary.
|
||||
*
|
||||
* @param value The value for which to retrieve the matched rules.
|
||||
* @return A set of {@link MatchedRule} associated with the given value, or null if no rules are found.
|
||||
*/
|
||||
public Set<MatchedRule> getMatchedRulesForLocalDictionaryEntry(String value) {
|
||||
|
||||
var cleanedValue = isCaseInsensitive() ? value.toLowerCase(Locale.US) : value;
|
||||
|
||||
return localEntriesWithMatchedRules.get(cleanedValue);
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+7
-24
@@ -76,9 +76,7 @@ public class SearchImplementation {
|
||||
if (ignoreCase) {
|
||||
textToCheck = textToCheck.toLowerCase(Locale.ROOT);
|
||||
}
|
||||
return this.pattern.matcher(textToCheck).results()
|
||||
.findAny()
|
||||
.isPresent();
|
||||
return this.pattern.matcher(textToCheck).results().findAny().isPresent();
|
||||
} else {
|
||||
return this.trie.containsMatch(textToCheck);
|
||||
}
|
||||
@@ -91,14 +89,9 @@ public class SearchImplementation {
|
||||
return new ArrayList<>();
|
||||
}
|
||||
if (this.pattern != null) {
|
||||
return this.pattern.matcher(text).results()
|
||||
.map(r -> new TextRange(r.start(), r.end()))
|
||||
.collect(Collectors.toList());
|
||||
return this.pattern.matcher(text).results().map(r -> new TextRange(r.start(), r.end())).collect(Collectors.toList());
|
||||
} else {
|
||||
return this.trie.parseText(text)
|
||||
.stream()
|
||||
.map(r -> new TextRange(r.getStart(), r.getEnd() + 1))
|
||||
.collect(Collectors.toList());
|
||||
return this.trie.parseText(text).stream().map(r -> new TextRange(r.getStart(), r.getEnd() + 1)).collect(Collectors.toList());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -110,14 +103,9 @@ public class SearchImplementation {
|
||||
}
|
||||
CharSequence subSequence = text.subSequence(region.start(), region.end());
|
||||
if (this.pattern != null) {
|
||||
return this.pattern.matcher(subSequence).results()
|
||||
.map(r -> new TextRange(r.start() + region.start(), r.end() + region.start()))
|
||||
.collect(Collectors.toList());
|
||||
return this.pattern.matcher(subSequence).results().map(r -> new TextRange(r.start() + region.start(), r.end() + region.start())).collect(Collectors.toList());
|
||||
} else {
|
||||
return this.trie.parseText(subSequence)
|
||||
.stream()
|
||||
.map(r -> new TextRange(r.getStart() + region.start(), r.getEnd() + region.start() + 1))
|
||||
.collect(Collectors.toList());
|
||||
return this.trie.parseText(subSequence).stream().map(r -> new TextRange(r.getStart() + region.start(), r.getEnd() + region.start() + 1)).collect(Collectors.toList());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -132,14 +120,9 @@ public class SearchImplementation {
|
||||
if (ignoreCase) {
|
||||
textToCheck = textToCheck.toLowerCase(Locale.ROOT);
|
||||
}
|
||||
return this.pattern.matcher(textToCheck).results()
|
||||
.map(r -> new MatchPosition(r.start(), r.end()))
|
||||
.collect(Collectors.toList());
|
||||
return this.pattern.matcher(textToCheck).results().map(r -> new MatchPosition(r.start(), r.end())).collect(Collectors.toList());
|
||||
} else {
|
||||
return this.trie.parseText(textToCheck)
|
||||
.stream()
|
||||
.map(r -> new MatchPosition(r.getStart(), r.getEnd() + 1))
|
||||
.collect(Collectors.toList());
|
||||
return this.trie.parseText(textToCheck).stream().map(r -> new MatchPosition(r.getStart(), r.getEnd() + 1)).collect(Collectors.toList());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+9
-21
@@ -40,10 +40,7 @@ public class DocumentTree {
|
||||
|
||||
public TextBlock buildTextBlock() {
|
||||
|
||||
return allEntriesInOrder().map(Entry::getNode)
|
||||
.filter(SemanticNode::isLeaf)
|
||||
.map(SemanticNode::getLeafTextBlock)
|
||||
.collect(new TextBlockCollector());
|
||||
return allEntriesInOrder().map(Entry::getNode).filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@@ -92,8 +89,8 @@ public class DocumentTree {
|
||||
if (treeId.isEmpty()) {
|
||||
return root != null;
|
||||
}
|
||||
Entry entry = root;
|
||||
for (int id : treeId) {
|
||||
Entry entry = root.children.get(treeId.get(0));
|
||||
for (int id : treeId.subList(1, treeId.size())) {
|
||||
if (id >= entry.children.size() || 0 > id) {
|
||||
return false;
|
||||
}
|
||||
@@ -117,16 +114,13 @@ public class DocumentTree {
|
||||
|
||||
public Stream<SemanticNode> childNodes(List<Integer> treeId) {
|
||||
|
||||
return getEntryById(treeId).children.stream()
|
||||
.map(Entry::getNode);
|
||||
return getEntryById(treeId).children.stream().map(Entry::getNode);
|
||||
}
|
||||
|
||||
|
||||
public Stream<SemanticNode> childNodesOfType(List<Integer> treeId, NodeType nodeType) {
|
||||
|
||||
return getEntryById(treeId).children.stream()
|
||||
.filter(entry -> entry.node.getType().equals(nodeType))
|
||||
.map(Entry::getNode);
|
||||
return getEntryById(treeId).children.stream().filter(entry -> entry.node.getType().equals(nodeType)).map(Entry::getNode);
|
||||
}
|
||||
|
||||
|
||||
@@ -205,32 +199,26 @@ public class DocumentTree {
|
||||
|
||||
public Stream<Entry> allEntriesInOrder() {
|
||||
|
||||
return Stream.of(root)
|
||||
.flatMap(DocumentTree::flatten);
|
||||
return Stream.of(root).flatMap(DocumentTree::flatten);
|
||||
}
|
||||
|
||||
|
||||
public Stream<Entry> allSubEntriesInOrder(List<Integer> parentId) {
|
||||
|
||||
return getEntryById(parentId).children.stream()
|
||||
.flatMap(DocumentTree::flatten);
|
||||
return getEntryById(parentId).children.stream().flatMap(DocumentTree::flatten);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return String.join("\n",
|
||||
allEntriesInOrder().map(Entry::toString)
|
||||
.toList());
|
||||
return String.join("\n", allEntriesInOrder().map(Entry::toString).toList());
|
||||
}
|
||||
|
||||
|
||||
private static Stream<Entry> flatten(Entry entry) {
|
||||
|
||||
return Stream.concat(Stream.of(entry),
|
||||
entry.children.stream()
|
||||
.flatMap(DocumentTree::flatten));
|
||||
return Stream.concat(Stream.of(entry), entry.children.stream().flatMap(DocumentTree::flatten));
|
||||
}
|
||||
|
||||
|
||||
|
||||
+6
-88
@@ -11,10 +11,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBl
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.Setter;
|
||||
|
||||
/**
|
||||
* Represents a range of text defined by a start and end index.
|
||||
* Provides functionality to check containment, intersection, and to adjust ranges based on specified conditions.
|
||||
*/
|
||||
@Setter
|
||||
@EqualsAndHashCode
|
||||
@SuppressWarnings("PMD.AvoidFieldNameMatchingMethodName")
|
||||
@@ -24,13 +20,6 @@ public class TextRange implements Comparable<TextRange> {
|
||||
private int end;
|
||||
|
||||
|
||||
/**
|
||||
* Constructs a TextRange with specified start and end indexes.
|
||||
*
|
||||
* @param start The starting index of the range.
|
||||
* @param end The ending index of the range.
|
||||
* @throws IllegalArgumentException If start is greater than end.
|
||||
*/
|
||||
public TextRange(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
@@ -41,11 +30,6 @@ public class TextRange implements Comparable<TextRange> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the length of the text range.
|
||||
*
|
||||
* @return The length of the range.
|
||||
*/
|
||||
public int length() {
|
||||
|
||||
return end - start;
|
||||
@@ -64,38 +48,18 @@ public class TextRange implements Comparable<TextRange> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this {@link TextRange} fully contains another TextRange.
|
||||
*
|
||||
* @param textRange The {@link TextRange} to check.
|
||||
* @return true if this range contains the specified range, false otherwise.
|
||||
*/
|
||||
public boolean contains(TextRange textRange) {
|
||||
|
||||
return start <= textRange.start() && textRange.end() <= end;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this {@link TextRange} is fully contained by another TextRange.
|
||||
*
|
||||
* @param textRange The {@link TextRange} to check against.
|
||||
* @return true if this range is contained by the specified range, false otherwise.
|
||||
*/
|
||||
public boolean containedBy(TextRange textRange) {
|
||||
|
||||
return textRange.contains(this);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this {@link TextRange} contains another range specified by start and end indices.
|
||||
*
|
||||
* @param start The starting index of the range to check.
|
||||
* @param end The ending index of the range to check.
|
||||
* @return true if this range fully contains the specified range, false otherwise.
|
||||
* @throws IllegalArgumentException If the start index is greater than the end index.
|
||||
*/
|
||||
public boolean contains(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
@@ -105,14 +69,6 @@ public class TextRange implements Comparable<TextRange> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this {@link TextRange} is fully contained within another range specified by start and end indices.
|
||||
*
|
||||
* @param start The starting index of the outer range.
|
||||
* @param end The ending index of the outer range.
|
||||
* @return true if this range is fully contained within the specified range, false otherwise.
|
||||
* @throws IllegalArgumentException If the start index is greater than the end index.
|
||||
*/
|
||||
public boolean containedBy(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
@@ -122,46 +78,22 @@ public class TextRange implements Comparable<TextRange> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the specified index is within this {@link TextRange}.
|
||||
*
|
||||
* @param index The index to check.
|
||||
* @return true if the index is within the range (inclusive of the start and exclusive of the end), false otherwise.
|
||||
*/
|
||||
public boolean contains(int index) {
|
||||
|
||||
return start <= index && index < end;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this {@link TextRange} intersects with another {@link TextRange}.
|
||||
*
|
||||
* @param textRange The {@link TextRange} to check for intersection.
|
||||
* @return true if the ranges intersect, false otherwise.
|
||||
*/
|
||||
public boolean intersects(TextRange textRange) {
|
||||
|
||||
return textRange.start() < this.end && this.start < textRange.end();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Splits this TextRange into multiple ranges based on a list of indices.
|
||||
*
|
||||
* @param splitIndices The indices at which to split the range.
|
||||
* @return A list of TextRanges resulting from the split.
|
||||
* @throws IndexOutOfBoundsException If any split index is outside this TextRange.
|
||||
*/
|
||||
public List<TextRange> split(List<Integer> splitIndices) {
|
||||
|
||||
if (splitIndices.stream()
|
||||
.anyMatch(idx -> !this.contains(idx))) {
|
||||
throw new IndexOutOfBoundsException(format("%s splitting indices are out of range for %s",
|
||||
splitIndices.stream()
|
||||
.filter(idx -> !this.contains(idx))
|
||||
.toList(),
|
||||
this));
|
||||
if (splitIndices.stream().anyMatch(idx -> !this.contains(idx))) {
|
||||
throw new IndexOutOfBoundsException(format("%s splitting indices are out of range for %s", splitIndices.stream().filter(idx -> !this.contains(idx)).toList(), this));
|
||||
}
|
||||
List<TextRange> splitBoundaries = new LinkedList<>();
|
||||
int previousIndex = start;
|
||||
@@ -179,23 +111,10 @@ public class TextRange implements Comparable<TextRange> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Merges a collection of TextRanges into a single Text range encompassing all.
|
||||
*
|
||||
* @param boundaries The collection of TextRanges to merge.
|
||||
* @return A new TextRange covering the entire span of the given ranges.
|
||||
* @throws IllegalArgumentException If boundaries are empty.
|
||||
*/
|
||||
public static TextRange merge(Collection<TextRange> boundaries) {
|
||||
|
||||
int minStart = boundaries.stream()
|
||||
.mapToInt(TextRange::start)
|
||||
.min()
|
||||
.orElseThrow(IllegalArgumentException::new);
|
||||
int maxEnd = boundaries.stream()
|
||||
.mapToInt(TextRange::end)
|
||||
.max()
|
||||
.orElseThrow(IllegalArgumentException::new);
|
||||
int minStart = boundaries.stream().mapToInt(TextRange::start).min().orElseThrow(IllegalArgumentException::new);
|
||||
int maxEnd = boundaries.stream().mapToInt(TextRange::end).max().orElseThrow(IllegalArgumentException::new);
|
||||
return new TextRange(minStart, maxEnd);
|
||||
}
|
||||
|
||||
@@ -222,17 +141,16 @@ public class TextRange implements Comparable<TextRange> {
|
||||
|
||||
|
||||
/**
|
||||
* Shrinks the boundary, such that textBlock.subSequence(boundary) returns a string without trailing or preceding whitespaces.
|
||||
* shrinks the boundary, such that textBlock.subSequence(boundary) returns a string without trailing or preceding whitespaces.
|
||||
*
|
||||
* @param textBlock TextBlock to check whitespaces against
|
||||
* @return Trimmed boundary
|
||||
* @return trimmed boundary
|
||||
*/
|
||||
public TextRange trim(TextBlock textBlock) {
|
||||
|
||||
if (this.length() == 0) {
|
||||
return this;
|
||||
}
|
||||
|
||||
int trimmedStart = this.start;
|
||||
while (textBlock.containsIndex(trimmedStart) && trimmedStart < end && Character.isWhitespace(textBlock.charAt(trimmedStart))) {
|
||||
trimmedStart++;
|
||||
|
||||
+1
-2
@@ -5,6 +5,5 @@ public enum EntityType {
|
||||
HINT,
|
||||
RECOMMENDATION,
|
||||
FALSE_POSITIVE,
|
||||
FALSE_RECOMMENDATION,
|
||||
DICTIONARY_REMOVAL
|
||||
FALSE_RECOMMENDATION
|
||||
}
|
||||
|
||||
+18
-195
@@ -12,173 +12,82 @@ import lombok.NonNull;
|
||||
|
||||
public interface IEntity {
|
||||
|
||||
/**
|
||||
* Gets the list of rules matched against this entity.
|
||||
*
|
||||
* @return A priority queue of matched rules.
|
||||
*/
|
||||
PriorityQueue<MatchedRule> getMatchedRuleList();
|
||||
|
||||
|
||||
/**
|
||||
* Gets the manual overwrite actions applied to this entity, if any.
|
||||
*
|
||||
* @return The manual overwrite details.
|
||||
*/
|
||||
ManualChangeOverwrite getManualOverwrite();
|
||||
|
||||
|
||||
/**
|
||||
* Gets the value of this entity as a string.
|
||||
*
|
||||
* @return The string value.
|
||||
*/
|
||||
String getValue();
|
||||
|
||||
|
||||
/**
|
||||
* Gets the range of text in the document associated with this entity.
|
||||
*
|
||||
* @return The text range.
|
||||
*/
|
||||
TextRange getTextRange();
|
||||
|
||||
|
||||
/**
|
||||
* Gets the type of this entity.
|
||||
*
|
||||
* @return The entity type.
|
||||
*/
|
||||
String type();
|
||||
|
||||
|
||||
/**
|
||||
* Calculates the length of the entity's value.
|
||||
*
|
||||
* @return The length of the value.
|
||||
*/
|
||||
default int length() {
|
||||
|
||||
return value().length();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the value of the entity, considering any manual overwrite.
|
||||
* If no manual overwrite value is found, return the value of the entity or an empty string
|
||||
* if that value is null.
|
||||
*
|
||||
* @return The possibly overwritten value
|
||||
*/
|
||||
default String value() {
|
||||
|
||||
return getManualOverwrite().getValue()
|
||||
.orElse(getValue() == null ? "" : getValue());
|
||||
return getManualOverwrite().getValue().orElse(getValue() == null ? "" : getValue());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the entity has been applied, considering manual overwrites.
|
||||
*
|
||||
* @return True if applied, false otherwise.
|
||||
*/
|
||||
// Don't use default accessor pattern (e.g. isApplied()), as it might lead to errors in drools due to property-specific optimization of the drools planner.
|
||||
default boolean applied() {
|
||||
|
||||
return getManualOverwrite().getApplied()
|
||||
.orElse(getMatchedRule().isApplied());
|
||||
return getManualOverwrite().getApplied().orElse(getMatchedRule().isApplied());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the entity has been skipped, based on its applied status.
|
||||
*
|
||||
* @return True if skipped, false otherwise.
|
||||
*/
|
||||
default boolean skipped() {
|
||||
|
||||
return !applied();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the entity has been ignored, considering manual overwrites.
|
||||
*
|
||||
* @return True if ignored, false otherwise.
|
||||
*/
|
||||
default boolean ignored() {
|
||||
|
||||
return getManualOverwrite().getIgnored()
|
||||
.orElse(getMatchedRule().isIgnored());
|
||||
return getManualOverwrite().getIgnored().orElse(getMatchedRule().isIgnored());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the entity has been removed, considering manual overwrites.
|
||||
*
|
||||
* @return True if removed, false otherwise.
|
||||
*/
|
||||
default boolean removed() {
|
||||
|
||||
return getManualOverwrite().getRemoved()
|
||||
.orElse(getMatchedRule().isRemoved());
|
||||
return getManualOverwrite().getRemoved().orElse(getMatchedRule().isRemoved());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the entity has been resized, considering manual overwrites.
|
||||
*
|
||||
* @return True if resized, false otherwise.
|
||||
*/
|
||||
default boolean resized() {
|
||||
|
||||
return getManualOverwrite().getResized()
|
||||
.orElse(false);
|
||||
return getManualOverwrite().getResized().orElse(false);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the entity is considered active, based on its removed and ignored status.
|
||||
* An active entry is not removed or ignored.
|
||||
*
|
||||
* @return True if active, false otherwise.
|
||||
*/
|
||||
default boolean active() {
|
||||
|
||||
return !(removed() || ignored());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if there are any manual changes applied to the entity.
|
||||
*
|
||||
* @return True if there are manual changes, false otherwise.
|
||||
*/
|
||||
default boolean hasManualChanges() {
|
||||
|
||||
return !getManualOverwrite().getManualChangeLog().isEmpty();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves a set of references associated with the entity's matched rule.
|
||||
*
|
||||
* @return A set of references.
|
||||
*/
|
||||
default Set<TextEntity> references() {
|
||||
|
||||
return getMatchedRule().getReferences();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Applies a redaction to the entity with a specified legal basis.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being applied.
|
||||
* @param reason The reason for the redaction.
|
||||
* @param legalBasis The legal basis for the redaction, which must not be blank or empty.
|
||||
* @throws IllegalArgumentException If the legal basis is blank or empty.
|
||||
*/
|
||||
default void redact(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
|
||||
|
||||
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
|
||||
@@ -188,143 +97,78 @@ public interface IEntity {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Applies a rule to the entity with an optional legal basis.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being applied.
|
||||
* @param reason The reason for applying the rule.
|
||||
* @param legalBasis The legal basis for the application, can be a default or unspecified value.
|
||||
*/
|
||||
default void apply(@NonNull String ruleIdentifier, String reason, String legalBasis) {
|
||||
|
||||
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).legalBasis(legalBasis).applied(true).build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Applies a rule to the entity without specifying a legal basis, which will be replaced by "n-a".
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being applied.
|
||||
* @param reason The reason for applying the rule.
|
||||
*/
|
||||
default void apply(@NonNull String ruleIdentifier, String reason) {
|
||||
|
||||
apply(ruleIdentifier, reason, "n-a");
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Marks the entity as skipped according to a specific rule.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being skipped.
|
||||
* @param reason The reason for skipping the rule.
|
||||
*/
|
||||
default void skip(@NonNull String ruleIdentifier, String reason) {
|
||||
|
||||
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Marks the entity as removed according to a specific rule.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule based on which the entity is removed.
|
||||
* @param reason The reason for the removal.
|
||||
*/
|
||||
default void remove(String ruleIdentifier, String reason) {
|
||||
|
||||
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).removed(true).build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Marks the entity as ignored according to a specific rule.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule based on which the entity is removed.
|
||||
* @param reason The reason for the removal.
|
||||
*/
|
||||
default void ignore(String ruleIdentifier, String reason) {
|
||||
|
||||
addMatchedRule(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).ignored(true).build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Applies a rule to the entity, indicating that the value should be written with line breaks.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being applied.
|
||||
* @param reason The reason for the rule application.
|
||||
* @param legalBasis The legal basis for the rule, which must not be empty.
|
||||
* @throws IllegalArgumentException If the legal basis is blank or empty.
|
||||
*/
|
||||
default void applyWithLineBreaks(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis) {
|
||||
|
||||
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
|
||||
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
|
||||
}
|
||||
getMatchedRuleList().add(MatchedRule.builder()
|
||||
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.reason(reason)
|
||||
.legalBasis(legalBasis)
|
||||
.applied(true)
|
||||
.writeValueWithLineBreaks(true)
|
||||
.build());
|
||||
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.reason(reason)
|
||||
.legalBasis(legalBasis)
|
||||
.applied(true)
|
||||
.writeValueWithLineBreaks(true)
|
||||
.build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Applies a rule to the entity with a collection of references.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being applied.
|
||||
* @param reason The reason for the rule application.
|
||||
* @param legalBasis The legal basis for the rule, which must not be empty.
|
||||
* @param references A collection of text entities that are referenced by this rule application.
|
||||
* @throws IllegalArgumentException If the legal basis is blank or empty.
|
||||
*/
|
||||
default void applyWithReferences(@NonNull String ruleIdentifier, String reason, @NonNull String legalBasis, Collection<TextEntity> references) {
|
||||
|
||||
if (legalBasis.isBlank() || legalBasis.isEmpty()) {
|
||||
throw new IllegalArgumentException("legal basis cannot be empty when redacting an entity");
|
||||
}
|
||||
getMatchedRuleList().add(MatchedRule.builder()
|
||||
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.reason(reason)
|
||||
.legalBasis(legalBasis)
|
||||
.applied(true)
|
||||
.references(new HashSet<>(references))
|
||||
.build());
|
||||
.ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.reason(reason)
|
||||
.legalBasis(legalBasis)
|
||||
.applied(true)
|
||||
.references(new HashSet<>(references))
|
||||
.build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Marks the entity as skipped for a specific rule and associates a collection of references.
|
||||
*
|
||||
* @param ruleIdentifier The identifier of the rule being skipped.
|
||||
* @param reason The reason for skipping the rule.
|
||||
* @param references A collection of text entities that are referenced by the skipped rule.
|
||||
*/
|
||||
default void skipWithReferences(@NonNull String ruleIdentifier, String reason, Collection<TextEntity> references) {
|
||||
|
||||
getMatchedRuleList().add(MatchedRule.builder().ruleIdentifier(RuleIdentifier.fromString(ruleIdentifier)).reason(reason).references(new HashSet<>(references)).build());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Adds a single matched rule to this entity.
|
||||
*
|
||||
* @param matchedRule The matched rule to add.
|
||||
*/
|
||||
default void addMatchedRule(MatchedRule matchedRule) {
|
||||
|
||||
getMatchedRuleList().add(matchedRule);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Adds a collection of matched rules to this entity.
|
||||
*
|
||||
* @param matchedRules The collection of matched rules to add.
|
||||
*/
|
||||
default void addMatchedRules(Collection<MatchedRule> matchedRules) {
|
||||
|
||||
if (getMatchedRuleList().equals(matchedRules)) {
|
||||
@@ -334,22 +178,12 @@ public interface IEntity {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the 'unit' value of the highest priority matched rule.
|
||||
*
|
||||
* @return The unit value of the matched rule.
|
||||
*/
|
||||
default int getMatchedRuleUnit() {
|
||||
|
||||
return getMatchedRule().getRuleIdentifier().unit();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Gets the highest priority matched rule for this entity.
|
||||
*
|
||||
* @return The matched rule.
|
||||
*/
|
||||
default MatchedRule getMatchedRule() {
|
||||
|
||||
if (getMatchedRuleList().isEmpty()) {
|
||||
@@ -359,11 +193,6 @@ public interface IEntity {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Builds a reason string for this entity, incorporating descriptions from manual changes.
|
||||
*
|
||||
* @return The built reason string.
|
||||
*/
|
||||
default String buildReasonWithManualChangeDescriptions() {
|
||||
|
||||
if (getManualOverwrite().getDescriptions().isEmpty()) {
|
||||
@@ -376,15 +205,9 @@ public interface IEntity {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Retrieves the legal basis for the action taken on this entity, considering any manual overwrite.
|
||||
*
|
||||
* @return The legal basis.
|
||||
*/
|
||||
default String legalBasis() {
|
||||
|
||||
return getManualOverwrite().getLegalBasis()
|
||||
.orElse(getMatchedRule().getLegalBasis());
|
||||
return getManualOverwrite().getLegalBasis().orElse(getMatchedRule().getLegalBasis());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-2
@@ -131,8 +131,6 @@ public class ManualChangeOverwrite {
|
||||
if (manualChange instanceof ManualRecategorization recategorization) {
|
||||
recategorized = true;
|
||||
type = recategorization.getType();
|
||||
section = recategorization.getSection();
|
||||
value = recategorization.getValue();
|
||||
if (recategorization.getLegalBasis() != null && !recategorization.getLegalBasis().isEmpty()) {
|
||||
legalBasis = recategorization.getLegalBasis();
|
||||
}
|
||||
|
||||
+3
-41
@@ -15,9 +15,6 @@ import lombok.EqualsAndHashCode;
|
||||
import lombok.Getter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a rule that has been matched during the document redaction process.
|
||||
*/
|
||||
@Getter
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -28,8 +25,7 @@ public final class MatchedRule implements Comparable<MatchedRule> {
|
||||
public static final RuleType FINAL_TYPE = RuleType.fromString("FINAL");
|
||||
public static final RuleType ELIMINATION_RULE_TYPE = RuleType.fromString("X");
|
||||
public static final RuleType IMPORTED_TYPE = RuleType.fromString("IMP");
|
||||
public static final RuleType DICTIONARY_TYPE = RuleType.fromString("DICT");
|
||||
private static final List<RuleType> RULE_TYPE_PRIORITIES = List.of(FINAL_TYPE, ELIMINATION_RULE_TYPE, IMPORTED_TYPE, DICTIONARY_TYPE);
|
||||
private static final List<RuleType> RULE_TYPE_PRIORITIES = List.of(FINAL_TYPE, ELIMINATION_RULE_TYPE, IMPORTED_TYPE);
|
||||
|
||||
RuleIdentifier ruleIdentifier;
|
||||
@Builder.Default
|
||||
@@ -45,33 +41,18 @@ public final class MatchedRule implements Comparable<MatchedRule> {
|
||||
Set<TextEntity> references = Collections.emptySet();
|
||||
|
||||
|
||||
/**
|
||||
* Creates an empty instance of {@link MatchedRule}.
|
||||
* This can be used as a placeholder or when no rule is actually matched.
|
||||
*
|
||||
* @return An empty {@link MatchedRule} instance.
|
||||
*/
|
||||
public static MatchedRule empty() {
|
||||
|
||||
return MatchedRule.builder().ruleIdentifier(RuleIdentifier.empty()).build();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns a modified instance of {@link MatchedRule} based on its applied status.
|
||||
* If the rule has been applied, it returns a new {@link MatchedRule} instance that retains all properties of the original
|
||||
* except for the 'applied' status, which is set to false.
|
||||
* If the rule has not been applied, it returns the original instance.
|
||||
*
|
||||
* @return A {@link MatchedRule} instance with 'applied' set to false.
|
||||
*/
|
||||
public MatchedRule asSkippedIfApplied() {
|
||||
|
||||
if (!this.isApplied()) {
|
||||
return this;
|
||||
}
|
||||
return MatchedRule.builder()
|
||||
.ruleIdentifier(getRuleIdentifier())
|
||||
return MatchedRule.builder().ruleIdentifier(getRuleIdentifier())
|
||||
.writeValueWithLineBreaks(this.isWriteValueWithLineBreaks())
|
||||
.legalBasis(this.getLegalBasis())
|
||||
.reason(this.getReason())
|
||||
@@ -80,13 +61,6 @@ public final class MatchedRule implements Comparable<MatchedRule> {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Compares this rule with another {@link MatchedRule} to establish a priority order.
|
||||
* The comparison is based on the rule type, unit, and ID, in that order.
|
||||
*
|
||||
* @param matchedRule The {@link MatchedRule} to compare against.
|
||||
* @return A negative integer, zero, or a positive integer as this rule is less than, equal to, or greater than the specified rule.
|
||||
*/
|
||||
@Override
|
||||
public int compareTo(MatchedRule matchedRule) {
|
||||
|
||||
@@ -123,19 +97,7 @@ public final class MatchedRule implements Comparable<MatchedRule> {
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return "MatchedRule[ruleIdentifier="
|
||||
+ ruleIdentifier
|
||||
+ ", reason="
|
||||
+ reason
|
||||
+ ", legalBasis="
|
||||
+ legalBasis
|
||||
+ ", applied="
|
||||
+ applied
|
||||
+ ", writeValueWithLineBreaks="
|
||||
+ writeValueWithLineBreaks
|
||||
+ ", references="
|
||||
+ references
|
||||
+ ']';
|
||||
return "MatchedRule[ruleIdentifier=" + ruleIdentifier + ", reason=" + reason + ", legalBasis=" + legalBasis + ", applied=" + applied + ", writeValueWithLineBreaks=" + writeValueWithLineBreaks + ", references=" + references + ']';
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+12
-51
@@ -40,7 +40,7 @@ public class TextEntity implements IEntity {
|
||||
TextRange textRange;
|
||||
@Builder.Default
|
||||
List<TextRange> duplicateTextRanges = new ArrayList<>();
|
||||
String type; // TODO: make final once ManualChangesApplicationService::recategorize is deleted
|
||||
String type; // TODO: make final once ManualChangesApplicatioService recategorize is deleted
|
||||
final EntityType entityType;
|
||||
|
||||
@Builder.Default
|
||||
@@ -67,13 +67,7 @@ public class TextEntity implements IEntity {
|
||||
|
||||
public static TextEntity initialEntityNode(TextRange textRange, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return TextEntity.builder()
|
||||
.id(buildId(node, textRange, type, entityType))
|
||||
.type(type)
|
||||
.entityType(entityType)
|
||||
.textRange(textRange)
|
||||
.manualOverwrite(new ManualChangeOverwrite(entityType))
|
||||
.build();
|
||||
return TextEntity.builder().id(buildId(node, textRange, type, entityType)).type(type).entityType(entityType).textRange(textRange).manualOverwrite(new ManualChangeOverwrite(entityType)).build();
|
||||
}
|
||||
|
||||
|
||||
@@ -86,13 +80,7 @@ public class TextEntity implements IEntity {
|
||||
private static String buildId(SemanticNode node, TextRange textRange, String type, EntityType entityType) {
|
||||
|
||||
Map<Page, List<Rectangle2D>> rectanglesPerLinePerPage = node.getPositionsPerPage(textRange);
|
||||
return IdBuilder.buildId(rectanglesPerLinePerPage.keySet(),
|
||||
rectanglesPerLinePerPage.values()
|
||||
.stream()
|
||||
.flatMap(Collection::stream)
|
||||
.toList(),
|
||||
type,
|
||||
entityType.name());
|
||||
return IdBuilder.buildId(rectanglesPerLinePerPage.keySet(), rectanglesPerLinePerPage.values().stream().flatMap(Collection::stream).toList(), type, entityType.name());
|
||||
}
|
||||
|
||||
|
||||
@@ -101,18 +89,15 @@ public class TextEntity implements IEntity {
|
||||
duplicateTextRanges.add(textRange);
|
||||
}
|
||||
|
||||
|
||||
public boolean occursInNodeOfType(Class<? extends SemanticNode> clazz) {
|
||||
|
||||
return intersectingNodes.stream()
|
||||
.anyMatch(clazz::isInstance);
|
||||
return intersectingNodes.stream().anyMatch(clazz::isInstance);
|
||||
}
|
||||
|
||||
|
||||
public boolean occursInNode(SemanticNode semanticNode) {
|
||||
|
||||
return intersectingNodes.stream()
|
||||
.anyMatch(node -> node.equals(semanticNode));
|
||||
return intersectingNodes.stream().anyMatch(node -> node.equals(semanticNode));
|
||||
}
|
||||
|
||||
|
||||
@@ -161,10 +146,7 @@ public class TextEntity implements IEntity {
|
||||
.min(Comparator.comparingInt(Page::getNumber))
|
||||
.orElseThrow(() -> new RuntimeException("No Positions found on any page!"));
|
||||
|
||||
positionsOnPagePerPage = rectanglesPerLinePerPage.entrySet()
|
||||
.stream()
|
||||
.map(entry -> buildPositionOnPage(firstPage, id, entry))
|
||||
.toList();
|
||||
positionsOnPagePerPage = rectanglesPerLinePerPage.entrySet().stream().map(entry -> buildPositionOnPage(firstPage, id, entry)).toList();
|
||||
}
|
||||
return positionsOnPagePerPage;
|
||||
}
|
||||
@@ -182,37 +164,19 @@ public class TextEntity implements IEntity {
|
||||
|
||||
public boolean containedBy(TextEntity textEntity) {
|
||||
|
||||
return this.textRange.containedBy(textEntity.getTextRange()) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> duplicateTextRange.containedBy(textEntity.textRange)) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
|
||||
.stream()
|
||||
.anyMatch(duplicateTextRange::containedBy));
|
||||
return this.textRange.containedBy(textEntity.getTextRange());
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(TextEntity textEntity) {
|
||||
|
||||
return this.textRange.contains(textEntity.getTextRange()) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> duplicateTextRange.contains(textEntity.textRange)) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
|
||||
.stream()
|
||||
.anyMatch(duplicateTextRange::contains));
|
||||
return this.textRange.contains(textEntity.getTextRange());
|
||||
}
|
||||
|
||||
|
||||
public boolean intersects(TextEntity textEntity) {
|
||||
|
||||
return this.textRange.intersects(textEntity.getTextRange()) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> duplicateTextRange.intersects(textEntity.textRange)) //
|
||||
|| duplicateTextRanges.stream()
|
||||
.anyMatch(duplicateTextRange -> textEntity.getDuplicateTextRanges()
|
||||
.stream()
|
||||
.anyMatch(duplicateTextRange::intersects));
|
||||
return this.textRange.intersects(textEntity.getTextRange());
|
||||
}
|
||||
|
||||
|
||||
@@ -230,8 +194,7 @@ public class TextEntity implements IEntity {
|
||||
|
||||
public boolean matchesAnnotationId(String manualRedactionId) {
|
||||
|
||||
return getPositionsOnPagePerPage().stream()
|
||||
.anyMatch(entityPosition -> entityPosition.getId().equals(manualRedactionId));
|
||||
return getPositionsOnPagePerPage().stream().anyMatch(entityPosition -> entityPosition.getId().equals(manualRedactionId));
|
||||
}
|
||||
|
||||
|
||||
@@ -261,16 +224,14 @@ public class TextEntity implements IEntity {
|
||||
@Override
|
||||
public String type() {
|
||||
|
||||
return getManualOverwrite().getType()
|
||||
.orElse(type);
|
||||
return getManualOverwrite().getType().orElse(type);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String value() {
|
||||
|
||||
return getManualOverwrite().getValue()
|
||||
.orElse(getMatchedRule().isWriteValueWithLineBreaks() ? getValueWithLineBreaks() : value);
|
||||
return getManualOverwrite().getValue().orElse(getMatchedRule().isWriteValueWithLineBreaks() ? getValueWithLineBreaks() : value);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+5
-32
@@ -24,9 +24,6 @@ import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents the entire document as a node within the document's semantic structure.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -60,32 +57,21 @@ public class Document implements GenericSemanticNode {
|
||||
public TextBlock getTextBlock() {
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = GenericSemanticNode.super.getTextBlock();
|
||||
textBlock = streamTerminalTextBlocksInOrder().collect(new TextBlockCollector());
|
||||
}
|
||||
return textBlock;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Gets the main sections of the document as a list.
|
||||
*
|
||||
* @return A list of main sections within the document.
|
||||
*/
|
||||
public List<Section> getMainSections() {
|
||||
|
||||
return streamChildrenOfType(NodeType.SECTION).map(node -> (Section) node)
|
||||
.collect(Collectors.toList());
|
||||
return streamChildrenOfType(NodeType.SECTION).map(node -> (Section) node).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Streams all terminal (leaf) text blocks within the document in their natural order.
|
||||
*
|
||||
* @return A stream of terminal {@link TextBlock}.
|
||||
*/
|
||||
public Stream<TextBlock> streamTerminalTextBlocksInOrder() {
|
||||
|
||||
return streamAllNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getTextBlock);
|
||||
return streamAllNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock);
|
||||
}
|
||||
|
||||
|
||||
@@ -106,29 +92,16 @@ public class Document implements GenericSemanticNode {
|
||||
@Override
|
||||
public Headline getHeadline() {
|
||||
|
||||
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node)
|
||||
.findFirst()
|
||||
.orElseGet(Headline::empty);
|
||||
return streamAllSubNodesOfType(NodeType.HEADLINE).map(node -> (Headline) node).findFirst().orElseGet(Headline::empty);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Streams all nodes within the document, regardless of type, in their natural order.
|
||||
*
|
||||
* @return A stream of all {@link SemanticNode} within the document.
|
||||
*/
|
||||
private Stream<SemanticNode> streamAllNodes() {
|
||||
|
||||
return documentTree.allEntriesInOrder()
|
||||
.map(DocumentTree.Entry::getNode);
|
||||
return documentTree.allEntriesInOrder().map(DocumentTree.Entry::getNode);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Streams all image nodes contained within the document.
|
||||
*
|
||||
* @return A stream of {@link Image} nodes.
|
||||
*/
|
||||
public Stream<Image> streamAllImages() {
|
||||
|
||||
return streamAllSubNodesOfType(NodeType.IMAGE).map(node -> (Image) node);
|
||||
|
||||
-34
@@ -1,34 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.model.document.nodes;
|
||||
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlockCollector;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
@SuperBuilder
|
||||
public class DuplicatedParagraph extends Paragraph {
|
||||
|
||||
TextBlock unsortedLeafTextBlock;
|
||||
|
||||
|
||||
@Override
|
||||
public TextBlock getTextBlock() {
|
||||
|
||||
return Stream.of(leafTextBlock, unsortedLeafTextBlock).collect(new TextBlockCollector());
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return super.toString();
|
||||
}
|
||||
|
||||
}
|
||||
-3
@@ -19,9 +19,6 @@ import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents the header part of a document page.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
|
||||
+1
-16
@@ -20,9 +20,6 @@ import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a headline in a document.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -101,27 +98,15 @@ public class Headline implements GenericSemanticNode {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates an empty headline with no text content.
|
||||
*
|
||||
* @return An empty {@link Headline} instance.
|
||||
*/
|
||||
public static Headline empty() {
|
||||
|
||||
return Headline.builder().leafTextBlock(AtomicTextBlock.empty(-1L, 0, new Page(), -1, null)).build();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this headline is associated with any paragraphs within its parent section or node.
|
||||
*
|
||||
* @return True if there are paragraphs associated with this headline, false otherwise.
|
||||
*/
|
||||
public boolean hasParagraphs() {
|
||||
|
||||
return getParent().streamAllSubNodesOfType(NodeType.PARAGRAPH)
|
||||
.findFirst()
|
||||
.isPresent();
|
||||
return getParent().streamAllSubNodesOfType(NodeType.PARAGRAPH).findFirst().isPresent();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+3
-7
@@ -28,10 +28,6 @@ import lombok.EqualsAndHashCode;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
*
|
||||
Represents an image within the document.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -102,14 +98,14 @@ public class Image implements GenericSemanticNode, IEntity {
|
||||
public String type() {
|
||||
|
||||
return getManualOverwrite().getType()
|
||||
.orElse(imageType.toString().toLowerCase(Locale.ENGLISH));
|
||||
.orElse(imageType.toString());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return treeId + ": " + getValue() + " " + position;
|
||||
return treeId + ": " + NodeType.IMAGE + ": " + imageType.toString() + " " + position;
|
||||
}
|
||||
|
||||
|
||||
@@ -140,7 +136,7 @@ public class Image implements GenericSemanticNode, IEntity {
|
||||
Map<Page, Rectangle2D> bboxImage = image.getBBox();
|
||||
Map<Page, Rectangle2D> bbox = this.getBBox();
|
||||
//image needs to be on the same page
|
||||
if (bboxImage.get(this.page) != null) {
|
||||
if(bboxImage.get(this.page) != null) {
|
||||
Rectangle2D intersection = bboxImage.get(this.page).createIntersection(bbox.get(this.page));
|
||||
double calculatedIntersection = intersection.getWidth() * intersection.getHeight();
|
||||
double area = bbox.get(this.page).getWidth() * bbox.get(this.page).getHeight();
|
||||
|
||||
+13
-1
@@ -6,10 +6,22 @@ public enum ImageType {
|
||||
LOGO,
|
||||
FORMULA,
|
||||
SIGNATURE,
|
||||
OTHER,
|
||||
OTHER {
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return "image";
|
||||
}
|
||||
},
|
||||
OCR;
|
||||
|
||||
|
||||
public String toString() {
|
||||
|
||||
return name().toLowerCase(Locale.ENGLISH);
|
||||
}
|
||||
|
||||
|
||||
public static ImageType fromString(String imageType) {
|
||||
|
||||
return switch (imageType.toLowerCase(Locale.ROOT)) {
|
||||
|
||||
+1
-13
@@ -17,9 +17,6 @@ import lombok.NoArgsConstructor;
|
||||
import lombok.Setter;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a single page in a document.
|
||||
*/
|
||||
@Getter
|
||||
@Setter
|
||||
@Builder
|
||||
@@ -46,17 +43,9 @@ public class Page {
|
||||
Set<Image> images = new HashSet<>();
|
||||
|
||||
|
||||
/**
|
||||
* Constructs and returns a {@link TextBlock} representing the concatenated text of all leaf semantic nodes in the main body.
|
||||
*
|
||||
* @return The main body text block.
|
||||
*/
|
||||
public TextBlock getMainBodyTextBlock() {
|
||||
|
||||
return mainBody.stream()
|
||||
.filter(SemanticNode::isLeaf)
|
||||
.map(SemanticNode::getLeafTextBlock)
|
||||
.collect(new TextBlockCollector());
|
||||
return mainBody.stream().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@@ -65,5 +54,4 @@ public class Page {
|
||||
|
||||
return String.valueOf(number);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+2
-6
@@ -17,15 +17,11 @@ import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.experimental.SuperBuilder;
|
||||
|
||||
/**
|
||||
* Represents a paragraph in the document.
|
||||
*/
|
||||
@Data
|
||||
@SuperBuilder
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PROTECTED)
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
|
||||
public class Paragraph implements GenericSemanticNode {
|
||||
|
||||
|
||||
+2
-23
@@ -21,9 +21,6 @@ import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Represents a section within a document, encapsulating both its textual content and semantic structure.
|
||||
*/
|
||||
@Slf4j
|
||||
@Data
|
||||
@Builder
|
||||
@@ -54,15 +51,9 @@ public class Section implements GenericSemanticNode {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if this section contains any tables.
|
||||
*
|
||||
* @return True if the section contains at least one table, false otherwise.
|
||||
*/
|
||||
public boolean hasTables() {
|
||||
|
||||
return streamAllSubNodesOfType(NodeType.TABLE).findAny()
|
||||
.isPresent();
|
||||
return streamAllSubNodesOfType(NodeType.TABLE).findAny().isPresent();
|
||||
}
|
||||
|
||||
|
||||
@@ -77,7 +68,7 @@ public class Section implements GenericSemanticNode {
|
||||
public TextBlock getTextBlock() {
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = GenericSemanticNode.super.getTextBlock();
|
||||
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
return textBlock;
|
||||
}
|
||||
@@ -99,24 +90,12 @@ public class Section implements GenericSemanticNode {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if any headline within this section or its sub-nodes contains a given string.
|
||||
*
|
||||
* @param value The string to search for within headlines, case-sensitive.
|
||||
* @return True if at least one headline contains the specified string, false otherwise.
|
||||
*/
|
||||
public boolean anyHeadlineContainsString(String value) {
|
||||
|
||||
return streamAllSubNodesOfType(NodeType.HEADLINE).anyMatch(h -> h.containsString(value));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if any headline within this section or its sub-nodes contains a given string, case-insensitive.
|
||||
*
|
||||
* @param value The string to search for within headlines, case-insensitive.
|
||||
* @return True if at least one headline contains the specified string, false otherwise.
|
||||
*/
|
||||
public boolean anyHeadlineContainsStringIgnoreCase(String value) {
|
||||
|
||||
return streamAllSubNodesOfType(NodeType.HEADLINE).anyMatch(h -> h.containsStringIgnoreCase(value));
|
||||
|
||||
+1
-36
@@ -10,9 +10,6 @@ import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a unique identifier for a section within a document.
|
||||
*/
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class SectionIdentifier {
|
||||
@@ -31,12 +28,6 @@ public class SectionIdentifier {
|
||||
boolean asChild;
|
||||
|
||||
|
||||
/**
|
||||
* Generates a SectionIdentifier from the headline text of a section, determining its format and structure.
|
||||
*
|
||||
* @param headline The headline text from which to generate the section identifier.
|
||||
* @return A {@link SectionIdentifier} instance corresponding to the headline text.
|
||||
*/
|
||||
public static SectionIdentifier fromSearchText(String headline) {
|
||||
|
||||
if (headline == null || headline.isEmpty() || headline.isBlank()) {
|
||||
@@ -52,34 +43,18 @@ public class SectionIdentifier {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Marks the current section identifier as a child of another section.
|
||||
*
|
||||
* @param sectionIdentifier The parent section identifier.
|
||||
* @return A new {@link SectionIdentifier} instance marked as a child.
|
||||
*/
|
||||
public static SectionIdentifier asChildOf(SectionIdentifier sectionIdentifier) {
|
||||
|
||||
return new SectionIdentifier(sectionIdentifier.format, sectionIdentifier.toString(), sectionIdentifier.identifiers, true);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Generates a SectionIdentifier that represents the entire document.
|
||||
*
|
||||
* @return A {@link SectionIdentifier} with a document-wide scope.
|
||||
*/
|
||||
public static SectionIdentifier document() {
|
||||
|
||||
return new SectionIdentifier(Format.DOCUMENT, "document", Collections.emptyList(), false);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Generates an empty SectionIdentifier.
|
||||
*
|
||||
* @return An empty {@link SectionIdentifier} instance.
|
||||
*/
|
||||
public static SectionIdentifier empty() {
|
||||
|
||||
return new SectionIdentifier(Format.EMPTY, "empty", Collections.emptyList(), false);
|
||||
@@ -97,11 +72,7 @@ public class SectionIdentifier {
|
||||
}
|
||||
identifiers.add(Integer.parseInt(numericalIdentifier.trim()));
|
||||
}
|
||||
return new SectionIdentifier(Format.NUMERICAL,
|
||||
identifierString,
|
||||
identifiers.stream()
|
||||
.toList(),
|
||||
false);
|
||||
return new SectionIdentifier(Format.NUMERICAL, identifierString, identifiers.stream().toList(), false);
|
||||
}
|
||||
|
||||
|
||||
@@ -134,12 +105,6 @@ public class SectionIdentifier {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Determines if the current section is a child of the given section, based on their identifiers.
|
||||
*
|
||||
* @param sectionIdentifier The section identifier to compare against.
|
||||
* @return True if the current section is a child of the given section, false otherwise.
|
||||
*/
|
||||
public boolean isChildOf(SectionIdentifier sectionIdentifier) {
|
||||
|
||||
if (this.format.equals(Format.DOCUMENT) || this.format.equals(Format.EMPTY)) {
|
||||
|
||||
+38
-111
@@ -19,7 +19,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.TextRange;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntity;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBlockCollector;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.NodeVisitor;
|
||||
import com.iqser.red.service.redaction.v1.server.utils.RectangleTransformations;
|
||||
import com.iqser.red.service.redaction.v1.server.utils.RedactionSearchUtility;
|
||||
@@ -42,12 +41,7 @@ public interface SemanticNode {
|
||||
*
|
||||
* @return TextBlock containing all AtomicTextBlocks that are located under this Node.
|
||||
*/
|
||||
default TextBlock getTextBlock() {
|
||||
|
||||
return streamAllSubNodes().filter(SemanticNode::isLeaf)
|
||||
.map(SemanticNode::getTextBlock)
|
||||
.collect(new TextBlockCollector());
|
||||
}
|
||||
TextBlock getTextBlock();
|
||||
|
||||
|
||||
/**
|
||||
@@ -77,10 +71,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default Page getFirstPage() {
|
||||
|
||||
return getTextBlock().getPages()
|
||||
.stream()
|
||||
.min(Comparator.comparingInt(Page::getNumber))
|
||||
.orElseThrow();
|
||||
return getTextBlock().getPages().stream().min(Comparator.comparingInt(Page::getNumber)).orElseThrow();
|
||||
}
|
||||
|
||||
|
||||
@@ -106,8 +97,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean onPage(int pageNumber) {
|
||||
|
||||
return getPages().stream()
|
||||
.anyMatch(page -> page.getNumber() == pageNumber);
|
||||
return getPages().stream().anyMatch(page -> page.getNumber() == pageNumber);
|
||||
}
|
||||
|
||||
|
||||
@@ -259,9 +249,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean hasEntitiesOfType(String type) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.anyMatch(redactionEntity -> redactionEntity.type().equals(type));
|
||||
return getEntities().stream().filter(TextEntity::active).anyMatch(redactionEntity -> redactionEntity.type().equals(type));
|
||||
}
|
||||
|
||||
|
||||
@@ -274,10 +262,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean hasEntitiesOfAnyType(String... types) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.anyMatch(redactionEntity -> Arrays.stream(types)
|
||||
.anyMatch(type -> redactionEntity.type().equals(type)));
|
||||
return getEntities().stream().filter(TextEntity::active).anyMatch(redactionEntity -> Arrays.stream(types).anyMatch(type -> redactionEntity.type().equals(type)));
|
||||
}
|
||||
|
||||
|
||||
@@ -290,12 +275,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean hasEntitiesOfAllTypes(String... types) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.map(TextEntity::type)
|
||||
.collect(Collectors.toUnmodifiableSet())
|
||||
.containsAll(Arrays.stream(types)
|
||||
.toList());
|
||||
return getEntities().stream().filter(TextEntity::active).map(TextEntity::type).collect(Collectors.toUnmodifiableSet()).containsAll(Arrays.stream(types).toList());
|
||||
}
|
||||
|
||||
|
||||
@@ -308,10 +288,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default List<TextEntity> getEntitiesOfType(String type) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.filter(redactionEntity -> redactionEntity.type().equals(type))
|
||||
.toList();
|
||||
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.type().equals(type)).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -324,10 +301,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default List<TextEntity> getEntitiesOfType(List<String> types) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.filter(redactionEntity -> redactionEntity.isAnyType(types))
|
||||
.toList();
|
||||
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.isAnyType(types)).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -340,11 +314,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default List<TextEntity> getEntitiesOfType(String... types) {
|
||||
|
||||
return getEntities().stream()
|
||||
.filter(TextEntity::active)
|
||||
.filter(redactionEntity -> redactionEntity.isAnyType(Arrays.stream(types)
|
||||
.toList()))
|
||||
.toList();
|
||||
return getEntities().stream().filter(TextEntity::active).filter(redactionEntity -> redactionEntity.isAnyType(Arrays.stream(types).toList())).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -358,8 +328,7 @@ public interface SemanticNode {
|
||||
|
||||
TextBlock textBlock = getTextBlock();
|
||||
if (!textBlock.getAtomicTextBlocks().isEmpty()) {
|
||||
return getTextBlock().getAtomicTextBlocks()
|
||||
.get(0).getNumberOnPage();
|
||||
return getTextBlock().getAtomicTextBlocks().get(0).getNumberOnPage();
|
||||
} else {
|
||||
return -1;
|
||||
}
|
||||
@@ -388,16 +357,14 @@ public interface SemanticNode {
|
||||
return getTextBlock().getSearchText().contains(string);
|
||||
}
|
||||
|
||||
|
||||
Set<LayoutEngine> getEngines();
|
||||
|
||||
|
||||
default void addEngine(LayoutEngine engine) {
|
||||
|
||||
getEngines().add(engine);
|
||||
}
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains all the provided Strings.
|
||||
*
|
||||
@@ -406,8 +373,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsAllStrings(String... strings) {
|
||||
|
||||
return Arrays.stream(strings)
|
||||
.allMatch(this::containsString);
|
||||
return Arrays.stream(strings).allMatch(this::containsString);
|
||||
}
|
||||
|
||||
|
||||
@@ -419,8 +385,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsAnyString(String... strings) {
|
||||
|
||||
return Arrays.stream(strings)
|
||||
.anyMatch(this::containsString);
|
||||
return Arrays.stream(strings).anyMatch(this::containsString);
|
||||
}
|
||||
|
||||
|
||||
@@ -432,16 +397,15 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsAnyString(List<String> strings) {
|
||||
|
||||
return strings.stream()
|
||||
.anyMatch(this::containsString);
|
||||
return strings.stream().anyMatch(this::containsString);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains all the provided Strings case-insensitive.
|
||||
* Checks whether this SemanticNode contains all the provided Strings ignoring case.
|
||||
*
|
||||
* @param string A String which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains the string case-insensitive
|
||||
* @return true, if this node's TextBlock contains the string ignoring case
|
||||
*/
|
||||
default boolean containsStringIgnoreCase(String string) {
|
||||
|
||||
@@ -450,28 +414,26 @@ public interface SemanticNode {
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains any of the provided Strings case-insensitive.
|
||||
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
|
||||
*
|
||||
* @param strings A List of Strings which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains any of the strings
|
||||
*/
|
||||
default boolean containsAnyStringIgnoreCase(String... strings) {
|
||||
|
||||
return Arrays.stream(strings)
|
||||
.anyMatch(this::containsStringIgnoreCase);
|
||||
return Arrays.stream(strings).anyMatch(this::containsStringIgnoreCase);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains any of the provided Strings case-insensitive.
|
||||
* Checks whether this SemanticNode contains any of the provided Strings ignoring case.
|
||||
*
|
||||
* @param strings A List of Strings which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains any of the strings
|
||||
*/
|
||||
default boolean containsAllStringsIgnoreCase(String... strings) {
|
||||
|
||||
return Arrays.stream(strings)
|
||||
.allMatch(this::containsStringIgnoreCase);
|
||||
return Arrays.stream(strings).allMatch(this::containsStringIgnoreCase);
|
||||
}
|
||||
|
||||
|
||||
@@ -483,24 +445,19 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsWord(String word) {
|
||||
|
||||
return getTextBlock().getWords()
|
||||
.stream()
|
||||
.anyMatch(s -> s.equals(word));
|
||||
return getTextBlock().getWords().stream().anyMatch(s -> s.equals(word));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains exactly the provided String as a word case-insensitive.
|
||||
* Checks whether this SemanticNode contains exactly the provided String as a word ignoring case.
|
||||
*
|
||||
* @param word - String which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains string
|
||||
*/
|
||||
default boolean containsWordIgnoreCase(String word) {
|
||||
|
||||
return getTextBlock().getWords()
|
||||
.stream()
|
||||
.map(String::toLowerCase)
|
||||
.anyMatch(s -> s.equals(word.toLowerCase(Locale.ENGLISH)));
|
||||
return getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(s -> s.equals(word.toLowerCase(Locale.ENGLISH)));
|
||||
}
|
||||
|
||||
|
||||
@@ -512,27 +469,19 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsAnyWord(String... words) {
|
||||
|
||||
return Arrays.stream(words)
|
||||
.anyMatch(word -> getTextBlock().getWords()
|
||||
.stream()
|
||||
.anyMatch(word::equals));
|
||||
return Arrays.stream(words).anyMatch(word -> getTextBlock().getWords().stream().anyMatch(word::equals));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains any of the provided Strings as a word case-insensitive.
|
||||
* Checks whether this SemanticNode contains any of the provided Strings as a word ignoring case.
|
||||
*
|
||||
* @param words - A List of Strings which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains any of the provided strings
|
||||
*/
|
||||
default boolean containsAnyWordIgnoreCase(String... words) {
|
||||
|
||||
return Arrays.stream(words)
|
||||
.map(String::toLowerCase)
|
||||
.anyMatch(word -> getTextBlock().getWords()
|
||||
.stream()
|
||||
.map(String::toLowerCase)
|
||||
.anyMatch(word::equals));
|
||||
return Arrays.stream(words).map(String::toLowerCase).anyMatch(word -> getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(word::equals));
|
||||
}
|
||||
|
||||
|
||||
@@ -544,27 +493,19 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean containsAllWords(String... words) {
|
||||
|
||||
return Arrays.stream(words)
|
||||
.allMatch(word -> getTextBlock().getWords()
|
||||
.stream()
|
||||
.anyMatch(word::equals));
|
||||
return Arrays.stream(words).allMatch(word -> getTextBlock().getWords().stream().anyMatch(word::equals));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode contains all the provided Strings as word case-insensitive.
|
||||
* Checks whether this SemanticNode contains all the provided Strings as word ignoring case.
|
||||
*
|
||||
* @param words - A List of Strings which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains all the provided strings
|
||||
*/
|
||||
default boolean containsAllWordsIgnoreCase(String... words) {
|
||||
|
||||
return Arrays.stream(words)
|
||||
.map(String::toLowerCase)
|
||||
.allMatch(word -> getTextBlock().getWords()
|
||||
.stream()
|
||||
.map(String::toLowerCase)
|
||||
.anyMatch(word::equals));
|
||||
return Arrays.stream(words).map(String::toLowerCase).allMatch(word -> getTextBlock().getWords().stream().map(String::toLowerCase).anyMatch(word::equals));
|
||||
}
|
||||
|
||||
|
||||
@@ -581,10 +522,10 @@ public interface SemanticNode {
|
||||
|
||||
|
||||
/**
|
||||
* Checks whether this SemanticNode matches the provided regex pattern case-insensitive.
|
||||
* Checks whether this SemanticNode matches the provided regex pattern ignoring case.
|
||||
*
|
||||
* @param regexPattern A String representing a regex pattern, which the TextBlock might contain
|
||||
* @return true, if this node's TextBlock contains the regex pattern case-insensitive
|
||||
* @return true, if this node's TextBlock contains the regex pattern ignoring case
|
||||
*/
|
||||
default boolean matchesRegexIgnoreCase(String regexPattern) {
|
||||
|
||||
@@ -604,11 +545,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default boolean intersectsRectangle(int x, int y, int w, int h, int pageNumber) {
|
||||
|
||||
return getBBox().entrySet()
|
||||
.stream()
|
||||
.filter(entry -> entry.getKey().getNumber() == pageNumber)
|
||||
.map(Map.Entry::getValue)
|
||||
.anyMatch(rect -> rect.intersects(x, y, w, h));
|
||||
return getBBox().entrySet().stream().filter(entry -> entry.getKey().getNumber() == pageNumber).map(Map.Entry::getValue).anyMatch(rect -> rect.intersects(x, y, w, h));
|
||||
}
|
||||
|
||||
|
||||
@@ -661,8 +598,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default Stream<SemanticNode> streamAllSubNodes() {
|
||||
|
||||
return getDocumentTree().allSubEntriesInOrder(getTreeId())
|
||||
.map(DocumentTree.Entry::getNode);
|
||||
return getDocumentTree().allSubEntriesInOrder(getTreeId()).map(DocumentTree.Entry::getNode);
|
||||
}
|
||||
|
||||
|
||||
@@ -673,9 +609,7 @@ public interface SemanticNode {
|
||||
*/
|
||||
default Stream<SemanticNode> streamAllSubNodesOfType(NodeType nodeType) {
|
||||
|
||||
return getDocumentTree().allSubEntriesInOrder(getTreeId())
|
||||
.filter(entry -> entry.getType().equals(nodeType))
|
||||
.map(DocumentTree.Entry::getNode);
|
||||
return getDocumentTree().allSubEntriesInOrder(getTreeId()).filter(entry -> entry.getType().equals(nodeType)).map(DocumentTree.Entry::getNode);
|
||||
}
|
||||
|
||||
|
||||
@@ -714,8 +648,7 @@ public interface SemanticNode {
|
||||
if (isLeaf()) {
|
||||
return getTextBlock().getPositionsPerPage(textRange);
|
||||
}
|
||||
Optional<SemanticNode> containingChildNode = streamChildren().filter(child -> child.getTextRange().contains(textRange))
|
||||
.findFirst();
|
||||
Optional<SemanticNode> containingChildNode = streamChildren().filter(child -> child.getTextRange().contains(textRange)).findFirst();
|
||||
if (containingChildNode.isEmpty()) {
|
||||
return getTextBlock().getPositionsPerPage(textRange);
|
||||
}
|
||||
@@ -765,12 +698,8 @@ public interface SemanticNode {
|
||||
private Map<Page, Rectangle2D> getBBoxFromChildren() {
|
||||
|
||||
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
|
||||
List<Map<Page, Rectangle2D>> childrenBBoxes = streamChildren().map(SemanticNode::getBBox)
|
||||
.toList();
|
||||
Set<Page> pages = childrenBBoxes.stream()
|
||||
.flatMap(map -> map.keySet()
|
||||
.stream())
|
||||
.collect(Collectors.toSet());
|
||||
List<Map<Page, Rectangle2D>> childrenBBoxes = streamChildren().map(SemanticNode::getBBox).toList();
|
||||
Set<Page> pages = childrenBBoxes.stream().flatMap(map -> map.keySet().stream()).collect(Collectors.toSet());
|
||||
for (Page page : pages) {
|
||||
Rectangle2D bBoxOnPage = childrenBBoxes.stream()
|
||||
.filter(childBboxPerPage -> childBboxPerPage.containsKey(page))
|
||||
@@ -788,9 +717,7 @@ public interface SemanticNode {
|
||||
private Map<Page, Rectangle2D> getBBoxFromLeafTextBlock() {
|
||||
|
||||
Map<Page, Rectangle2D> bBoxPerPage = new HashMap<>();
|
||||
Map<Page, List<AtomicTextBlock>> atomicTextBlockPerPage = getTextBlock().getAtomicTextBlocks()
|
||||
.stream()
|
||||
.collect(Collectors.groupingBy(AtomicTextBlock::getPage));
|
||||
Map<Page, List<AtomicTextBlock>> atomicTextBlockPerPage = getTextBlock().getAtomicTextBlocks().stream().collect(Collectors.groupingBy(AtomicTextBlock::getPage));
|
||||
atomicTextBlockPerPage.forEach((page, atomicTextBlocks) -> bBoxPerPage.put(page, RectangleTransformations.atomicTextBlockBBox(atomicTextBlocks)));
|
||||
return bBoxPerPage;
|
||||
}
|
||||
|
||||
+3
-4
@@ -26,9 +26,6 @@ import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a table within a document.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -411,7 +408,9 @@ public class Table implements SemanticNode {
|
||||
public TextBlock getTextBlock() {
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = SemanticNode.super.getTextBlock();
|
||||
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf)
|
||||
.map(SemanticNode::getLeafTextBlock)
|
||||
.collect(new TextBlockCollector());
|
||||
}
|
||||
return textBlock;
|
||||
}
|
||||
|
||||
+1
-6
@@ -20,9 +20,6 @@ import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
/**
|
||||
* Represents a single table cell within a table.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -82,9 +79,7 @@ public class TableCell implements GenericSemanticNode {
|
||||
}
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf)
|
||||
.map(SemanticNode::getLeafTextBlock)
|
||||
.collect(new TextBlockCollector());
|
||||
textBlock = streamAllSubNodes().filter(SemanticNode::isLeaf).map(SemanticNode::getLeafTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
return textBlock;
|
||||
}
|
||||
|
||||
+15
-24
@@ -61,7 +61,6 @@ public class AtomicTextBlock implements TextBlock {
|
||||
return lineBreaks.size() + 1;
|
||||
}
|
||||
|
||||
|
||||
public static AtomicTextBlock empty(Long textBlockIdx, int stringOffset, Page page, int numberOnPage, SemanticNode parent) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
@@ -78,7 +77,10 @@ public class AtomicTextBlock implements TextBlock {
|
||||
}
|
||||
|
||||
|
||||
public static AtomicTextBlock fromAtomicTextBlockData(DocumentTextData atomicTextBlockData, DocumentPositionData atomicPositionBlockData, SemanticNode parent, Page page) {
|
||||
public static AtomicTextBlock fromAtomicTextBlockData(DocumentTextData atomicTextBlockData,
|
||||
DocumentPositionData atomicPositionBlockData,
|
||||
SemanticNode parent,
|
||||
Page page) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(atomicTextBlockData.getId())
|
||||
@@ -86,10 +88,8 @@ public class AtomicTextBlock implements TextBlock {
|
||||
.page(page)
|
||||
.textRange(new TextRange(atomicTextBlockData.getStart(), atomicTextBlockData.getEnd()))
|
||||
.searchText(atomicTextBlockData.getSearchText())
|
||||
.lineBreaks(Arrays.stream(atomicTextBlockData.getLineBreaks()).boxed()
|
||||
.toList())
|
||||
.stringIdxToPositionIdx(Arrays.stream(atomicPositionBlockData.getStringIdxToPositionIdx()).boxed()
|
||||
.toList())
|
||||
.lineBreaks(Arrays.stream(atomicTextBlockData.getLineBreaks()).boxed().toList())
|
||||
.stringIdxToPositionIdx(Arrays.stream(atomicPositionBlockData.getStringIdxToPositionIdx()).boxed().toList())
|
||||
.positions(toRectangle2DList(atomicPositionBlockData.getPositions()))
|
||||
.parent(parent)
|
||||
.build();
|
||||
@@ -98,9 +98,7 @@ public class AtomicTextBlock implements TextBlock {
|
||||
|
||||
private static List<Rectangle2D> toRectangle2DList(float[][] positions) {
|
||||
|
||||
return Arrays.stream(positions)
|
||||
.map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3]))
|
||||
.toList();
|
||||
return Arrays.stream(positions).map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3])).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -120,7 +118,6 @@ public class AtomicTextBlock implements TextBlock {
|
||||
return new TextRange(lineBreaks.get(lineNumber - 1) + textRange.start(), lineBreaks.get(lineNumber) + textRange.start());
|
||||
}
|
||||
|
||||
|
||||
public List<String> getWords() {
|
||||
|
||||
if (words == null) {
|
||||
@@ -147,9 +144,9 @@ public class AtomicTextBlock implements TextBlock {
|
||||
public int getNextLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak > fromIndex - textRange.start()) //
|
||||
.findFirst() //
|
||||
.orElse(searchText.length()) + textRange.start();
|
||||
.filter(linebreak -> linebreak > fromIndex - textRange.start()) //
|
||||
.findFirst() //
|
||||
.orElse(searchText.length()) + textRange.start();
|
||||
}
|
||||
|
||||
|
||||
@@ -157,9 +154,9 @@ public class AtomicTextBlock implements TextBlock {
|
||||
public int getPreviousLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak <= fromIndex - textRange.start())//
|
||||
.reduce((a, b) -> b)//
|
||||
.orElse(0) + textRange.start();
|
||||
.filter(linebreak -> linebreak <= fromIndex - textRange.start())//
|
||||
.reduce((a, b) -> b)//
|
||||
.orElse(0) + textRange.start();
|
||||
}
|
||||
|
||||
|
||||
@@ -212,10 +209,7 @@ public class AtomicTextBlock implements TextBlock {
|
||||
return "";
|
||||
}
|
||||
|
||||
Set<Integer> lbInBoundary = lineBreaks.stream()
|
||||
.map(i -> i + textRange.start())
|
||||
.filter(textRange::contains)
|
||||
.collect(Collectors.toSet());
|
||||
Set<Integer> lbInBoundary = lineBreaks.stream().map(i -> i + textRange.start()).filter(textRange::contains).collect(Collectors.toSet());
|
||||
if (textRange.end() == getTextRange().end()) {
|
||||
lbInBoundary.add(getTextRange().end());
|
||||
}
|
||||
@@ -241,10 +235,7 @@ public class AtomicTextBlock implements TextBlock {
|
||||
|
||||
private List<Integer> getAllLineBreaksInBoundary(TextRange textRange) {
|
||||
|
||||
return getLineBreaks().stream()
|
||||
.map(linebreak -> linebreak + this.textRange.start())
|
||||
.filter(textRange::contains)
|
||||
.toList();
|
||||
return getLineBreaks().stream().map(linebreak -> linebreak + this.textRange.start()).filter(textRange::contains).toList();
|
||||
}
|
||||
|
||||
|
||||
|
||||
+7
-22
@@ -44,8 +44,7 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
this.atomicTextBlocks.add(firstTextBlock);
|
||||
textRange = new TextRange(firstTextBlock.getTextRange().start(), firstTextBlock.getTextRange().end());
|
||||
|
||||
atomicTextBlocks.subList(1, atomicTextBlocks.size())
|
||||
.forEach(this::concat);
|
||||
atomicTextBlocks.subList(1, atomicTextBlocks.size()).forEach(this::concat);
|
||||
}
|
||||
|
||||
|
||||
@@ -66,10 +65,7 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
|
||||
private AtomicTextBlock getAtomicTextBlockByStringIndex(int stringIdx) {
|
||||
|
||||
return atomicTextBlocks.stream()
|
||||
.filter(textBlock -> textBlock.getTextRange().contains(stringIdx))
|
||||
.findAny()
|
||||
.orElseThrow(IndexOutOfBoundsException::new);
|
||||
return atomicTextBlocks.stream().filter(textBlock -> textBlock.getTextRange().contains(stringIdx)).findAny().orElseThrow(IndexOutOfBoundsException::new);
|
||||
}
|
||||
|
||||
|
||||
@@ -103,18 +99,14 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
@Override
|
||||
public List<String> getWords() {
|
||||
|
||||
return atomicTextBlocks.stream()
|
||||
.map(AtomicTextBlock::getWords)
|
||||
.flatMap(Collection::stream)
|
||||
.toList();
|
||||
return atomicTextBlocks.stream().map(AtomicTextBlock::getWords).flatMap(Collection::stream).toList();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int numberOfLines() {
|
||||
|
||||
return atomicTextBlocks.stream()
|
||||
.mapToInt(AtomicTextBlock::numberOfLines).sum();
|
||||
return atomicTextBlocks.stream().mapToInt(AtomicTextBlock::numberOfLines).sum();
|
||||
}
|
||||
|
||||
|
||||
@@ -135,10 +127,7 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
@Override
|
||||
public List<Integer> getLineBreaks() {
|
||||
|
||||
return getAtomicTextBlocks().stream()
|
||||
.flatMap(atomicTextBlock -> atomicTextBlock.getLineBreaks()
|
||||
.stream())
|
||||
.toList();
|
||||
return getAtomicTextBlocks().stream().flatMap(atomicTextBlock -> atomicTextBlock.getLineBreaks().stream()).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -213,8 +202,7 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
|
||||
AtomicTextBlock lastTextBlock = textBlocks.get(textBlocks.size() - 1);
|
||||
rectanglesPerLinePerPage = mergeEntityPositionsWithSamePageNode(rectanglesPerLinePerPage,
|
||||
lastTextBlock.getPositionsPerPage(new TextRange(lastTextBlock.getTextRange().start(),
|
||||
stringTextRange.end())));
|
||||
lastTextBlock.getPositionsPerPage(new TextRange(lastTextBlock.getTextRange().start(), stringTextRange.end())));
|
||||
|
||||
return rectanglesPerLinePerPage;
|
||||
}
|
||||
@@ -251,10 +239,7 @@ public class ConcatenatedTextBlock implements TextBlock {
|
||||
private Map<Page, List<Rectangle2D>> mergeEntityPositionsWithSamePageNode(Map<Page, List<Rectangle2D>> map1, Map<Page, List<Rectangle2D>> map2) {
|
||||
|
||||
Map<Page, List<Rectangle2D>> mergedMap = new HashMap<>(map1);
|
||||
map2.forEach((pageNode, rectangles) -> mergedMap.merge(pageNode,
|
||||
rectangles,
|
||||
(l1, l2) -> Stream.concat(l1.stream(), l2.stream())
|
||||
.toList()));
|
||||
map2.forEach((pageNode, rectangles) -> mergedMap.merge(pageNode, rectangles, (l1, l2) -> Stream.concat(l1.stream(), l2.stream()).toList()));
|
||||
return mergedMap;
|
||||
}
|
||||
|
||||
|
||||
+2
-6
@@ -18,10 +18,8 @@ public interface TextBlock extends CharSequence {
|
||||
|
||||
String getSearchText();
|
||||
|
||||
|
||||
List<String> getWords();
|
||||
|
||||
|
||||
List<AtomicTextBlock> getAtomicTextBlocks();
|
||||
|
||||
|
||||
@@ -37,6 +35,7 @@ public interface TextBlock extends CharSequence {
|
||||
TextRange getLineTextRange(int lineNumber);
|
||||
|
||||
|
||||
|
||||
List<Integer> getLineBreaks();
|
||||
|
||||
|
||||
@@ -72,7 +71,6 @@ public interface TextBlock extends CharSequence {
|
||||
return RectangleTransformations.rectangle2DBBox(getLinePositions(lineNumber));
|
||||
}
|
||||
|
||||
|
||||
default String searchTextWithLineBreaks() {
|
||||
|
||||
return subSequenceWithLineBreaks(getTextRange());
|
||||
@@ -87,9 +85,7 @@ public interface TextBlock extends CharSequence {
|
||||
|
||||
default Set<Page> getPages() {
|
||||
|
||||
return getAtomicTextBlocks().stream()
|
||||
.map(AtomicTextBlock::getPage)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
return getAtomicTextBlocks().stream().map(AtomicTextBlock::getPage).collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
|
||||
|
||||
+1
-2
@@ -9,8 +9,7 @@ public record RuleClass(RuleType ruleType, List<RuleUnit> ruleUnits) {
|
||||
public Optional<RuleUnit> findRuleUnitByInteger(Integer unit) {
|
||||
|
||||
return ruleUnits.stream()
|
||||
.filter(ruleUnit -> Objects.equals(ruleUnit.unit(), unit))
|
||||
.findFirst();
|
||||
.filter(ruleUnit -> Objects.equals(ruleUnit.unit(), unit)).findFirst();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+8
-21
@@ -10,7 +10,7 @@ import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
@@ -32,22 +32,18 @@ public final class RuleFileBluePrint {
|
||||
int globalsLine;
|
||||
List<BasicQuery> queries;
|
||||
List<RuleClass> ruleClasses;
|
||||
DroolsValidation droolsValidation;
|
||||
DroolsSyntaxValidation droolsSyntaxValidation;
|
||||
|
||||
|
||||
public Optional<RuleClass> findRuleClassByType(RuleType ruleType) {
|
||||
|
||||
return ruleClasses.stream()
|
||||
.filter(ruleClass -> Objects.equals(ruleClass.ruleType(), ruleType))
|
||||
.findFirst();
|
||||
return ruleClasses.stream().filter(ruleClass -> Objects.equals(ruleClass.ruleType(), ruleType)).findFirst();
|
||||
}
|
||||
|
||||
|
||||
public Set<String> getImportSplitByKeyword() {
|
||||
|
||||
return Arrays.stream(imports.replaceAll("\n", "").split("import"))
|
||||
.map(String::trim)
|
||||
.collect(Collectors.toSet());
|
||||
return Arrays.stream(imports.replaceAll("\n", "").split("import")).map(String::trim).collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
|
||||
@@ -57,15 +53,11 @@ public final class RuleFileBluePrint {
|
||||
return findRuleClassByType(ruleIdentifier.type()).map(RuleClass::ruleUnits)
|
||||
.orElse(Collections.emptyList())
|
||||
.stream()
|
||||
.flatMap(ruleUnit -> ruleUnit.rules()
|
||||
.stream()
|
||||
.filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
|
||||
.flatMap(ruleUnit -> ruleUnit.rules().stream().filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
|
||||
.toList();
|
||||
}
|
||||
return findRuleClassByType(ruleIdentifier.type()).flatMap(ruleClass -> ruleClass.findRuleUnitByInteger(ruleIdentifier.unit()))
|
||||
.map(ruleUnit -> ruleUnit.rules()
|
||||
.stream()
|
||||
.filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
|
||||
.map(ruleUnit -> ruleUnit.rules().stream().filter(rule -> rule.getIdentifier().matches(ruleIdentifier)))
|
||||
.orElse(Stream.empty())
|
||||
.toList();
|
||||
}
|
||||
@@ -73,18 +65,13 @@ public final class RuleFileBluePrint {
|
||||
|
||||
public List<RuleIdentifier> getAllRuleIdentifiers() {
|
||||
|
||||
return streamAllRules().map(BasicRule::getIdentifier)
|
||||
.collect(Collectors.toList());
|
||||
return streamAllRules().map(BasicRule::getIdentifier).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
public Stream<BasicRule> streamAllRules() {
|
||||
|
||||
return getRuleClasses().stream()
|
||||
.map(RuleClass::ruleUnits)
|
||||
.flatMap(Collection::stream)
|
||||
.map(RuleUnit::rules)
|
||||
.flatMap(Collection::stream);
|
||||
return getRuleClasses().stream().map(RuleClass::ruleUnits).flatMap(Collection::stream).map(RuleUnit::rules).flatMap(Collection::stream);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+2
-2
@@ -42,8 +42,8 @@ public record RuleIdentifier(@NonNull RuleType type, Integer unit, Integer id) {
|
||||
public boolean matches(RuleIdentifier ruleIdentifier) {
|
||||
|
||||
return ruleIdentifier.type().equals(this.type()) && //
|
||||
(Objects.isNull(ruleIdentifier.unit()) || Objects.isNull(this.unit()) || Objects.equals(this.unit(), ruleIdentifier.unit())) && //
|
||||
(Objects.isNull(ruleIdentifier.id()) || Objects.isNull(this.id()) || Objects.equals(this.id(), ruleIdentifier.id()));
|
||||
(Objects.isNull(ruleIdentifier.unit()) || Objects.isNull(this.unit()) || Objects.equals(this.unit(), ruleIdentifier.unit())) && //
|
||||
(Objects.isNull(ruleIdentifier.id()) || Objects.isNull(this.id()) || Objects.equals(this.id(), ruleIdentifier.id()));
|
||||
|
||||
}
|
||||
|
||||
|
||||
-1
@@ -70,7 +70,6 @@ public class MessagingConfiguration {
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
@Bean
|
||||
public Queue redactionAnalysisResponseQueue() {
|
||||
|
||||
|
||||
+20
-18
@@ -51,14 +51,14 @@ public class RedactionMessageReceiver {
|
||||
// This prevents from endless retries oom errors.
|
||||
if (message.getMessageProperties().isRedelivered()) {
|
||||
var errorMessage = format("Error during last processing of request with dossierId: %s and fileId: %s, do not retry.",
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId());
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId());
|
||||
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(errorMessage,
|
||||
priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE,
|
||||
"redaction-service",
|
||||
OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS)));
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(errorMessage,
|
||||
priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE,
|
||||
"redaction-service",
|
||||
OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS)));
|
||||
throw new AmqpRejectAndDontRequeueException(errorMessage);
|
||||
}
|
||||
|
||||
@@ -84,9 +84,9 @@ public class RedactionMessageReceiver {
|
||||
log.debug(analyzeRequest.getManualRedactions().toString());
|
||||
result = analyzeService.analyze(analyzeRequest);
|
||||
log.info("Successfully analyzed dossier {} file {} took: {} s",
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
format("%.2f", result.getDuration() / 1000.0));
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
format("%.2f", result.getDuration() / 1000.0));
|
||||
log.info("----------------------------------------------------------------------------------");
|
||||
break;
|
||||
|
||||
@@ -96,9 +96,9 @@ public class RedactionMessageReceiver {
|
||||
log.debug(analyzeRequest.getManualRedactions().toString());
|
||||
result = analyzeService.reanalyze(analyzeRequest);
|
||||
log.info("Successfully reanalyzed dossier {} file {} took: {} s",
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
format("%.2f", result.getDuration() / 1000.0));
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
format("%.2f", result.getDuration() / 1000.0));
|
||||
log.info("----------------------------------------------------------------------------------");
|
||||
break;
|
||||
case SURROUNDING_TEXT_ANALYSIS:
|
||||
@@ -106,7 +106,9 @@ public class RedactionMessageReceiver {
|
||||
log.info("Starting Surrounding Text Analysis for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
log.debug(analyzeRequest.getManualRedactions().toString());
|
||||
unprocessedChangesService.analyseSurroundingText(analyzeRequest);
|
||||
log.info("Successful Surrounding Text Analysis dossier {} file {} ", analyzeRequest.getDossierId(), analyzeRequest.getFileId());
|
||||
log.info("Successful Surrounding Text Analysis dossier {} file {} ",
|
||||
analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId());
|
||||
log.info("-------------------------------------------------------------------------------------------------");
|
||||
shouldRespond = false;
|
||||
break;
|
||||
@@ -135,8 +137,8 @@ public class RedactionMessageReceiver {
|
||||
log.warn("Failed to process analyze request: {}", analyzeRequest, e);
|
||||
var timestamp = OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS);
|
||||
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(e.getMessage(), priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE, "redaction-service", timestamp));
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(e.getMessage(), priority ? REDACTION_PRIORITY_QUEUE : REDACTION_QUEUE, "redaction-service", timestamp));
|
||||
}
|
||||
|
||||
|
||||
@@ -151,8 +153,8 @@ public class RedactionMessageReceiver {
|
||||
timestamp = timestamp != null ? timestamp : OffsetDateTime.now().truncatedTo(ChronoUnit.MILLIS);
|
||||
log.info("Failed to process analyze request, errorCause: {}, timestamp: {}", errorCause, timestamp);
|
||||
fileStatusProcessingUpdateClient.analysisFailed(analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(errorCause, REDACTION_DQL, "redaction-service", timestamp));
|
||||
analyzeRequest.getFileId(),
|
||||
new FileErrorInfo(errorCause, REDACTION_DQL, "redaction-service", timestamp));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+28
-54
@@ -1,8 +1,5 @@
|
||||
package com.iqser.red.service.redaction.v1.server.service;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.service.document.SectionFinderService.getRelevantManuallyModifiedAnnotationIds;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
@@ -25,9 +22,16 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogChanges;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.legalbasis.LegalBasis;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogChanges;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLogLegalBasis;
|
||||
import com.iqser.red.service.redaction.v1.server.RedactionServiceSettings;
|
||||
import com.iqser.red.service.redaction.v1.server.client.LegalBasisClient;
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.NerEntitiesModel;
|
||||
import com.iqser.red.service.redaction.v1.server.model.KieWrapper;
|
||||
import com.iqser.red.service.redaction.v1.server.model.PrecursorEntity;
|
||||
import com.iqser.red.service.redaction.v1.server.model.NerEntities;
|
||||
import com.iqser.red.service.redaction.v1.server.model.component.Component;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
|
||||
@@ -83,7 +87,7 @@ public class AnalyzeService {
|
||||
public AnalyzeResult reanalyze(@RequestBody AnalyzeRequest analyzeRequest) {
|
||||
|
||||
long startTime = System.currentTimeMillis();
|
||||
EntityLog entityLogWithoutEntries = redactionStorageService.getEntityLogWithoutEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
|
||||
EntityLog previousEntityLog = redactionStorageService.getEntityLog(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
|
||||
log.info("Loaded previous entity log for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
|
||||
Document document = DocumentGraphMapper.toDocumentGraph(observedStorageService.getDocumentData(analyzeRequest.getDossierId(), analyzeRequest.getFileId()));
|
||||
@@ -93,36 +97,25 @@ public class AnalyzeService {
|
||||
log.info("Loaded Imported Redactions for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
|
||||
// not yet ready for reanalysis
|
||||
if (entityLogWithoutEntries == null || document == null || document.getNumberOfPages() == 0) {
|
||||
if (previousEntityLog == null || document == null || document.getNumberOfPages() == 0) {
|
||||
return analyze(analyzeRequest);
|
||||
}
|
||||
|
||||
DictionaryIncrement dictionaryIncrement = dictionaryService.getDictionaryIncrements(analyzeRequest.getDossierTemplateId(),
|
||||
new DictionaryVersion(entityLogWithoutEntries.getDictionaryVersion(),
|
||||
entityLogWithoutEntries.getDossierDictionaryVersion()),
|
||||
new DictionaryVersion(previousEntityLog.getDictionaryVersion(),
|
||||
previousEntityLog.getDossierDictionaryVersion()),
|
||||
analyzeRequest.getDossierId());
|
||||
|
||||
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
|
||||
|
||||
Set<Integer> sectionsToReanalyseIds = redactionStorageService.findIdsOfSectionsToReanalyse(analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
relevantManuallyModifiedAnnotationIds);
|
||||
sectionsToReanalyseIds.addAll(getSectionsToReanalyseIds(analyzeRequest,
|
||||
document,
|
||||
dictionaryIncrement,
|
||||
importedRedactions,
|
||||
relevantManuallyModifiedAnnotationIds));
|
||||
|
||||
Set<Integer> sectionsToReanalyseIds = getSectionsToReanalyseIds(analyzeRequest, previousEntityLog, document, dictionaryIncrement, importedRedactions);
|
||||
List<SemanticNode> sectionsToReAnalyse = getSectionsToReAnalyse(document, sectionsToReanalyseIds);
|
||||
log.info("{} Sections to reanalyze found for file {} in dossier {}", sectionsToReanalyseIds.size(), analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
|
||||
if (sectionsToReAnalyse.isEmpty()) {
|
||||
|
||||
EntityLogChanges entityLogChanges = entityLogCreatorService.updateVersionsAndReturnChanges(entityLogWithoutEntries,
|
||||
EntityLogChanges entityLogChanges = entityLogCreatorService.updateVersionsAndReturnChanges(previousEntityLog,
|
||||
dictionaryIncrement.getDictionaryVersion(),
|
||||
analyzeRequest,
|
||||
new ArrayList<>(),
|
||||
new ArrayList<>());
|
||||
false);
|
||||
|
||||
return finalizeAnalysis(analyzeRequest,
|
||||
startTime,
|
||||
@@ -167,8 +160,8 @@ public class AnalyzeService {
|
||||
|
||||
EntityLogChanges entityLogChanges = entityLogCreatorService.updatePreviousEntityLog(analyzeRequest,
|
||||
document,
|
||||
entityLogWithoutEntries,
|
||||
notFoundManualOrImportedEntries,
|
||||
previousEntityLog,
|
||||
sectionsToReanalyseIds,
|
||||
dictionary.getVersion());
|
||||
|
||||
@@ -231,18 +224,18 @@ public class AnalyzeService {
|
||||
nerEntities);
|
||||
log.info("Finished entity rule execution for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
|
||||
EntityLogChanges entityLogChanges = entityLogCreatorService.createInitialEntityLog(analyzeRequest,
|
||||
document,
|
||||
notFoundManualOrImportedEntries,
|
||||
dictionary.getVersion(),
|
||||
kieWrapperEntityRules.rulesVersion());
|
||||
EntityLog entityLog = entityLogCreatorService.createInitialEntityLog(analyzeRequest,
|
||||
document,
|
||||
notFoundManualOrImportedEntries,
|
||||
dictionary.getVersion(),
|
||||
kieWrapperEntityRules.rulesVersion());
|
||||
|
||||
notFoundImportedEntitiesService.processEntityLog(entityLogChanges.getEntityLog(), analyzeRequest, notFoundImportedEntries);
|
||||
notFoundImportedEntitiesService.processEntityLog(entityLog, analyzeRequest, notFoundImportedEntries);
|
||||
|
||||
return finalizeAnalysis(analyzeRequest,
|
||||
startTime,
|
||||
kieWrapperComponentRules,
|
||||
entityLogChanges,
|
||||
new EntityLogChanges(entityLog, false),
|
||||
document,
|
||||
document.getNumberOfPages(),
|
||||
dictionary.getVersion(),
|
||||
@@ -262,25 +255,10 @@ public class AnalyzeService {
|
||||
Set<FileAttribute> addedFileAttributes) {
|
||||
|
||||
EntityLog entityLog = entityLogChanges.getEntityLog();
|
||||
|
||||
// as workaround for duplicate key exceptions occurring due to simultaneous analyses and reanalyses save instead of insert is used
|
||||
// also analysis numbers should be incremented in every follow-up request, so checking if the log exists is not needed
|
||||
if (!redactionStorageService.entityLogExists(analyzeRequest.getDossierId(), analyzeRequest.getFileId())) {
|
||||
redactionStorageService.saveEntityLog(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLog);
|
||||
|
||||
} else {
|
||||
redactionStorageService.updateEntityLogWithoutEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLog);
|
||||
|
||||
if (!entityLogChanges.getNewEntityLogEntries().isEmpty()) {
|
||||
redactionStorageService.saveEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLogChanges.getNewEntityLogEntries());
|
||||
}
|
||||
if (!entityLogChanges.getUpdatedEntityLogEntries().isEmpty()) {
|
||||
redactionStorageService.updateEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), entityLogChanges.getUpdatedEntityLogEntries());
|
||||
}
|
||||
}
|
||||
redactionStorageService.storeObject(analyzeRequest.getDossierId(), analyzeRequest.getFileId(), FileType.ENTITY_LOG, entityLogChanges.getEntityLog());
|
||||
|
||||
log.info("Created entity log for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
if (entityLogChanges.hasChanges() || !isReanalysis) {
|
||||
if (entityLogChanges.isHasChanges() || !isReanalysis) {
|
||||
computeComponentsWhenRulesArePresent(analyzeRequest, kieWrapperComponentRules, document, addedFileAttributes, entityLogChanges, dictionaryVersion);
|
||||
}
|
||||
|
||||
@@ -295,7 +273,7 @@ public class AnalyzeService {
|
||||
.fileId(analyzeRequest.getFileId())
|
||||
.duration(duration)
|
||||
.numberOfPages(numberOfPages)
|
||||
.hasUpdates(entityLogChanges.hasChanges())
|
||||
.hasUpdates(entityLogChanges.isHasChanges())
|
||||
.analysisVersion(redactionServiceSettings.getAnalysisVersion())
|
||||
.analysisNumber(analyzeRequest.getAnalysisNumber())
|
||||
.rulesVersion(entityLog.getRulesVersion())
|
||||
@@ -345,16 +323,12 @@ public class AnalyzeService {
|
||||
|
||||
|
||||
private Set<Integer> getSectionsToReanalyseIds(AnalyzeRequest analyzeRequest,
|
||||
EntityLog entityLog,
|
||||
Document document,
|
||||
DictionaryIncrement dictionaryIncrement,
|
||||
ImportedRedactions importedRedactions,
|
||||
Set<String> relevantManuallyModifiedAnnotationIds) {
|
||||
ImportedRedactions importedRedactions) {
|
||||
|
||||
return sectionFinderService.findSectionsToReanalyse(dictionaryIncrement,
|
||||
document,
|
||||
analyzeRequest,
|
||||
importedRedactions,
|
||||
relevantManuallyModifiedAnnotationIds);
|
||||
return sectionFinderService.findSectionsToReanalyse(dictionaryIncrement, entityLog, document, analyzeRequest, importedRedactions);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+12
-26
@@ -23,15 +23,13 @@ public class ComponentLogCreatorService {
|
||||
public ComponentLog buildComponentLog(int analysisNumber, List<Component> components, long componentRulesVersion) {
|
||||
|
||||
Map<String, List<ComponentLogEntryValue>> map = new HashMap<>();
|
||||
components.stream()
|
||||
.sorted(ComponentComparator.first())
|
||||
.forEach(component -> {
|
||||
ComponentLogEntryValue componentLogEntryValue = buildComponentLogEntry(component);
|
||||
map.computeIfAbsent(component.getName(), k -> new ArrayList<>()).add(componentLogEntryValue);
|
||||
});
|
||||
List<ComponentLogEntry> componentLogComponents = map.entrySet()
|
||||
.stream()
|
||||
.map(entry -> new ComponentLogEntry(entry.getKey(), entry.getValue()))
|
||||
components.stream().sorted(ComponentComparator.first()).forEach(component -> {
|
||||
ComponentLogEntryValue componentLogEntryValue = buildComponentLogEntry(component);
|
||||
map.computeIfAbsent(component.getName(), k -> new ArrayList<>()).add(componentLogEntryValue);
|
||||
});
|
||||
List<ComponentLogEntry> componentLogComponents = map
|
||||
.entrySet()
|
||||
.stream().map(entry -> new ComponentLogEntry(entry.getKey(), entry.getValue()))
|
||||
.toList();
|
||||
return new ComponentLog(analysisNumber, componentRulesVersion, componentLogComponents);
|
||||
}
|
||||
@@ -40,36 +38,24 @@ public class ComponentLogCreatorService {
|
||||
private ComponentLogEntryValue buildComponentLogEntry(Component component) {
|
||||
|
||||
return ComponentLogEntryValue.builder()
|
||||
.value(component.getValue())
|
||||
.originalValue(component.getValue())
|
||||
.value(component.getValue()).originalValue(component.getValue())
|
||||
.componentRuleId(component.getMatchedRule().toString())
|
||||
.valueDescription(component.getValueDescription())
|
||||
.componentLogEntityReferences(toComponentEntityReferences(component.getReferences()
|
||||
.stream()
|
||||
.sorted(EntityComparators.first())
|
||||
.toList()))
|
||||
.componentLogEntityReferences(toComponentEntityReferences(component.getReferences().stream().sorted(EntityComparators.first()).toList()))
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private List<ComponentLogEntityReference> toComponentEntityReferences(List<Entity> references) {
|
||||
|
||||
return references.stream()
|
||||
.map(this::toComponentEntityReference)
|
||||
.toList();
|
||||
return references.stream().map(this::toComponentEntityReference).toList();
|
||||
}
|
||||
|
||||
|
||||
private ComponentLogEntityReference toComponentEntityReference(Entity entity) {
|
||||
|
||||
return ComponentLogEntityReference.builder()
|
||||
.id(entity.getId())
|
||||
.page(entity.getPositions()
|
||||
.stream()
|
||||
.findFirst()
|
||||
.map(Position::getPageNumber)
|
||||
.orElse(0))
|
||||
.entityRuleId(entity.getMatchedRule())
|
||||
return ComponentLogEntityReference.builder().id(entity.getId())
|
||||
.page(entity.getPositions().stream().findFirst().map(Position::getPageNumber).orElse(0)).entityRuleId(entity.getMatchedRule())
|
||||
.type(entity.getType())
|
||||
.build();
|
||||
}
|
||||
|
||||
+7
-13
@@ -6,11 +6,10 @@ import java.util.Set;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Engine;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.DictionaryModel;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.entity.EntityType;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNode;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.Dictionary;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.EntityCreationService;
|
||||
import com.iqser.red.service.redaction.v1.server.service.document.EntityEnrichmentService;
|
||||
|
||||
@@ -39,13 +38,10 @@ public class DictionarySearchService {
|
||||
@Observed(name = "DictionarySearchService", contextualName = "add-dictionary-entries")
|
||||
public void addDictionaryEntities(Dictionary dictionary, SemanticNode node) {
|
||||
|
||||
for (DictionaryModel model : dictionary.getDictionaryModels()) {
|
||||
for (var model : dictionary.getDictionaryModels()) {
|
||||
bySearchImplementationAsDictionary(model.getEntriesSearch(), model.getType(), model.isHint() ? EntityType.HINT : EntityType.ENTITY, node, model.isDossierDictionary());
|
||||
bySearchImplementationAsDictionary(model.getFalsePositiveSearch(), model.getType(), EntityType.FALSE_POSITIVE, node, model.isDossierDictionary());
|
||||
bySearchImplementationAsDictionary(model.getFalseRecommendationsSearch(), model.getType(), EntityType.FALSE_RECOMMENDATION, node, model.isDossierDictionary());
|
||||
if (model.isDossierDictionary()) {
|
||||
bySearchImplementationAsDictionary(model.getDeletionEntriesSearch(), model.getType(), EntityType.DICTIONARY_REMOVAL, node, model.isDossierDictionary());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -56,16 +52,14 @@ public class DictionarySearchService {
|
||||
SemanticNode node,
|
||||
boolean isDossierDictionaryEntry) {
|
||||
|
||||
Set<Engine> engines = isDossierDictionaryEntry ? Set.of(Engine.DOSSIER_DICTIONARY) : Set.of(Engine.DICTIONARY);
|
||||
EntityCreationService entityCreationService = new EntityCreationService(entityEnrichmentService);
|
||||
searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
|
||||
.stream()
|
||||
.filter(boundary -> entityCreationService.isValidEntityTextRange(node.getTextBlock(), boundary))
|
||||
.forEach(bounds -> entityCreationService.byTextRangeWithEngine(bounds, type, entityType, node, engines)
|
||||
.ifPresent(entity -> {
|
||||
entity.setDictionaryEntry(true);
|
||||
entity.setDossierDictionaryEntry(isDossierDictionaryEntry);
|
||||
}));
|
||||
.forEach(bounds -> entityCreationService.byTextRangeWithEngine(bounds, type, entityType, node, Set.of(Engine.DICTIONARY)).ifPresent(entity -> {
|
||||
entity.setDictionaryEntry(true);
|
||||
entity.setDossierDictionaryEntry(isDossierDictionaryEntry);
|
||||
}));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+135
-190
@@ -4,7 +4,6 @@ import java.awt.Color;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Optional;
|
||||
@@ -92,8 +91,7 @@ public class DictionaryService {
|
||||
updateDictionaryEntry(dossierTemplateId, dossierDictionaryVersion, getVersion(dossierDictionary), dossierId);
|
||||
}
|
||||
|
||||
return DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateDictionaryVersion).dossierVersion(dossierDictionaryVersion)
|
||||
.build();
|
||||
return DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateDictionaryVersion).dossierVersion(dossierDictionaryVersion).build();
|
||||
}
|
||||
|
||||
|
||||
@@ -108,47 +106,41 @@ public class DictionaryService {
|
||||
List<DictionaryModel> dictionaryModels = getDossierTemplateDictionary(dossierTemplateId).getDictionary();
|
||||
|
||||
dictionaryModels.forEach(dictionaryModel -> {
|
||||
dictionaryModel.getEntries()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalsePositives()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalseRecommendations()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getEntries().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierTemplateVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
if (dossierDictionaryExists(dossierId)) {
|
||||
dictionaryModels = getDossierDictionary(dossierId).getDictionary();
|
||||
dictionaryModels.forEach(dictionaryModel -> {
|
||||
dictionaryModel.getEntries()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalsePositives()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalseRecommendations()
|
||||
.forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getEntries().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalsePositives().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
dictionaryModel.getFalseRecommendations().forEach(dictionaryEntry -> {
|
||||
if (dictionaryEntry.getVersion() > fromVersion.getDossierVersion()) {
|
||||
newValues.add(new DictionaryIncrementValue(dictionaryEntry.getValue(), dictionaryModel.isCaseInsensitive()));
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -163,120 +155,84 @@ public class DictionaryService {
|
||||
DictionaryRepresentation dictionaryRepresentation = new DictionaryRepresentation();
|
||||
|
||||
var typeResponse = dossierId == null ? dictionaryClient.getAllTypesForDossierTemplate(dossierTemplateId, true) : dictionaryClient.getAllTypesForDossier(dossierId,
|
||||
true);
|
||||
true);
|
||||
if (CollectionUtils.isNotEmpty(typeResponse)) {
|
||||
|
||||
List<DictionaryModel> dictionary = typeResponse.stream()
|
||||
.map(t -> {
|
||||
List<DictionaryModel> dictionary = typeResponse.stream().map(t -> {
|
||||
|
||||
Optional<DictionaryModel> optionalOldModel;
|
||||
if (dossierId == null) {
|
||||
var representation = getDossierTemplateDictionary(dossierTemplateId);
|
||||
optionalOldModel = representation != null ? representation.getDictionary()
|
||||
.stream()
|
||||
.filter(f -> f.getType().equals(t.getType()))
|
||||
.findAny() : Optional.empty();
|
||||
} else {
|
||||
var representation = getDossierDictionary(dossierId);
|
||||
optionalOldModel = representation != null ? representation.getDictionary()
|
||||
.stream()
|
||||
.filter(f -> f.getType().equals(t.getType()))
|
||||
.findAny() : Optional.empty();
|
||||
}
|
||||
Optional<DictionaryModel> optionalOldModel;
|
||||
if (dossierId == null) {
|
||||
var representation = getDossierTemplateDictionary(dossierTemplateId);
|
||||
optionalOldModel = representation != null ? representation.getDictionary()
|
||||
.stream()
|
||||
.filter(f -> f.getType().equals(t.getType()))
|
||||
.findAny() : Optional.empty();
|
||||
} else {
|
||||
var representation = getDossierDictionary(dossierId);
|
||||
optionalOldModel = representation != null ? representation.getDictionary()
|
||||
.stream()
|
||||
.filter(f -> f.getType().equals(t.getType()))
|
||||
.findAny() : Optional.empty();
|
||||
}
|
||||
|
||||
Set<DictionaryEntryModel> entries = new HashSet<>();
|
||||
Set<DictionaryEntryModel> falsePositives = new HashSet<>();
|
||||
Set<DictionaryEntryModel> falseRecommendations = new HashSet<>();
|
||||
Set<DictionaryEntryModel> entries = new HashSet<>();
|
||||
Set<DictionaryEntryModel> falsePositives = new HashSet<>();
|
||||
Set<DictionaryEntryModel> falseRecommendations = new HashSet<>();
|
||||
|
||||
DictionaryEntries newEntries = getEntries(t.getId(), currentVersion);
|
||||
DictionaryEntries newEntries = getEntries(t.getId(), currentVersion);
|
||||
|
||||
var newValues = newEntries.getEntries()
|
||||
var newValues = newEntries.getEntries().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
|
||||
var newFalsePositivesValues = newEntries.getFalsePositives().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
|
||||
var newFalseRecommendationsValues = newEntries.getFalseRecommendations().stream().map(DictionaryEntry::getValue).collect(Collectors.toSet());
|
||||
|
||||
optionalOldModel.ifPresent(oldDictionaryModel -> {
|
||||
|
||||
});
|
||||
if (optionalOldModel.isPresent()) {
|
||||
var oldModel = optionalOldModel.get();
|
||||
if (oldModel.isCaseInsensitive() && !t.isCaseInsensitive()) {
|
||||
// add old entries from existing DictionaryModel but exclude lower case representation
|
||||
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue())).toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives()
|
||||
.stream()
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toSet());
|
||||
var newFalsePositivesValues = newEntries.getFalsePositives()
|
||||
.filter(f -> !newFalsePositivesValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue()))
|
||||
.toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations()
|
||||
.stream()
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toSet());
|
||||
var newFalseRecommendationsValues = newEntries.getFalseRecommendations()
|
||||
.filter(f -> !newFalseRecommendationsValues.stream().map(s -> s.toLowerCase(Locale.ROOT)).toList().contains(f.getValue()))
|
||||
.toList());
|
||||
} else if (!oldModel.isCaseInsensitive() && t.isCaseInsensitive()) {
|
||||
// add old entries from existing DictionaryModel but exclude upper case representation
|
||||
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.contains(f.getValue().toLowerCase(Locale.ROOT))).toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives().stream().filter(f -> !newFalsePositivesValues.contains(f.getValue().toLowerCase(Locale.ROOT))).toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations()
|
||||
.stream()
|
||||
.map(DictionaryEntry::getValue)
|
||||
.collect(Collectors.toSet());
|
||||
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
|
||||
.toList());
|
||||
|
||||
optionalOldModel.ifPresent(oldDictionaryModel -> {
|
||||
} else {
|
||||
// add old entries from existing DictionaryModel
|
||||
entries.addAll(oldModel.getEntries().stream().filter(f -> !newValues.contains(f.getValue())).toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives().stream().filter(f -> !newFalsePositivesValues.contains(f.getValue())).toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations().stream().filter(f -> !newFalseRecommendationsValues.contains(f.getValue())).toList());
|
||||
}
|
||||
}
|
||||
|
||||
});
|
||||
if (optionalOldModel.isPresent()) {
|
||||
var oldModel = optionalOldModel.get();
|
||||
if (oldModel.isCaseInsensitive() && !t.isCaseInsensitive()) {
|
||||
// add old entries from existing DictionaryModel but exclude lower case representation
|
||||
entries.addAll(oldModel.getEntries()
|
||||
.stream()
|
||||
.filter(f -> !newValues.stream()
|
||||
.map(s -> s.toLowerCase(Locale.ROOT))
|
||||
.toList().contains(f.getValue()))
|
||||
.toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives()
|
||||
.stream()
|
||||
.filter(f -> !newFalsePositivesValues.stream()
|
||||
.map(s -> s.toLowerCase(Locale.ROOT))
|
||||
.toList().contains(f.getValue()))
|
||||
.toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations()
|
||||
.stream()
|
||||
.filter(f -> !newFalseRecommendationsValues.stream()
|
||||
.map(s -> s.toLowerCase(Locale.ROOT))
|
||||
.toList().contains(f.getValue()))
|
||||
.toList());
|
||||
} else if (!oldModel.isCaseInsensitive() && t.isCaseInsensitive()) {
|
||||
// add old entries from existing DictionaryModel but exclude upper case representation
|
||||
entries.addAll(oldModel.getEntries()
|
||||
.stream()
|
||||
.filter(f -> !newValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
|
||||
.toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives()
|
||||
.stream()
|
||||
.filter(f -> !newFalsePositivesValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
|
||||
.toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations()
|
||||
.stream()
|
||||
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue().toLowerCase(Locale.ROOT)))
|
||||
.toList());
|
||||
// Add Increments
|
||||
entries.addAll(newEntries.getEntries());
|
||||
falsePositives.addAll(newEntries.getFalsePositives());
|
||||
falseRecommendations.addAll(newEntries.getFalseRecommendations());
|
||||
|
||||
} else {
|
||||
// add old entries from existing DictionaryModel
|
||||
entries.addAll(oldModel.getEntries()
|
||||
.stream()
|
||||
.filter(f -> !newValues.contains(f.getValue()))
|
||||
.toList());
|
||||
falsePositives.addAll(oldModel.getFalsePositives()
|
||||
.stream()
|
||||
.filter(f -> !newFalsePositivesValues.contains(f.getValue()))
|
||||
.toList());
|
||||
falseRecommendations.addAll(oldModel.getFalseRecommendations()
|
||||
.stream()
|
||||
.filter(f -> !newFalseRecommendationsValues.contains(f.getValue()))
|
||||
.toList());
|
||||
}
|
||||
}
|
||||
|
||||
// Add Increments
|
||||
entries.addAll(newEntries.getEntries());
|
||||
falsePositives.addAll(newEntries.getFalsePositives());
|
||||
falseRecommendations.addAll(newEntries.getFalseRecommendations());
|
||||
|
||||
return new DictionaryModel(t.getType(),
|
||||
t.getRank(),
|
||||
convertColor(t.getHexColor()),
|
||||
t.isCaseInsensitive(),
|
||||
t.isHint(),
|
||||
entries,
|
||||
falsePositives,
|
||||
falseRecommendations,
|
||||
dossierId != null);
|
||||
})
|
||||
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
|
||||
.collect(Collectors.toList());
|
||||
return new DictionaryModel(t.getType(),
|
||||
t.getRank(),
|
||||
convertColor(t.getHexColor()),
|
||||
t.isCaseInsensitive(),
|
||||
t.isHint(),
|
||||
entries,
|
||||
falsePositives,
|
||||
falseRecommendations,
|
||||
dossierId != null);
|
||||
}).sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed()).collect(Collectors.toList());
|
||||
|
||||
dictionary.forEach(dm -> dictionaryRepresentation.getLocalAccessMap().put(dm.getType(), dm));
|
||||
|
||||
@@ -308,17 +264,17 @@ public class DictionaryService {
|
||||
var type = dictionaryClient.getDictionaryForType(typeId, fromVersion);
|
||||
|
||||
Set<DictionaryEntryModel> entries = type.getEntries() != null ? new HashSet<>(type.getEntries()
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
Set<DictionaryEntryModel> falsePositives = type.getFalsePositiveEntries() != null ? new HashSet<>(type.getFalsePositiveEntries()
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
Set<DictionaryEntryModel> falseRecommendations = type.getFalseRecommendationEntries() != null ? new HashSet<>(type.getFalseRecommendationEntries()
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
.stream()
|
||||
.map(DictionaryEntryModel::new)
|
||||
.collect(Collectors.toSet())) : new HashSet<>();
|
||||
|
||||
if (type.isCaseInsensitive()) {
|
||||
entries.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
|
||||
@@ -326,10 +282,10 @@ public class DictionaryService {
|
||||
falseRecommendations.forEach(entry -> entry.setValue(entry.getValue().toLowerCase(Locale.ROOT)));
|
||||
}
|
||||
log.debug("Dictionary update returned {} entries {} falsePositives and {} falseRecommendations for type {}",
|
||||
entries.size(),
|
||||
falsePositives.size(),
|
||||
falseRecommendations.size(),
|
||||
typeId);
|
||||
entries.size(),
|
||||
falsePositives.size(),
|
||||
falseRecommendations.size(),
|
||||
typeId);
|
||||
return new DictionaryEntries(entries, falsePositives, falseRecommendations);
|
||||
}
|
||||
|
||||
@@ -344,8 +300,7 @@ public class DictionaryService {
|
||||
@SneakyThrows
|
||||
public float[] getColor(String type, String dossierTemplateId) {
|
||||
|
||||
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap()
|
||||
.get(type);
|
||||
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap().get(type);
|
||||
if (model != null) {
|
||||
return model.getColor();
|
||||
}
|
||||
@@ -356,8 +311,7 @@ public class DictionaryService {
|
||||
@SneakyThrows
|
||||
public boolean isHint(String type, String dossierTemplateId) {
|
||||
|
||||
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap()
|
||||
.get(type);
|
||||
DictionaryModel model = getDossierTemplateDictionary(dossierTemplateId).getLocalAccessMap().get(type);
|
||||
if (model != null) {
|
||||
return model.isHint();
|
||||
}
|
||||
@@ -370,33 +324,26 @@ public class DictionaryService {
|
||||
@Observed(name = "DictionaryService", contextualName = "deep-copy-dictionary")
|
||||
public Dictionary getDeepCopyDictionary(String dossierTemplateId, String dossierId) {
|
||||
|
||||
List<DictionaryModel> mergedDictionaries = new LinkedList<>();
|
||||
List<DictionaryModel> mergedDictionaries;
|
||||
|
||||
DictionaryRepresentation dossierTemplateRepresentation = getDossierTemplateDictionary(dossierTemplateId);
|
||||
List<DictionaryModel> dossierTemplateDictionaries = dossierTemplateRepresentation.getDictionary();
|
||||
dossierTemplateDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
|
||||
var dossierTemplateRepresentation = getDossierTemplateDictionary(dossierTemplateId);
|
||||
var dossierTemplateDictionaries = dossierTemplateRepresentation.getDictionary();
|
||||
|
||||
// add dossier
|
||||
// merge dictionaries if they have same names
|
||||
long dossierDictionaryVersion = -1;
|
||||
if (dossierDictionaryExists(dossierId)) {
|
||||
DictionaryRepresentation dossierRepresentation = getDossierDictionary(dossierId);
|
||||
List<DictionaryModel> dossierDictionaries = dossierRepresentation.getDictionary();
|
||||
dossierDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
|
||||
return getDictionary(mergedDictionaries, dossierTemplateRepresentation, dossierRepresentation.getDictionaryVersion());
|
||||
var dossierRepresentation = getDossierDictionary(dossierId);
|
||||
var dossierDictionaries = dossierRepresentation.getDictionary();
|
||||
mergedDictionaries = convertCommonsDictionaryModel(dictionaryMergeService.getMergedDictionary(convertDictionaryModel(dossierTemplateDictionaries),
|
||||
convertDictionaryModel(dossierDictionaries)));
|
||||
dossierDictionaryVersion = dossierRepresentation.getDictionaryVersion();
|
||||
} else {
|
||||
return getDictionary(mergedDictionaries, dossierTemplateRepresentation, dossierDictionaryVersion);
|
||||
mergedDictionaries = new ArrayList<>();
|
||||
dossierTemplateDictionaries.forEach(dm -> mergedDictionaries.add(SerializationUtils.clone(dm)));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
private Dictionary getDictionary(List<DictionaryModel> mergedDictionaries, DictionaryRepresentation dossierTemplateRepresentation, long dossierDictionaryVersion) {
|
||||
|
||||
return new Dictionary(mergedDictionaries.stream()
|
||||
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
|
||||
.collect(Collectors.toList()),
|
||||
DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateRepresentation.getDictionaryVersion()).dossierVersion(dossierDictionaryVersion)
|
||||
.build());
|
||||
return new Dictionary(mergedDictionaries.stream().sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed()).collect(Collectors.toList()),
|
||||
DictionaryVersion.builder().dossierTemplateVersion(dossierTemplateRepresentation.getDictionaryVersion()).dossierVersion(dossierDictionaryVersion).build());
|
||||
}
|
||||
|
||||
|
||||
@@ -424,16 +371,14 @@ public class DictionaryService {
|
||||
@SneakyThrows
|
||||
private DictionaryRepresentation getDossierTemplateDictionary(String dossierTemplateId) {
|
||||
|
||||
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossierTemplate()
|
||||
.get(dossierTemplateId);
|
||||
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossierTemplate().get(dossierTemplateId);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
private DictionaryRepresentation getDossierDictionary(String dossierId) {
|
||||
|
||||
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossier()
|
||||
.get(dossierId);
|
||||
return tenantDictionaryCache.get(TenantContext.getTenantId()).getDictionariesByDossier().get(dossierId);
|
||||
}
|
||||
|
||||
|
||||
@@ -476,14 +421,14 @@ public class DictionaryService {
|
||||
|
||||
return commonsDictionaries.stream()
|
||||
.map(cd -> new DictionaryModel(cd.getType(),
|
||||
cd.getRank(),
|
||||
cd.getColor(),
|
||||
cd.isCaseInsensitive(),
|
||||
cd.isHint(),
|
||||
cd.getEntries(),
|
||||
cd.getFalsePositives(),
|
||||
cd.getFalseRecommendations(),
|
||||
cd.isDossierDictionary()))
|
||||
cd.getRank(),
|
||||
cd.getColor(),
|
||||
cd.isCaseInsensitive(),
|
||||
cd.isHint(),
|
||||
cd.getEntries(),
|
||||
cd.getFalsePositives(),
|
||||
cd.getFalseRecommendations(),
|
||||
cd.isDossierDictionary()))
|
||||
.collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
+43
-30
@@ -1,7 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.server.service;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
@@ -13,6 +13,9 @@ import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ChangeType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntryState;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.ManualChange;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.annotations.entitymapped.IdRemoval;
|
||||
|
||||
import io.micrometer.core.annotation.Timed;
|
||||
import lombok.AccessLevel;
|
||||
@@ -27,60 +30,75 @@ import lombok.extern.slf4j.Slf4j;
|
||||
public class EntityChangeLogService {
|
||||
|
||||
@Timed("redactmanager_computeChanges")
|
||||
public EntryChanges computeChanges(List<EntityLogEntry> previousEntityLogEntries, List<EntityLogEntry> newEntityLogEntries, int analysisNumber) {
|
||||
public boolean computeChanges(List<EntityLogEntry> previousEntityLogEntries, List<EntityLogEntry> newEntityLogEntries, ManualRedactions manualRedactions, int analysisNumber) {
|
||||
|
||||
var now = OffsetDateTime.now();
|
||||
if (previousEntityLogEntries.isEmpty()) {
|
||||
newEntityLogEntries.forEach(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.ADDED, now)));
|
||||
return new EntryChanges(newEntityLogEntries, new ArrayList<>());
|
||||
return true;
|
||||
}
|
||||
|
||||
List<EntityLogEntry> toInsert = new ArrayList<>();
|
||||
List<EntityLogEntry> toUpdate = new ArrayList<>();
|
||||
boolean hasChanges = false;
|
||||
|
||||
for (EntityLogEntry entityLogEntry : newEntityLogEntries) {
|
||||
Optional<EntityLogEntry> optionalPreviousEntity = previousEntityLogEntries.stream()
|
||||
.filter(entry -> entry.getId().equals(entityLogEntry.getId()))
|
||||
.findAny();
|
||||
if (optionalPreviousEntity.isEmpty()) {
|
||||
hasChanges = true;
|
||||
entityLogEntry.getChanges().add(new Change(analysisNumber, ChangeType.ADDED, now));
|
||||
toInsert.add(entityLogEntry);
|
||||
continue;
|
||||
}
|
||||
|
||||
EntityLogEntry previousEntity = optionalPreviousEntity.get();
|
||||
entityLogEntry.getChanges().addAll(previousEntity.getChanges());
|
||||
|
||||
if (!previousEntity.equals(entityLogEntry)) {
|
||||
if(!previousEntity.getState().equals(entityLogEntry.getState())) {
|
||||
ChangeType changeType = calculateChangeType(entityLogEntry.getState(), previousEntity.getState());
|
||||
entityLogEntry.getChanges().add(new Change(analysisNumber, changeType, now));
|
||||
}
|
||||
toUpdate.add(entityLogEntry);
|
||||
if (!previousEntity.getState().equals(entityLogEntry.getState())) {
|
||||
hasChanges = true;
|
||||
ChangeType changeType = calculateChangeType(entityLogEntry.getState(), previousEntity.getState());
|
||||
entityLogEntry.getChanges().add(new Change(analysisNumber, changeType, now));
|
||||
}
|
||||
}
|
||||
|
||||
toUpdate.addAll(addRemovedEntriesAsRemoved(previousEntityLogEntries, newEntityLogEntries, analysisNumber, now));
|
||||
return new EntryChanges(toInsert, toUpdate);
|
||||
addRemovedEntriesAsRemoved(previousEntityLogEntries, newEntityLogEntries, manualRedactions, analysisNumber, now);
|
||||
return hasChanges;
|
||||
}
|
||||
|
||||
|
||||
private List<EntityLogEntry> addRemovedEntriesAsRemoved(List<EntityLogEntry> previousEntityLogEntries,
|
||||
List<EntityLogEntry> newEntityLogEntries,
|
||||
int analysisNumber,
|
||||
OffsetDateTime now) {
|
||||
private void addRemovedEntriesAsRemoved(List<EntityLogEntry> previousEntityLogEntries,
|
||||
List<EntityLogEntry> newEntityLogEntries,
|
||||
ManualRedactions manualRedactions,
|
||||
int analysisNumber,
|
||||
OffsetDateTime now) {
|
||||
|
||||
Set<String> existingIds = newEntityLogEntries.stream()
|
||||
.map(EntityLogEntry::getId)
|
||||
.collect(Collectors.toSet());
|
||||
List<EntityLogEntry> removedEntries = previousEntityLogEntries.stream()
|
||||
.filter(entry -> !existingIds.contains(entry.getId()))
|
||||
.collect(Collectors.toList());
|
||||
List<EntityLogEntry> removedDossierRedaction = removedEntries.stream()
|
||||
.filter(e -> e.getState() == EntryState.REMOVED && e.getType().equals("dossier_redaction"))
|
||||
.toList();
|
||||
removedEntries.stream()
|
||||
.filter(entry -> !entry.getState().equals(EntryState.REMOVED))
|
||||
.peek(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.REMOVED, now)))
|
||||
.forEach(entry -> entry.setState(EntryState.REMOVED));
|
||||
previousEntityLogEntries.removeAll(removedDossierRedaction);
|
||||
removedEntries.removeAll(removedDossierRedaction);
|
||||
removedEntries.forEach(entry -> entry.getChanges().add(new Change(analysisNumber, ChangeType.REMOVED, now)));
|
||||
removedEntries.forEach(entry -> entry.setState(EntryState.REMOVED));
|
||||
removedEntries.forEach(entry -> addManualChangeForDictionaryRemovals(entry, manualRedactions));
|
||||
newEntityLogEntries.addAll(removedEntries);
|
||||
return removedEntries;
|
||||
}
|
||||
|
||||
|
||||
private void addManualChangeForDictionaryRemovals(EntityLogEntry entry, ManualRedactions manualRedactions) {
|
||||
|
||||
if (manualRedactions == null || manualRedactions.getIdsToRemove().isEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
manualRedactions.getIdsToRemove()
|
||||
.stream()
|
||||
.filter(IdRemoval::isRemoveFromDictionary)//
|
||||
.filter(removed -> removed.getAnnotationId().equals(entry.getId()))//
|
||||
.findFirst()//
|
||||
.ifPresent(idRemove -> entry.getManualChanges().add(ManualChangeFactory.toManualChange(idRemove, false)));
|
||||
}
|
||||
|
||||
|
||||
@@ -104,9 +122,4 @@ public class EntityChangeLogService {
|
||||
return (state.equals(EntryState.REMOVED) || state.equals(EntryState.IGNORED));
|
||||
}
|
||||
|
||||
|
||||
public record EntryChanges(List<EntityLogEntry> inserted, List<EntityLogEntry> updated) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+38
-41
@@ -32,7 +32,6 @@ import com.iqser.red.service.redaction.v1.server.model.document.entity.TextEntit
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.ImageType;
|
||||
import com.iqser.red.service.redaction.v1.server.service.EntityChangeLogService.EntryChanges;
|
||||
import com.iqser.red.service.redaction.v1.server.storage.RedactionStorageService;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
@@ -53,19 +52,17 @@ public class EntityLogCreatorService {
|
||||
RedactionStorageService redactionStorageService;
|
||||
|
||||
|
||||
private static boolean notFalsePositiveOrFalseRecommendationOrRemoval(TextEntity textEntity) {
|
||||
private static boolean notFalsePositiveOrFalseRecommendation(TextEntity textEntity) {
|
||||
|
||||
return !(textEntity.getEntityType().equals(EntityType.FALSE_POSITIVE) //
|
||||
|| textEntity.getEntityType().equals(EntityType.FALSE_RECOMMENDATION) //
|
||||
|| textEntity.getEntityType().equals(EntityType.DICTIONARY_REMOVAL));
|
||||
return !(textEntity.getEntityType().equals(EntityType.FALSE_POSITIVE) || textEntity.getEntityType().equals(EntityType.FALSE_RECOMMENDATION));
|
||||
}
|
||||
|
||||
|
||||
public EntityLogChanges createInitialEntityLog(AnalyzeRequest analyzeRequest,
|
||||
Document document,
|
||||
List<PrecursorEntity> notFoundEntities,
|
||||
DictionaryVersion dictionaryVersion,
|
||||
long rulesVersion) {
|
||||
public EntityLog createInitialEntityLog(AnalyzeRequest analyzeRequest,
|
||||
Document document,
|
||||
List<PrecursorEntity> notFoundEntities,
|
||||
DictionaryVersion dictionaryVersion,
|
||||
long rulesVersion) {
|
||||
|
||||
List<EntityLogEntry> entityLogEntries = createEntityLogEntries(document, analyzeRequest, notFoundEntities);
|
||||
|
||||
@@ -73,20 +70,16 @@ public class EntityLogCreatorService {
|
||||
|
||||
List<EntityLogEntry> previousExistingEntityLogEntries = getPreviousEntityLogEntries(analyzeRequest.getDossierId(), analyzeRequest.getFileId());
|
||||
|
||||
EntryChanges entryChanges = entityChangeLogService.computeChanges(previousExistingEntityLogEntries, entityLogEntries, analyzeRequest.getAnalysisNumber());
|
||||
entityChangeLogService.computeChanges(previousExistingEntityLogEntries, entityLogEntries, analyzeRequest.getManualRedactions(), analyzeRequest.getAnalysisNumber());
|
||||
|
||||
return EntityLogChanges.builder()
|
||||
.entityLog(new EntityLog(redactionServiceSettings.getAnalysisVersion(),
|
||||
analyzeRequest.getAnalysisNumber(),
|
||||
entityLogEntries,
|
||||
toEntityLogLegalBasis(legalBasis),
|
||||
dictionaryVersion.getDossierTemplateVersion(),
|
||||
dictionaryVersion.getDossierVersion(),
|
||||
rulesVersion,
|
||||
legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId())))
|
||||
.updatedEntityLogEntries(entryChanges.updated())
|
||||
.newEntityLogEntries(entryChanges.inserted())
|
||||
.build();
|
||||
return new EntityLog(redactionServiceSettings.getAnalysisVersion(),
|
||||
analyzeRequest.getAnalysisNumber(),
|
||||
entityLogEntries,
|
||||
toEntityLogLegalBasis(legalBasis),
|
||||
dictionaryVersion.getDossierTemplateVersion(),
|
||||
dictionaryVersion.getDossierVersion(),
|
||||
rulesVersion,
|
||||
legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId()));
|
||||
}
|
||||
|
||||
|
||||
@@ -100,11 +93,7 @@ public class EntityLogCreatorService {
|
||||
}
|
||||
|
||||
|
||||
public EntityLogChanges updateVersionsAndReturnChanges(EntityLog entityLog,
|
||||
DictionaryVersion dictionaryVersion,
|
||||
AnalyzeRequest analyzeRequest,
|
||||
List<EntityLogEntry> newEntries,
|
||||
List<EntityLogEntry> updatedEntries) {
|
||||
public EntityLogChanges updateVersionsAndReturnChanges(EntityLog entityLog, DictionaryVersion dictionaryVersion, AnalyzeRequest analyzeRequest, boolean hasChanges) {
|
||||
|
||||
List<LegalBasis> legalBasis = legalBasisClient.getLegalBasisMapping(analyzeRequest.getDossierTemplateId());
|
||||
entityLog.setLegalBasisVersion(legalBasisClient.getVersion(analyzeRequest.getDossierTemplateId()));
|
||||
@@ -113,14 +102,14 @@ public class EntityLogCreatorService {
|
||||
entityLog.setDossierDictionaryVersion(dictionaryVersion.getDossierVersion());
|
||||
entityLog.setAnalysisNumber(analyzeRequest.getAnalysisNumber());
|
||||
|
||||
return EntityLogChanges.builder().entityLog(entityLog).newEntityLogEntries(newEntries).updatedEntityLogEntries(updatedEntries).build();
|
||||
return new EntityLogChanges(entityLog, hasChanges);
|
||||
}
|
||||
|
||||
|
||||
public EntityLogChanges updatePreviousEntityLog(AnalyzeRequest analyzeRequest,
|
||||
Document document,
|
||||
EntityLog entityLogWithoutEntries,
|
||||
List<PrecursorEntity> notFoundEntries,
|
||||
EntityLog previousEntityLog,
|
||||
Set<Integer> sectionsToReanalyseIds,
|
||||
DictionaryVersion dictionaryVersion) {
|
||||
|
||||
@@ -128,14 +117,24 @@ public class EntityLogCreatorService {
|
||||
.filter(entry -> entry.getContainingNodeId().isEmpty() || sectionsToReanalyseIds.contains(entry.getContainingNodeId()
|
||||
.get(0)))
|
||||
.collect(Collectors.toList());
|
||||
Set<String> newEntityIds = newEntityLogEntries.stream()
|
||||
.map(EntityLogEntry::getId)
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
List<EntityLogEntry> previousEntriesFromReAnalyzedSections = redactionStorageService.findEntriesContainedBySectionsOrNotContained(analyzeRequest.getDossierId(),
|
||||
analyzeRequest.getFileId(),
|
||||
sectionsToReanalyseIds);
|
||||
List<EntityLogEntry> previousEntriesFromReAnalyzedSections = previousEntityLog.getEntityLogEntry()
|
||||
.stream()
|
||||
.filter(entry -> (newEntityIds.contains(entry.getId()) || entry.getContainingNodeId().isEmpty() || sectionsToReanalyseIds.contains(entry.getContainingNodeId()
|
||||
.get(0))))
|
||||
.collect(Collectors.toList());
|
||||
previousEntityLog.getEntityLogEntry().removeAll(previousEntriesFromReAnalyzedSections);
|
||||
|
||||
EntryChanges entryChanges = entityChangeLogService.computeChanges(previousEntriesFromReAnalyzedSections, newEntityLogEntries, analyzeRequest.getAnalysisNumber());
|
||||
boolean hasChanges = entityChangeLogService.computeChanges(previousEntriesFromReAnalyzedSections,
|
||||
newEntityLogEntries,
|
||||
analyzeRequest.getManualRedactions(),
|
||||
analyzeRequest.getAnalysisNumber());
|
||||
previousEntityLog.getEntityLogEntry().addAll(newEntityLogEntries);
|
||||
|
||||
return updateVersionsAndReturnChanges(entityLogWithoutEntries, dictionaryVersion, analyzeRequest, entryChanges.inserted(), entryChanges.updated());
|
||||
return updateVersionsAndReturnChanges(previousEntityLog, dictionaryVersion, analyzeRequest, hasChanges);
|
||||
}
|
||||
|
||||
|
||||
@@ -148,7 +147,7 @@ public class EntityLogCreatorService {
|
||||
document.getEntities()
|
||||
.stream()
|
||||
.filter(entity -> !entity.getValue().isEmpty())
|
||||
.filter(EntityLogCreatorService::notFalsePositiveOrFalseRecommendationOrRemoval)
|
||||
.filter(EntityLogCreatorService::notFalsePositiveOrFalseRecommendation)
|
||||
.filter(entity -> !entity.removed())
|
||||
.forEach(entityNode -> entries.addAll(toEntityLogEntries(entityNode)));
|
||||
document.streamAllImages()
|
||||
@@ -187,12 +186,11 @@ public class EntityLogCreatorService {
|
||||
|
||||
private EntityLogEntry createEntityLogEntry(Image image, String dossierTemplateId) {
|
||||
|
||||
String imageType = image.getImageType().equals(ImageType.OTHER) ? "image" : image.getImageType().toString().toLowerCase(Locale.ENGLISH);
|
||||
boolean isHint = dictionaryService.isHint(imageType, dossierTemplateId);
|
||||
boolean isHint = dictionaryService.isHint(image.type(), dossierTemplateId);
|
||||
return EntityLogEntry.builder()
|
||||
.id(image.getId())
|
||||
.value(image.getValue())
|
||||
.type(imageType)
|
||||
.value(image.value())
|
||||
.type(image.type())
|
||||
.reason(image.buildReasonWithManualChangeDescriptions())
|
||||
.legalBasis(image.legalBasis())
|
||||
.matchedRule(image.getMatchedRule().getRuleIdentifier().toString())
|
||||
@@ -344,7 +342,6 @@ public class EntityLogCreatorService {
|
||||
case FALSE_POSITIVE -> EntryType.FALSE_POSITIVE;
|
||||
case RECOMMENDATION -> EntryType.RECOMMENDATION;
|
||||
case FALSE_RECOMMENDATION -> EntryType.FALSE_RECOMMENDATION;
|
||||
case DICTIONARY_REMOVAL -> EntryType.FALSE_POSITIVE;
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
+2
-6
@@ -30,12 +30,8 @@ public class ManualChangeFactory {
|
||||
public ManualChange toManualChange(BaseAnnotation baseAnnotation, boolean isHint) {
|
||||
|
||||
ManualChange manualChange = ManualChange.from(baseAnnotation);
|
||||
if (baseAnnotation instanceof ManualRecategorization recategorization) {
|
||||
manualChange.withManualRedactionType(ManualRedactionType.RECATEGORIZE)
|
||||
.withChange("type", recategorization.getType())
|
||||
.withChange("section", recategorization.getSection())
|
||||
.withChange("legalBasis", recategorization.getLegalBasis())
|
||||
.withChange("value", recategorization.getValue());
|
||||
if (baseAnnotation instanceof ManualRecategorization imageRecategorization) {
|
||||
manualChange.withManualRedactionType(ManualRedactionType.RECATEGORIZE).withChange("type", imageRecategorization.getType());
|
||||
} else if (baseAnnotation instanceof IdRemoval manualRemoval) {
|
||||
manualChange.withManualRedactionType(manualRemoval.isRemoveFromDictionary() ? ManualRedactionType.REMOVE_FROM_DICTIONARY : ManualRedactionType.REMOVE);
|
||||
} else if (baseAnnotation instanceof ManualForceRedaction manualForceRedaction) {
|
||||
|
||||
+10
-35
@@ -58,12 +58,6 @@ public class ManualChangesApplicationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Resizes a text entity based on manual resize redaction details.
|
||||
*
|
||||
* @param entityToBeResized The entity to resize.
|
||||
* @param manualResizeRedaction The details of the resize operation.
|
||||
*/
|
||||
public void resize(TextEntity entityToBeResized, ManualResizeRedaction manualResizeRedaction) {
|
||||
|
||||
resizeEntityAndReinsert(entityToBeResized, manualResizeRedaction);
|
||||
@@ -80,9 +74,9 @@ public class ManualChangesApplicationService {
|
||||
.orElseThrow(() -> new NoSuchElementException("No redaction position with matching annotation id found!"));
|
||||
|
||||
positionOnPageToBeResized.setRectanglePerLine(manualResizeRedaction.getPositions()
|
||||
.stream()
|
||||
.map(ManualChangesApplicationService::toRectangle2D)
|
||||
.collect(Collectors.toList()));
|
||||
.stream()
|
||||
.map(ManualChangesApplicationService::toRectangle2D)
|
||||
.collect(Collectors.toList()));
|
||||
|
||||
entityToBeResized.getManualOverwrite().addChange(manualResizeRedaction);
|
||||
|
||||
@@ -96,17 +90,11 @@ public class ManualChangesApplicationService {
|
||||
|
||||
if (closestEntity.isPresent()) {
|
||||
copyValuesFromClosestEntity(entityToBeResized, manualResizeRedaction, closestEntity.get());
|
||||
possibleEntities.values()
|
||||
.stream()
|
||||
.flatMap(Collection::stream)
|
||||
.forEach(TextEntity::removeFromGraph);
|
||||
possibleEntities.values().stream().flatMap(Collection::stream).forEach(TextEntity::removeFromGraph);
|
||||
return;
|
||||
}
|
||||
|
||||
possibleEntities.values()
|
||||
.stream()
|
||||
.flatMap(Collection::stream)
|
||||
.forEach(TextEntity::removeFromGraph);
|
||||
possibleEntities.values().stream().flatMap(Collection::stream).forEach(TextEntity::removeFromGraph);
|
||||
|
||||
if (node.hasParent()) {
|
||||
node = node.getParent();
|
||||
@@ -122,18 +110,14 @@ public class ManualChangesApplicationService {
|
||||
Set<SemanticNode> currentIntersectingNodes = new HashSet<>(entityToBeResized.getIntersectingNodes());
|
||||
Set<SemanticNode> newIntersectingNodes = new HashSet<>(closestEntity.getIntersectingNodes());
|
||||
|
||||
Sets.difference(currentIntersectingNodes, newIntersectingNodes)
|
||||
.forEach(removedNode -> removedNode.getEntities().remove(entityToBeResized));
|
||||
Sets.difference(newIntersectingNodes, currentIntersectingNodes)
|
||||
.forEach(addedNode -> addedNode.getEntities().add(entityToBeResized));
|
||||
Sets.difference(currentIntersectingNodes, newIntersectingNodes).forEach(removedNode -> removedNode.getEntities().remove(entityToBeResized));
|
||||
Sets.difference(newIntersectingNodes, currentIntersectingNodes).forEach(addedNode -> addedNode.getEntities().add(entityToBeResized));
|
||||
|
||||
Set<Page> currentIntersectingPages = new HashSet<>(entityToBeResized.getPages());
|
||||
Set<Page> newIntersectingPages = new HashSet<>(closestEntity.getPages());
|
||||
|
||||
Sets.difference(currentIntersectingPages, newIntersectingPages)
|
||||
.forEach(removedPage -> removedPage.getEntities().remove(entityToBeResized));
|
||||
Sets.difference(newIntersectingPages, currentIntersectingPages)
|
||||
.forEach(addedPage -> addedPage.getEntities().add(entityToBeResized));
|
||||
Sets.difference(currentIntersectingPages, newIntersectingPages).forEach(removedPage -> removedPage.getEntities().remove(entityToBeResized));
|
||||
Sets.difference(newIntersectingPages, currentIntersectingPages).forEach(addedPage -> addedPage.getEntities().add(entityToBeResized));
|
||||
|
||||
entityToBeResized.setDeepestFullyContainingNode(closestEntity.getDeepestFullyContainingNode());
|
||||
entityToBeResized.setIntersectingNodes(new ArrayList<>(newIntersectingNodes));
|
||||
@@ -146,21 +130,12 @@ public class ManualChangesApplicationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Resizes an image entity based on manual resize redaction instructions.
|
||||
*
|
||||
* @param image The image to resize.
|
||||
* @param manualResizeRedaction The details of the resize operation.
|
||||
*/
|
||||
public void resizeImage(Image image, ManualResizeRedaction manualResizeRedaction) {
|
||||
|
||||
if (manualResizeRedaction.getPositions().isEmpty() || manualResizeRedaction.getPositions() == null) {
|
||||
return;
|
||||
}
|
||||
var bBox = RectangleTransformations.rectangle2DBBox(manualResizeRedaction.getPositions()
|
||||
.stream()
|
||||
.map(ManualChangesApplicationService::toRectangle2D)
|
||||
.toList());
|
||||
var bBox = RectangleTransformations.rectangle2DBBox(manualResizeRedaction.getPositions().stream().map(ManualChangesApplicationService::toRectangle2D).toList());
|
||||
image.setPosition(bBox);
|
||||
image.getManualOverwrite().addChange(manualResizeRedaction);
|
||||
}
|
||||
|
||||
+5
-11
@@ -53,13 +53,10 @@ public class NotFoundImportedEntitiesService {
|
||||
if (!notFoundEntities.isEmpty()) {
|
||||
// imported redactions present, intersections must be added with merged imported redactions
|
||||
Map<Integer, List<PrecursorEntity>> importedRedactionsMap = mapImportedRedactionsOnPage(notFoundEntities);
|
||||
entityLog.getEntityLogEntry()
|
||||
.stream()
|
||||
.filter(entry -> !entry.getEngines().contains(Engine.IMPORTED))
|
||||
.forEach(redactionLogEntry -> {
|
||||
redactionLogEntry.setImportedRedactionIntersections(new HashSet<>());
|
||||
addIntersections(redactionLogEntry, importedRedactionsMap, analysisNumber);
|
||||
});
|
||||
entityLog.getEntityLogEntry().stream().filter(entry -> !entry.getEngines().contains(Engine.IMPORTED)).forEach(redactionLogEntry -> {
|
||||
redactionLogEntry.setImportedRedactionIntersections(new HashSet<>());
|
||||
addIntersections(redactionLogEntry, importedRedactionsMap, analysisNumber);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -73,10 +70,7 @@ public class NotFoundImportedEntitiesService {
|
||||
.map(RectangleWithPage::pageNumber)
|
||||
.collect(Collectors.toSet());
|
||||
pageNumbers.forEach(pageNumber -> importedRedactionsMap.put(pageNumber,
|
||||
importedEntities.stream()
|
||||
.filter(i -> pageNumber == i.getEntityPosition()
|
||||
.get(0).pageNumber())
|
||||
.collect(Collectors.toList())));
|
||||
importedEntities.stream().filter(i -> pageNumber == i.getEntityPosition().get(0).pageNumber()).collect(Collectors.toList())));
|
||||
return importedRedactionsMap;
|
||||
}
|
||||
|
||||
|
||||
+2
-6
@@ -15,12 +15,8 @@ public class ComponentComparator implements Comparator<Component> {
|
||||
@Override
|
||||
public int compare(Component component1, Component component2) {
|
||||
|
||||
var firstEntity1 = component1.getReferences()
|
||||
.stream()
|
||||
.min(EntityComparators.first());
|
||||
var firstEntity2 = component2.getReferences()
|
||||
.stream()
|
||||
.min(EntityComparators.first());
|
||||
var firstEntity1 = component1.getReferences().stream().min(EntityComparators.first());
|
||||
var firstEntity2 = component2.getReferences().stream().min(EntityComparators.first());
|
||||
if (firstEntity1.isEmpty() && firstEntity2.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
+68
-104
@@ -40,8 +40,7 @@ public class ComponentCreationService {
|
||||
|
||||
private static List<Entity> findEntitiesFromLongestSection(Collection<Entity> entities) {
|
||||
|
||||
var entitiesBySection = entities.stream()
|
||||
.collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
|
||||
var entitiesBySection = entities.stream().collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
|
||||
Optional<SemanticNode> longestSection = entitiesBySection.entrySet()
|
||||
.stream()
|
||||
.sorted(Comparator.comparingInt(ComponentCreationService::getTotalLengthOfEntities).reversed())
|
||||
@@ -80,20 +79,14 @@ public class ComponentCreationService {
|
||||
public void firstOrElse(String ruleIdentifier, String name, Collection<Entity> entities, String fallback) {
|
||||
|
||||
String valueDescription = String.format("First found value of type %s or else '%s'", joinTypes(entities), fallback);
|
||||
String value = entities.stream()
|
||||
.min(EntityComparators.first())
|
||||
.map(Entity::getValue)
|
||||
.orElse(fallback);
|
||||
String value = entities.stream().min(EntityComparators.first()).map(Entity::getValue).orElse(fallback);
|
||||
create(ruleIdentifier, name, value, valueDescription, entities);
|
||||
}
|
||||
|
||||
|
||||
private static String joinTypes(Collection<Entity> entities) {
|
||||
|
||||
return entities.stream()
|
||||
.map(Entity::getType)
|
||||
.distinct()
|
||||
.collect(Collectors.joining(", "));
|
||||
return entities.stream().map(Entity::getType).distinct().collect(Collectors.joining(", "));
|
||||
}
|
||||
|
||||
|
||||
@@ -111,12 +104,12 @@ public class ComponentCreationService {
|
||||
referencedEntities.addAll(references);
|
||||
|
||||
kieSession.insert(Component.builder()
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription(valueDescription)
|
||||
.references(new LinkedList<>(references))
|
||||
.build());
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription(valueDescription)
|
||||
.references(new LinkedList<>(references))
|
||||
.build());
|
||||
}
|
||||
|
||||
|
||||
@@ -149,11 +142,8 @@ public class ComponentCreationService {
|
||||
|
||||
private static List<Entity> findEntitiesFromFirstSection(Collection<Entity> entities) {
|
||||
|
||||
var entitiesBySection = entities.stream()
|
||||
.collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
|
||||
Optional<SemanticNode> firstSection = entitiesBySection.keySet()
|
||||
.stream()
|
||||
.min(SemanticNodeComparators.first());
|
||||
var entitiesBySection = entities.stream().collect(Collectors.groupingBy(entity -> entity.getContainingNode().getHighestParent()));
|
||||
Optional<SemanticNode> firstSection = entitiesBySection.keySet().stream().min(SemanticNodeComparators.first());
|
||||
if (firstSection.isEmpty()) {
|
||||
return Collections.emptyList();
|
||||
}
|
||||
@@ -198,10 +188,7 @@ public class ComponentCreationService {
|
||||
public void joining(String ruleIdentifier, String name, Collection<Entity> entities, String delimiter) {
|
||||
|
||||
String valueDescription = String.format("Joining all values of type %s with '%s'", joinTypes(entities), delimiter);
|
||||
String value = entities.stream()
|
||||
.sorted(EntityComparators.first())
|
||||
.map(Entity::getValue)
|
||||
.collect(Collectors.joining(delimiter));
|
||||
String value = entities.stream().sorted(EntityComparators.first()).map(Entity::getValue).collect(Collectors.joining(delimiter));
|
||||
create(ruleIdentifier, name, value, valueDescription, entities);
|
||||
}
|
||||
|
||||
@@ -244,20 +231,14 @@ public class ComponentCreationService {
|
||||
public void joiningUnique(String ruleIdentifier, String name, Collection<Entity> entities, String delimiter) {
|
||||
|
||||
String valueDescription = String.format("Joining all unique values of type %s with '%s'", joinTypes(entities), delimiter);
|
||||
String value = entities.stream()
|
||||
.sorted(EntityComparators.first())
|
||||
.map(Entity::getValue)
|
||||
.distinct()
|
||||
.collect(Collectors.joining(delimiter));
|
||||
String value = entities.stream().sorted(EntityComparators.first()).map(Entity::getValue).distinct().collect(Collectors.joining(delimiter));
|
||||
create(ruleIdentifier, name, value, valueDescription, entities);
|
||||
}
|
||||
|
||||
|
||||
private static int getTotalLengthOfEntities(Map.Entry<SemanticNode, List<Entity>> entry) {
|
||||
|
||||
return entry.getValue()
|
||||
.stream()
|
||||
.mapToInt(Entity::getLength).sum();
|
||||
return entry.getValue().stream().mapToInt(Entity::getLength).sum();
|
||||
}
|
||||
|
||||
|
||||
@@ -312,10 +293,7 @@ public class ComponentCreationService {
|
||||
*/
|
||||
public void uniqueValueCount(String ruleIdentifier, String name, Collection<Entity> entities) {
|
||||
|
||||
long count = entities.stream()
|
||||
.map(Entity::getValue)
|
||||
.distinct()
|
||||
.count();
|
||||
long count = entities.stream().map(Entity::getValue).distinct().count();
|
||||
create(ruleIdentifier, name, String.valueOf(count), "Number of unique values in the entity references", entities);
|
||||
}
|
||||
|
||||
@@ -329,20 +307,18 @@ public class ComponentCreationService {
|
||||
*/
|
||||
public void rowValueCount(String ruleIdentifier, String name, Collection<Entity> entities) {
|
||||
|
||||
entities.stream()
|
||||
.collect(Collectors.groupingBy(this::getFirstTable))
|
||||
.forEach((optionalTable, groupedEntities) -> {
|
||||
entities.stream().collect(Collectors.groupingBy(this::getFirstTable)).forEach((optionalTable, groupedEntities) -> {
|
||||
|
||||
if (optionalTable.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
if (optionalTable.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
long count = groupedEntities.stream()
|
||||
.collect(Collectors.groupingBy(entity -> getFirstTableCell(entity).map(TableCell::getRow)
|
||||
.orElse(-1))).size();
|
||||
long count = groupedEntities.stream()
|
||||
.collect(Collectors.groupingBy(entity -> getFirstTableCell(entity).map(TableCell::getRow).orElse(-1)))
|
||||
.size();
|
||||
|
||||
create(ruleIdentifier, name, String.valueOf(count), "Count rows with values in the entity references in same table", entities);
|
||||
});
|
||||
create(ruleIdentifier, name, String.valueOf(count), "Count rows with values in the entity references in same table", entities);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -358,20 +334,18 @@ public class ComponentCreationService {
|
||||
if (entities.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
entities.stream()
|
||||
.sorted(EntityComparators.first())
|
||||
.forEach(entity -> {
|
||||
BreakIterator iterator = BreakIterator.getSentenceInstance(Locale.ENGLISH);
|
||||
iterator.setText(entity.getValue());
|
||||
int start = iterator.first();
|
||||
for (int end = iterator.next(); end != BreakIterator.DONE; start = end, end = iterator.next()) {
|
||||
create(ruleIdentifier,
|
||||
name,
|
||||
entity.getValue().substring(start, end).replaceAll("\\n", "").trim(),
|
||||
String.format("Values of type '%s' as sentences", entity.getType()),
|
||||
entity);
|
||||
}
|
||||
});
|
||||
entities.stream().sorted(EntityComparators.first()).forEach(entity -> {
|
||||
BreakIterator iterator = BreakIterator.getSentenceInstance(Locale.ENGLISH);
|
||||
iterator.setText(entity.getValue());
|
||||
int start = iterator.first();
|
||||
for (int end = iterator.next(); end != BreakIterator.DONE; start = end, end = iterator.next()) {
|
||||
create(ruleIdentifier,
|
||||
name,
|
||||
entity.getValue().substring(start, end).replaceAll("\\n", "").trim(),
|
||||
String.format("Values of type '%s' as sentences", entity.getType()),
|
||||
entity);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -392,12 +366,12 @@ public class ComponentCreationService {
|
||||
List<Entity> referenceList = new LinkedList<>();
|
||||
referenceList.add(reference);
|
||||
kieSession.insert(Component.builder()
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription(valueDescription)
|
||||
.references(referenceList)
|
||||
.build());
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription(valueDescription)
|
||||
.references(referenceList)
|
||||
.build());
|
||||
}
|
||||
|
||||
|
||||
@@ -454,10 +428,8 @@ public class ComponentCreationService {
|
||||
}
|
||||
|
||||
String formattedDateStrings = Stream.concat(//
|
||||
dates.stream()
|
||||
.sorted()
|
||||
.map(date -> DateConverter.convertDate(date, resultFormat)), //
|
||||
unparsedDates.stream())//
|
||||
dates.stream().sorted().map(date -> DateConverter.convertDate(date, resultFormat)), //
|
||||
unparsedDates.stream())//
|
||||
.collect(Collectors.joining(", "));
|
||||
|
||||
create(ruleIdentifier, name, formattedDateStrings, valueDescription, entities);
|
||||
@@ -473,34 +445,26 @@ public class ComponentCreationService {
|
||||
*/
|
||||
public void joiningFromSameTableRow(String ruleIdentifier, String name, Collection<Entity> entities) {
|
||||
|
||||
String types = entities.stream()
|
||||
.map(Entity::getType)
|
||||
.sorted(Comparator.reverseOrder())
|
||||
.distinct()
|
||||
.collect(Collectors.joining(", "));
|
||||
String types = entities.stream().map(Entity::getType).sorted(Comparator.reverseOrder()).distinct().collect(Collectors.joining(", "));
|
||||
String valueDescription = String.format("Combine values of %s that are in same table row", types);
|
||||
entities.stream()
|
||||
.collect(Collectors.groupingBy(this::getFirstTable))
|
||||
.forEach((optionalTable, groupedEntities) -> {
|
||||
if (optionalTable.isEmpty()) {
|
||||
groupedEntities.forEach(entity -> create(ruleIdentifier, name, entity.getValue(), valueDescription, entity));
|
||||
}
|
||||
entities.stream().collect(Collectors.groupingBy(this::getFirstTable)).forEach((optionalTable, groupedEntities) -> {
|
||||
if (optionalTable.isEmpty()) {
|
||||
groupedEntities.forEach(entity -> create(ruleIdentifier, name, entity.getValue(), valueDescription, entity));
|
||||
}
|
||||
|
||||
groupedEntities.stream()
|
||||
.filter(entity -> entity.getContainingNode() instanceof TableCell)
|
||||
.collect(Collectors.groupingBy(entity -> ((TableCell) entity.getContainingNode()).getRow())).entrySet()
|
||||
.stream()
|
||||
.sorted(Comparator.comparingInt(Map.Entry::getKey))
|
||||
.map(Map.Entry::getValue)
|
||||
.forEach(entitiesInSameRow -> create(ruleIdentifier,
|
||||
name,
|
||||
entitiesInSameRow.stream()
|
||||
.sorted(EntityComparators.first())
|
||||
.map(Entity::getValue)
|
||||
.collect(Collectors.joining(", ")),
|
||||
valueDescription,
|
||||
entitiesInSameRow));
|
||||
});
|
||||
groupedEntities.stream()
|
||||
.filter(entity -> entity.getContainingNode() instanceof TableCell)
|
||||
.collect(Collectors.groupingBy(entity -> ((TableCell) entity.getContainingNode()).getRow()))
|
||||
.entrySet()
|
||||
.stream()
|
||||
.sorted(Comparator.comparingInt(Map.Entry::getKey))
|
||||
.map(Map.Entry::getValue)
|
||||
.forEach(entitiesInSameRow -> create(ruleIdentifier,
|
||||
name,
|
||||
entitiesInSameRow.stream().sorted(EntityComparators.first()).map(Entity::getValue).collect(Collectors.joining(", ")),
|
||||
valueDescription,
|
||||
entitiesInSameRow));
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -557,12 +521,12 @@ public class ComponentCreationService {
|
||||
public void create(String ruleIdentifier, String name, String value) {
|
||||
|
||||
kieSession.insert(Component.builder()
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription("")
|
||||
.references(Collections.emptyList())
|
||||
.build());
|
||||
.matchedRule(RuleIdentifier.fromString(ruleIdentifier))
|
||||
.name(name)
|
||||
.value(value)
|
||||
.valueDescription("")
|
||||
.references(Collections.emptyList())
|
||||
.build());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+13
-33
@@ -9,7 +9,6 @@ import java.util.NoSuchElementException;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.DocumentTree;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.DuplicatedParagraph;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Footer;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Header;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.nodes.Image;
|
||||
@@ -41,9 +40,7 @@ public class DocumentGraphMapper {
|
||||
DocumentTree documentTree = new DocumentTree(document);
|
||||
Context context = new Context(documentData, documentTree);
|
||||
|
||||
context.pageData.addAll(Arrays.stream(documentData.getDocumentPages())
|
||||
.map(DocumentGraphMapper::buildPage)
|
||||
.toList());
|
||||
context.pageData.addAll(Arrays.stream(documentData.getDocumentPages()).map(DocumentGraphMapper::buildPage).toList());
|
||||
|
||||
context.documentTree.getRoot().getChildren().addAll(buildEntries(documentData.getDocumentStructure().getRoot().getChildren(), context));
|
||||
|
||||
@@ -61,13 +58,11 @@ public class DocumentGraphMapper {
|
||||
List<DocumentTree.Entry> newEntries = new LinkedList<>();
|
||||
for (DocumentStructure.EntryData entryData : entries) {
|
||||
|
||||
List<Page> pages = Arrays.stream(entryData.getPageNumbers())
|
||||
.map(pageNumber -> getPage(pageNumber, context))
|
||||
.toList();
|
||||
List<Page> pages = Arrays.stream(entryData.getPageNumbers()).map(pageNumber -> getPage(pageNumber, context)).toList();
|
||||
|
||||
SemanticNode node = switch (entryData.getType()) {
|
||||
case SECTION -> buildSection(context);
|
||||
case PARAGRAPH -> buildParagraph(context, entryData.getProperties());
|
||||
case PARAGRAPH -> buildParagraph(context);
|
||||
case HEADLINE -> buildHeadline(context);
|
||||
case HEADER -> buildHeader(context);
|
||||
case FOOTER -> buildFooter(context);
|
||||
@@ -81,10 +76,8 @@ public class DocumentGraphMapper {
|
||||
TextBlock textBlock = toTextBlock(entryData.getAtomicBlockIds(), context, node);
|
||||
node.setLeafTextBlock(textBlock);
|
||||
}
|
||||
List<Integer> treeId = Arrays.stream(entryData.getTreeId()).boxed()
|
||||
.toList();
|
||||
entryData.getEngines()
|
||||
.forEach(engine -> node.addEngine(engine));
|
||||
List<Integer> treeId = Arrays.stream(entryData.getTreeId()).boxed().toList();
|
||||
entryData.getEngines().forEach(engine -> node.addEngine(engine));
|
||||
node.setTreeId(treeId);
|
||||
|
||||
switch (entryData.getType()) {
|
||||
@@ -149,35 +142,24 @@ public class DocumentGraphMapper {
|
||||
}
|
||||
|
||||
|
||||
private Paragraph buildParagraph(Context context, Map<String, String> properties) {
|
||||
|
||||
if (PropertiesMapper.isDuplicateParagraph(properties)) {
|
||||
|
||||
DuplicatedParagraph duplicatedParagraph = DuplicatedParagraph.builder().documentTree(context.documentTree).build();
|
||||
|
||||
Long[] unsortedTextblockIds = PropertiesMapper.getUnsortedTextblockIds(properties);
|
||||
duplicatedParagraph.setUnsortedLeafTextBlock(toTextBlock(unsortedTextblockIds, context, duplicatedParagraph));
|
||||
return duplicatedParagraph;
|
||||
}
|
||||
private Paragraph buildParagraph(Context context) {
|
||||
|
||||
return Paragraph.builder().documentTree(context.documentTree).build();
|
||||
}
|
||||
|
||||
|
||||
private TextBlock toTextBlock(Long[] atomicTextBlockIds, Context context, SemanticNode parent) {
|
||||
private TextBlock toTextBlock(Long[] atomicTextBlockIds, Context context, SemanticNode parent) {
|
||||
|
||||
return Arrays.stream(atomicTextBlockIds)
|
||||
.map(atomicTextBlockId -> getAtomicTextBlock(context, parent, atomicTextBlockId))
|
||||
.collect(new TextBlockCollector());
|
||||
return Arrays.stream(atomicTextBlockIds).map(atomicTextBlockId -> getAtomicTextBlock(context, parent, atomicTextBlockId)).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock getAtomicTextBlock(Context context, SemanticNode parent, Long atomicTextBlockId) {
|
||||
|
||||
return AtomicTextBlock.fromAtomicTextBlockData(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)),
|
||||
context.documentPositionData.get(Math.toIntExact(atomicTextBlockId)),
|
||||
parent,
|
||||
getPage(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)).getPage(), context));
|
||||
context.documentPositionData.get(Math.toIntExact(atomicTextBlockId)),
|
||||
parent,
|
||||
getPage(context.documentTextData.get(Math.toIntExact(atomicTextBlockId)).getPage(), context));
|
||||
}
|
||||
|
||||
|
||||
@@ -208,10 +190,8 @@ public class DocumentGraphMapper {
|
||||
|
||||
this.documentTree = documentTree;
|
||||
this.pageData = new LinkedList<>();
|
||||
this.documentTextData = Arrays.stream(documentData.getDocumentTextData())
|
||||
.toList();
|
||||
this.documentPositionData = Arrays.stream(documentData.getDocumentPositionData())
|
||||
.toList();
|
||||
this.documentTextData = Arrays.stream(documentData.getDocumentTextData()).toList();
|
||||
this.documentPositionData = Arrays.stream(documentData.getDocumentPositionData()).toList();
|
||||
|
||||
}
|
||||
|
||||
|
||||
-2
@@ -11,7 +11,6 @@ public abstract class EntityComparators implements Comparator<Entity> {
|
||||
return new FirstEntity();
|
||||
}
|
||||
|
||||
|
||||
public static class LongestEntity implements Comparator<Entity> {
|
||||
|
||||
@Override
|
||||
@@ -28,7 +27,6 @@ public abstract class EntityComparators implements Comparator<Entity> {
|
||||
return new LongestEntity();
|
||||
}
|
||||
|
||||
|
||||
public static class FirstEntity implements Comparator<Entity> {
|
||||
|
||||
@Override
|
||||
|
||||
+31
-550
@@ -54,17 +54,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, case-sensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStrings(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -76,17 +65,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, case-insensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -98,17 +76,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the start string in the entity, case-sensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeStart(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -125,17 +92,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the start string in the entity, case-insensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeStartIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -152,17 +108,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the end string in the entity, case-sensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeEnd(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -179,17 +124,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the end string in the entity, case-insensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -206,17 +140,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the start and end string in the entity, case-sensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeStartAndEnd(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -237,17 +160,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities found between specified start and stop strings, including the start and end string in the entity, case-insensitive.
|
||||
*
|
||||
* @param start The starting string to search for.
|
||||
* @param stop The stopping string to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> betweenStringsIncludeStartAndEndIgnoreCase(String start, String stop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(start, stop);
|
||||
@@ -268,17 +180,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node, case-sensitive.
|
||||
*
|
||||
* @param starts A list of start strings to search for.
|
||||
* @param stops A list of stop strings to search for.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> shortestBetweenAnyString(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(starts, stops);
|
||||
@@ -290,17 +191,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node, case-insensitive.
|
||||
*
|
||||
* @param starts A list of start strings to search for.
|
||||
* @param stops A list of stop strings to search for.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which to search.
|
||||
* @return A stream of {@link TextEntity} identified objects.
|
||||
* @throws IllegalArgumentException if both start and stop strings are empty, indicating there's nothing to search for.
|
||||
*/
|
||||
public Stream<TextEntity> shortestBetweenAnyStringIgnoreCase(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(starts, stops);
|
||||
@@ -312,18 +202,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies the shortest text entities found between any of the given start and stop strings within a specified semantic node,
|
||||
* case-insensitive, with a length limit.
|
||||
*
|
||||
* @param starts A list of start strings to search for, case-insensitively.
|
||||
* @param stops A list of stop strings to search for, case-insensitively.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @param limit The maximum length of the entity text.
|
||||
* @return A stream of {@link TextEntity} objects found between any of the start and stop strings, case-insensitively, and within the specified limit.
|
||||
*/
|
||||
public Stream<TextEntity> shortestBetweenAnyStringIgnoreCase(List<String> starts, List<String> stops, String type, EntityType entityType, SemanticNode node, int limit) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(starts, stops);
|
||||
@@ -335,16 +213,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities based on the boundaries identified between start and stop regular expressions within a specified semantic node.
|
||||
*
|
||||
* @param regexStart The regular expression defining the start boundary.
|
||||
* @param regexStop The regular expression defining the stop boundary.
|
||||
* @param type The type of entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects identified between the start and stop regular expressions.
|
||||
*/
|
||||
public Stream<TextEntity> betweenRegexes(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -355,17 +223,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities based on the boundaries identified between start and stop regular expressions within a specified semantic node,
|
||||
* case-insensitive.
|
||||
*
|
||||
* @param regexStart The regular expression defining the start boundary, case-insensitive.
|
||||
* @param regexStop The regular expression defining the stop boundary, case-insensitive.
|
||||
* @param type The type of entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects identified between the start and stop regular expressions, case-insensitively.
|
||||
*/
|
||||
public Stream<TextEntity> betweenRegexesIgnoreCase(String regexStart, String regexStop, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -376,35 +233,12 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities based on the boundaries identified between specified start and stop text ranges within a semantic node.
|
||||
* This is a more general method that can be used directly with lists of start and stop {@link TextRange} objects.
|
||||
*
|
||||
* @param startBoundaries A list of start text range boundaries.
|
||||
* @param stopBoundaries A list of stop text range boundaries.
|
||||
* @param type The type of entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects identified between the start and stop text ranges.
|
||||
*/
|
||||
public Stream<TextEntity> betweenTextRanges(List<TextRange> startBoundaries, List<TextRange> stopBoundaries, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return betweenTextRanges(startBoundaries, stopBoundaries, type, entityType, node, 0);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates entities based on the boundaries identified between specified start and stop text ranges within a semantic node,
|
||||
* with an optional length limit for the entities.
|
||||
*
|
||||
* @param startBoundaries A list of start text range boundaries.
|
||||
* @param stopBoundaries A list of stop text range boundaries.
|
||||
* @param type The type of entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @param limit The maximum length of the entity text; use 0 for no limit.
|
||||
* @return A stream of {@link TextEntity} objects identified between the start and stop text ranges, within the specified limit.
|
||||
*/
|
||||
public Stream<TextEntity> betweenTextRanges(List<TextRange> startBoundaries, List<TextRange> stopBoundaries, String type, EntityType entityType, SemanticNode node, int limit) {
|
||||
|
||||
List<TextRange> entityBoundaries = findNonOverlappingBoundariesBetweenBoundariesWithMinimalDistances(startBoundaries, stopBoundaries);
|
||||
@@ -442,22 +276,12 @@ public class EntityCreationService {
|
||||
"this is some text. a here is more text" and "here is more text". We only want to keep the latter.
|
||||
*/
|
||||
return entityTextRanges.stream()
|
||||
.filter(boundary -> entityTextRanges.stream()
|
||||
.noneMatch(innerBoundary -> !innerBoundary.equals(boundary) && innerBoundary.containedBy(boundary)))
|
||||
.filter(boundary -> entityTextRanges.stream().noneMatch(innerBoundary -> !innerBoundary.equals(boundary) && innerBoundary.containedBy(boundary)))
|
||||
.toList();
|
||||
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates text entities based on boundaries identified by a search implementation within a specified semantic node.
|
||||
*
|
||||
* @param searchImplementation The search implementation to use for identifying boundaries.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects corresponding to the identified boundaries.
|
||||
*/
|
||||
public Stream<TextEntity> bySearchImplementation(SearchImplementation searchImplementation, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
|
||||
@@ -469,15 +293,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities located immediately after the specified strings within a semantic node.
|
||||
*
|
||||
* @param strings A list of strings to search for. The text immediately following each string is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found immediately after the specified strings.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterStrings(List<String> strings, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -492,15 +307,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities located immediately after the specified strings within a semantic node, case-insensitive.
|
||||
*
|
||||
* @param strings A list of strings to search for, case-insensitive. The text immediately following each string is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found immediately after the specified strings, case-insensitively.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterStringsIgnoreCase(List<String> strings, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -515,15 +321,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies a text entity located immediately after a specified string within a semantic node.
|
||||
*
|
||||
* @param string The string to search for. The text immediately following this string is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found immediately after the specified string.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterString(String string, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -537,15 +334,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies a text entity located immediately after a specified string within a semantic node, case-insensitive.
|
||||
*
|
||||
* @param string The string to search for, case-insensitive. The text immediately following this string is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param node The semantic node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found immediately after the specified string, case-insensitively.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterStringIgnoreCase(String string, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextBlock textBlock = node.getTextBlock();
|
||||
@@ -559,43 +347,25 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities located immediately after a specified string across table cell columns within a table node.
|
||||
*
|
||||
* @param string The string to search for. The text immediately following this string in subsequent table cells is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param tableNode The table node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found across table cell columns immediately after the specified string.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterStringAcrossColumns(String string, String type, EntityType entityType, Table tableNode) {
|
||||
|
||||
return tableNode.streamTableCells()
|
||||
.flatMap(tableCell -> lineAfterBoundariesAcrossColumns(RedactionSearchUtility.findTextRangesByString(string, tableCell.getTextBlock()),
|
||||
tableCell,
|
||||
type,
|
||||
entityType,
|
||||
tableNode));
|
||||
tableCell,
|
||||
type,
|
||||
entityType,
|
||||
tableNode));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities located immediately after a specified string across table cell columns within a table node, case-insensitive.
|
||||
*
|
||||
* @param string The string to search for, case-insensitive. The text immediately following this string in subsequent table cells is considered for entity creation.
|
||||
* @param type The type of the entity to be created.
|
||||
* @param entityType The detailed classification of the entity.
|
||||
* @param tableNode The table node within which the search is performed.
|
||||
* @return A stream of {@link TextEntity} objects found across table cell columns immediately after the specified string, case-insensitively.
|
||||
*/
|
||||
public Stream<TextEntity> lineAfterStringAcrossColumnsIgnoreCase(String string, String type, EntityType entityType, Table tableNode) {
|
||||
|
||||
return tableNode.streamTableCells()
|
||||
.flatMap(tableCell -> lineAfterBoundariesAcrossColumns(RedactionSearchUtility.findTextRangesByStringIgnoreCase(string, tableCell.getTextBlock()),
|
||||
tableCell,
|
||||
type,
|
||||
entityType,
|
||||
tableNode));
|
||||
tableCell,
|
||||
type,
|
||||
entityType,
|
||||
tableNode));
|
||||
}
|
||||
|
||||
|
||||
@@ -626,15 +396,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Attempts to create a text entity for text within a semantic node, immediately after a specified string.
|
||||
*
|
||||
* @param semanticNode The semantic node within which to search for the string.
|
||||
* @param string The string after which the entity should be created.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if the string is not found.
|
||||
*/
|
||||
public Optional<TextEntity> semanticNodeAfterString(SemanticNode semanticNode, String string, String type, EntityType entityType) {
|
||||
|
||||
var textBlock = semanticNode.getTextBlock();
|
||||
@@ -653,77 +414,30 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block,
|
||||
* considering line breaks in the text.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node containing the text block to search.
|
||||
* @return A stream of identified {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexWithLineBreaks(String regexPattern, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return byRegexWithLineBreaks(regexPattern, type, entityType, 0, node);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block, considering line breaks in the text, case-insensitive.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node containing the text block to search.
|
||||
* @return A stream of identified {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return byRegexWithLineBreaksIgnoreCase(regexPattern, type, entityType, 0, node);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node containing the text block to search.
|
||||
* @return A stream of identified {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> byRegex(String regexPattern, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return byRegex(regexPattern, type, entityType, 0, node);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on matches to a regular expression pattern within a semantic node's text block, case-insensitive.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node containing the text block to search.
|
||||
* @return A stream of identified {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return byRegexIgnoreCase(regexPattern, type, entityType, 0, node);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities within a semantic node's text block based on a regex pattern that includes line breaks.
|
||||
*
|
||||
* @param regexPattern Regex pattern to match, including handling for line breaks.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param group The regex group to target for entity creation.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the regex pattern.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexWithLineBreaks(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByRegexWithLineBreaks(regexPattern, group, node.getTextBlock())
|
||||
@@ -734,16 +448,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities within a semantic node's text block based on a regex pattern that includes line breaks, case-insensitive.
|
||||
*
|
||||
* @param regexPattern Regex pattern to match, including handling for line breaks.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param group The regex group to target for entity creation.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the regex pattern.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexWithLineBreaksIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByRegexWithLineBreaksIgnoreCase(regexPattern, group, node.getTextBlock())
|
||||
@@ -754,16 +458,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on a simple regex pattern.
|
||||
*
|
||||
* @param regexPattern Regex pattern to match, including handling for line breaks.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param group The regex group to target for entity creation.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the regex pattern.
|
||||
*/
|
||||
public Stream<TextEntity> byRegex(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByRegex(regexPattern, group, node.getTextBlock())
|
||||
@@ -774,16 +468,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on a simple regex pattern, case-insensitive.
|
||||
*
|
||||
* @param regexPattern Regex pattern to match, including handling for line breaks.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param group The regex group to target for entity creation.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the regex pattern.
|
||||
*/
|
||||
public Stream<TextEntity> byRegexIgnoreCase(String regexPattern, String type, EntityType entityType, int group, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByRegexIgnoreCase(regexPattern, group, node.getTextBlock())
|
||||
@@ -794,15 +478,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on an exact string match within a semantic node's text block.
|
||||
*
|
||||
* @param keyword String keyword to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the exact string.
|
||||
*/
|
||||
public Stream<TextEntity> byString(String keyword, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByString(keyword, node.getTextBlock())
|
||||
@@ -813,15 +488,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies text entities based on an exact string match within a semantic node's text block, case-insensitive.
|
||||
*
|
||||
* @param keyword String keyword to search for.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects that match the exact string, case-insensitive.
|
||||
*/
|
||||
public Stream<TextEntity> byStringIgnoreCase(String keyword, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return RedactionSearchUtility.findTextRangesByStringIgnoreCase(keyword, node.getTextBlock())
|
||||
@@ -832,31 +498,12 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Extracts text entities from paragraphs only, within a given semantic node.
|
||||
*
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of {@link TextEntity} objects extracted from paragraphs only.
|
||||
*/
|
||||
public Stream<TextEntity> bySemanticNodeParagraphsOnly(SemanticNode node, String type, EntityType entityType) {
|
||||
|
||||
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH)
|
||||
.map(semanticNode -> bySemanticNode(semanticNode, type, entityType))
|
||||
.filter(Optional::isPresent)
|
||||
.map(Optional::get);
|
||||
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH).map(semanticNode -> bySemanticNode(semanticNode, type, entityType)).filter(Optional::isPresent).map(Optional::get);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Merges consecutive paragraphs into a single text entity within a given semantic node.
|
||||
*
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node to search within.
|
||||
* @return A stream of merged {@link TextEntity} objects from consecutive paragraphs.
|
||||
*/
|
||||
public Stream<TextEntity> bySemanticNodeParagraphsOnlyMergeConsecutive(SemanticNode node, String type, EntityType entityType) {
|
||||
|
||||
return node.streamAllSubNodesOfType(NodeType.PARAGRAPH)
|
||||
@@ -869,15 +516,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates a text entity immediately following a specified string within a semantic node.
|
||||
*
|
||||
* @param string The string after which to create the entity.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node to search within.
|
||||
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if not found.
|
||||
*/
|
||||
public Optional<TextEntity> semanticNodeAfterString(String string, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
if (!node.containsString(string)) {
|
||||
@@ -888,14 +526,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates a text entity based on the entire text range of a semantic node.
|
||||
*
|
||||
* @param node The semantic node to base the text entity on.
|
||||
* @param type The type of entity to create.
|
||||
* @param entityType The entity's classification.
|
||||
* @return An {@link Optional} containing the created {@link TextEntity}, or {@link Optional#empty()} if not valid.
|
||||
*/
|
||||
public Optional<TextEntity> bySemanticNode(SemanticNode node, String type, EntityType entityType) {
|
||||
|
||||
TextRange textRange = node.getTextBlock().getTextRange();
|
||||
@@ -910,13 +540,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Expands a text entity's start boundary based on a regex pattern match.
|
||||
*
|
||||
* @param entity The original text entity to expand.
|
||||
* @param regexPattern The regex pattern used to find the new start boundary.
|
||||
* @return An {@link Optional} containing the expanded {@link TextEntity}, or {@link Optional#empty()} if not valid.
|
||||
*/
|
||||
public Optional<TextEntity> byPrefixExpansionRegex(TextEntity entity, String regexPattern) {
|
||||
|
||||
int expandedStart = RedactionSearchUtility.getExpandedStartByRegex(entity, regexPattern);
|
||||
@@ -924,13 +547,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Expands a text entity's end boundary based on a regex pattern match.
|
||||
*
|
||||
* @param entity The original text entity to expand.
|
||||
* @param regexPattern The regex pattern used to find the new end boundary.
|
||||
* @return An {@link Optional} containing the expanded {@link TextEntity}, or {@link Optional#empty()} if not valid.
|
||||
*/
|
||||
public Optional<TextEntity> bySuffixExpansionRegex(TextEntity entity, String regexPattern) {
|
||||
|
||||
int expandedEnd = RedactionSearchUtility.getExpandedEndByRegex(entity, regexPattern);
|
||||
@@ -974,16 +590,9 @@ public class EntityCreationService {
|
||||
throw new IllegalArgumentException(String.format("%s is not in the %s of the provided semantic node %s", textRange, node.getTextRange(), node));
|
||||
}
|
||||
TextRange trimmedTextRange = textRange.trim(node.getTextBlock());
|
||||
if (trimmedTextRange.length() == 0) {
|
||||
return Optional.empty();
|
||||
}
|
||||
TextEntity entity = TextEntity.initialEntityNode(trimmedTextRange, type, entityType, node);
|
||||
if (node.getEntities().contains(entity)) {
|
||||
Optional<TextEntity> optionalTextEntity = node.getEntities()
|
||||
.stream()
|
||||
.filter(e -> e.equals(entity) && e.type().equals(type))
|
||||
.peek(e -> e.addEngines(engines))
|
||||
.findAny();
|
||||
Optional<TextEntity> optionalTextEntity = node.getEntities().stream().filter(e -> e.equals(entity) && e.type().equals(type)).peek(e -> e.addEngines(engines)).findAny();
|
||||
if (optionalTextEntity.isEmpty()) {
|
||||
return optionalTextEntity; // Entity has been recategorized and should not be created at all.
|
||||
}
|
||||
@@ -1026,16 +635,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Merges a list of text entities into a single entity, assuming they intersect and are of the same type.
|
||||
*
|
||||
* @param entitiesToMerge The list of entities to merge.
|
||||
* @param type The type for the merged entity.
|
||||
* @param entityType The entity's classification.
|
||||
* @param node The semantic node related to these entities.
|
||||
* @return A single merged {@link TextEntity}.
|
||||
* @throws IllegalArgumentException If entities do not intersect or have different types.
|
||||
*/
|
||||
public TextEntity mergeEntitiesOfSameType(List<TextEntity> entitiesToMerge, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
if (!allEntitiesIntersectAndHaveSameTypes(entitiesToMerge)) {
|
||||
@@ -1048,63 +647,30 @@ public class EntityCreationService {
|
||||
return entitiesToMerge.get(0);
|
||||
}
|
||||
|
||||
TextEntity mergedEntity = TextEntity.initialEntityNode(TextRange.merge(entitiesToMerge.stream()
|
||||
.map(TextEntity::getTextRange)
|
||||
.toList()), type, entityType, node);
|
||||
mergedEntity.addEngines(entitiesToMerge.stream()
|
||||
.flatMap(entityNode -> entityNode.getEngines()
|
||||
.stream())
|
||||
.collect(Collectors.toSet()));
|
||||
entitiesToMerge.stream()
|
||||
.map(TextEntity::getMatchedRuleList)
|
||||
.flatMap(Collection::stream)
|
||||
.forEach(matchedRule -> mergedEntity.getMatchedRuleList().add(matchedRule));
|
||||
TextEntity mergedEntity = TextEntity.initialEntityNode(TextRange.merge(entitiesToMerge.stream().map(TextEntity::getTextRange).toList()), type, entityType, node);
|
||||
mergedEntity.addEngines(entitiesToMerge.stream().flatMap(entityNode -> entityNode.getEngines().stream()).collect(Collectors.toSet()));
|
||||
entitiesToMerge.stream().map(TextEntity::getMatchedRuleList).flatMap(Collection::stream).forEach(matchedRule -> mergedEntity.getMatchedRuleList().add(matchedRule));
|
||||
entitiesToMerge.stream()
|
||||
.map(TextEntity::getManualOverwrite)
|
||||
.map(ManualChangeOverwrite::getManualChangeLog)
|
||||
.flatMap(Collection::stream)
|
||||
.forEach(manualChange -> mergedEntity.getManualOverwrite().addChange(manualChange));
|
||||
|
||||
mergedEntity.setDictionaryEntry(entitiesToMerge.stream()
|
||||
.anyMatch(TextEntity::isDictionaryEntry));
|
||||
mergedEntity.setDossierDictionaryEntry(entitiesToMerge.stream()
|
||||
.anyMatch(TextEntity::isDossierDictionaryEntry));
|
||||
mergedEntity.setDictionaryEntry(entitiesToMerge.stream().anyMatch(TextEntity::isDictionaryEntry));
|
||||
mergedEntity.setDossierDictionaryEntry(entitiesToMerge.stream().anyMatch(TextEntity::isDossierDictionaryEntry));
|
||||
|
||||
addEntityToGraph(mergedEntity, node);
|
||||
insertToKieSession(mergedEntity);
|
||||
|
||||
entitiesToMerge.stream()
|
||||
.filter(e -> !e.equals(mergedEntity))
|
||||
.forEach(node.getEntities()::remove);
|
||||
return mergedEntity;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Copies a list of text entities, creating a new entity for each in the list with the same properties.
|
||||
*
|
||||
* @param entities The list of entities to copy.
|
||||
* @param type The type for the copied entities.
|
||||
* @param entityType The classification for the copied entities.
|
||||
* @param node The semantic node related to these entities.
|
||||
* @return A stream of copied {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> copyEntities(List<TextEntity> entities, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
return entities.stream()
|
||||
.map(entity -> copyEntity(entity, type, entityType, node));
|
||||
return entities.stream().map(entity -> copyEntity(entity, type, entityType, node));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Copies a single text entity, preserving all its matched rules.
|
||||
*
|
||||
* @param entity The entity to copy.
|
||||
* @param type The type for the copied entity.
|
||||
* @param entityType The classification for the copied entity.
|
||||
* @param node The semantic node related to the entity.
|
||||
* @return A copied {@link TextEntity} with matched rules.
|
||||
*/
|
||||
public TextEntity copyEntity(TextEntity entity, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
var newEntity = copyEntityWithoutRules(entity, type, entityType, node);
|
||||
@@ -1113,15 +679,6 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Copies a single text entity without its matched rules.
|
||||
*
|
||||
* @param entity The entity to copy.
|
||||
* @param type The type for the copied entity.
|
||||
* @param entityType The classification for the copied entity.
|
||||
* @param node The semantic node related to the entity.
|
||||
* @return A copied {@link TextEntity} without matched rules.
|
||||
*/
|
||||
public TextEntity copyEntityWithoutRules(TextEntity entity, String type, EntityType entityType, SemanticNode node) {
|
||||
|
||||
TextEntity newEntity = byTextRangeWithEngine(entity.getTextRange(), type, entityType, node, entity.getEngines()).orElseThrow(() -> new NotFoundException(
|
||||
@@ -1133,27 +690,14 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Inserts a text entity into the kieSession for further processing.
|
||||
*
|
||||
* @param textEntity The merged text entity to insert.
|
||||
*/
|
||||
public void insertToKieSession(TextEntity textEntity) {
|
||||
public void insertToKieSession(TextEntity mergedEntity) {
|
||||
|
||||
if (kieSession != null) {
|
||||
kieSession.insert(textEntity);
|
||||
kieSession.insert(mergedEntity);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates a text entity based on a Named Entity Recognition (NER) entity.
|
||||
*
|
||||
* @param nerEntity The NER entity used for creating the text entity.
|
||||
* @param entityType The entity's classification.
|
||||
* @param semanticNode The semantic node related to the NER entity.
|
||||
* @return A new {@link TextEntity} based on the NER entity.
|
||||
*/
|
||||
public TextEntity byNerEntity(NerEntities.NerEntity nerEntity, EntityType entityType, SemanticNode semanticNode) {
|
||||
|
||||
return byTextRangeWithEngine(nerEntity.textRange(), nerEntity.type(), entityType, semanticNode, Set.of(Engine.NER)).orElseThrow(() -> new NotFoundException(
|
||||
@@ -1161,59 +705,24 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Creates a text entity based on a Named Entity Recognition (NER) entity, with a specified type.
|
||||
*
|
||||
* @param nerEntity The NER entity used for creating the text entity.
|
||||
* @param type Type of the entity.
|
||||
* @param entityType The entity's classification.
|
||||
* @param semanticNode The semantic node related to the NER entity.
|
||||
* @return A new {@link TextEntity} based on the NER entity.
|
||||
*/
|
||||
public TextEntity byNerEntity(NerEntities.NerEntity nerEntity, String type, EntityType entityType, SemanticNode semanticNode) {
|
||||
|
||||
return byTextRangeWithEngine(nerEntity.textRange(), type, entityType, semanticNode, Set.of(Engine.NER)).orElseThrow(() -> new NotFoundException("No entity present!"));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Optionally creates a text entity based on a Named Entity Recognition (NER) entity.
|
||||
*
|
||||
* @param nerEntity The NER entity used for creating the text entity.
|
||||
* @param entityType The entity's classification.
|
||||
* @param semanticNode The semantic node related to the NER entity.
|
||||
* @return An {@link Optional} containing the new {@link TextEntity} based on the NER entity, or {@link Optional#empty()} if not created.
|
||||
*/
|
||||
public Optional<TextEntity> optionalByNerEntity(NerEntities.NerEntity nerEntity, EntityType entityType, SemanticNode semanticNode) {
|
||||
|
||||
return byTextRangeWithEngine(nerEntity.textRange(), nerEntity.type(), entityType, semanticNode, Set.of(Engine.NER));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Optionally creates a text entity based on a Named Entity Recognition (NER) entity, with a specified type.
|
||||
*
|
||||
* @param nerEntity The NER entity used for creating the text entity.
|
||||
* @param type Type of the entity.
|
||||
* @param entityType The entity's classification.
|
||||
* @param semanticNode The semantic node related to the NER entity.
|
||||
* @return An {@link Optional} containing the new {@link TextEntity} based on the NER entity, or {@link Optional#empty()} if not created.
|
||||
*/
|
||||
public Optional<TextEntity> optionalByNerEntity(NerEntities.NerEntity nerEntity, String type, EntityType entityType, SemanticNode semanticNode) {
|
||||
|
||||
return byTextRangeWithEngine(nerEntity.textRange(), type, entityType, semanticNode, Set.of(Engine.NER));
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Combines multiple NER entities into a single text entity.
|
||||
*
|
||||
* @param nerEntities The collection of NER entities to combine.
|
||||
* @param type The type for the combined entity.
|
||||
* @param entityType The classification for the combined entity.
|
||||
* @param semanticNode The semantic node related to these entities.
|
||||
* @return A stream of combined {@link TextEntity} objects.
|
||||
*/
|
||||
public Stream<TextEntity> combineNerEntitiesToCbiAddressDefaults(NerEntities nerEntities, String type, EntityType entityType, SemanticNode semanticNode) {
|
||||
|
||||
return NerEntitiesAdapter.combineNerEntitiesToCbiAddressDefaults(nerEntities)
|
||||
@@ -1223,63 +732,37 @@ public class EntityCreationService {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Validates if a given text range within a text block represents a valid entity.
|
||||
*
|
||||
* @param textBlock The text block containing the text range.
|
||||
* @param textRange The text range to validate.
|
||||
* @return true if the text range represents a valid entity, false otherwise.
|
||||
*/
|
||||
public boolean isValidEntityTextRange(TextBlock textBlock, TextRange textRange) {
|
||||
|
||||
return textRange.length() > 0 && boundaryIsSurroundedBySeparators(textBlock, textRange);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Adds a text entity to its related semantic node and updates the document tree accordingly.
|
||||
*
|
||||
* @param entity The text entity to add.
|
||||
* @param node The semantic node related to the entity.
|
||||
*/
|
||||
public void addEntityToGraph(TextEntity entity, SemanticNode node) {
|
||||
|
||||
DocumentTree documentTree = node.getDocumentTree();
|
||||
try {
|
||||
if (node.getEntities().contains(entity)) {
|
||||
// If entity already exists and it has a different text range, we add the text range to the list of duplicated text ranges
|
||||
Optional<TextEntity> optionalTextEntity = node.getEntities()
|
||||
.stream()//
|
||||
node.getEntities().stream()//
|
||||
.filter(e -> e.equals(entity))//
|
||||
.filter(e -> !e.getTextRange().equals(entity.getTextRange()))//
|
||||
.findAny();
|
||||
if (optionalTextEntity.isPresent()) {
|
||||
addDuplicateEntityToGraph(optionalTextEntity.get(), entity.getTextRange(), node);
|
||||
} else {
|
||||
node.getEntities().remove(entity);
|
||||
addNewEntityToGraph(entity, documentTree);
|
||||
}
|
||||
|
||||
.findAny()//
|
||||
.ifPresent(entityToDuplicate -> addDuplicateEntityToGraph(entityToDuplicate, entity.getTextRange(), node));
|
||||
} else {
|
||||
entity.addIntersectingNode(documentTree.getRoot().getNode());
|
||||
addEntityToGraph(entity, documentTree);
|
||||
}
|
||||
} catch (NoSuchElementException e) {
|
||||
addNewEntityToGraph(entity, documentTree);
|
||||
entity.setDeepestFullyContainingNode(documentTree.getRoot().getNode());
|
||||
entityEnrichmentService.enrichEntity(entity, entity.getDeepestFullyContainingNode().getTextBlock());
|
||||
entity.addIntersectingNode(documentTree.getRoot().getNode());
|
||||
addToPages(entity);
|
||||
addEntityToNodeEntitySets(entity);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addNewEntityToGraph(TextEntity entity, DocumentTree documentTree) {
|
||||
|
||||
entity.setDeepestFullyContainingNode(documentTree.getRoot().getNode());
|
||||
entityEnrichmentService.enrichEntity(entity, entity.getDeepestFullyContainingNode().getTextBlock());
|
||||
entity.addIntersectingNode(documentTree.getRoot().getNode());
|
||||
addToPages(entity);
|
||||
addEntityToNodeEntitySets(entity);
|
||||
}
|
||||
|
||||
|
||||
private void addDuplicateEntityToGraph(TextEntity entityToDuplicate, TextRange newTextRange, SemanticNode node) {
|
||||
|
||||
entityToDuplicate.addTextRange(newTextRange);
|
||||
@@ -1287,10 +770,8 @@ public class EntityCreationService {
|
||||
SemanticNode deepestSharedNode = entityToDuplicate.getIntersectingNodes()
|
||||
.stream()
|
||||
.sorted(Comparator.comparingInt(n -> -n.getTreeId().size()))
|
||||
.filter(intersectingNode -> entityToDuplicate.getDuplicateTextRanges()
|
||||
.stream()
|
||||
.allMatch(tr -> intersectingNode.getTextRange().contains(tr)) && //
|
||||
intersectingNode.getTextRange().contains(entityToDuplicate.getTextRange()))
|
||||
.filter(intersectingNode -> entityToDuplicate.getDuplicateTextRanges().stream().allMatch(tr -> intersectingNode.getTextRange().contains(tr)) && //
|
||||
intersectingNode.getTextRange().contains(entityToDuplicate.getTextRange()))
|
||||
.findFirst()
|
||||
.orElse(node.getDocumentTree().getRoot().getNode());
|
||||
|
||||
@@ -1303,8 +784,7 @@ public class EntityCreationService {
|
||||
return;
|
||||
}
|
||||
additionalIntersectingNode.getEntities().add(entityToDuplicate);
|
||||
additionalIntersectingNode.getPages(newTextRange)
|
||||
.forEach(page -> page.getEntities().add(entityToDuplicate));
|
||||
additionalIntersectingNode.getPages(newTextRange).forEach(page -> page.getEntities().add(entityToDuplicate));
|
||||
entityToDuplicate.addIntersectingNode(additionalIntersectingNode);
|
||||
});
|
||||
}
|
||||
@@ -1326,4 +806,5 @@ public class EntityCreationService {
|
||||
addEntityToNodeEntitySets(entity);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
+2
-2
@@ -11,6 +11,7 @@ import com.iqser.red.service.redaction.v1.server.model.document.textblock.TextBl
|
||||
|
||||
public class EntityCreationUtility {
|
||||
|
||||
|
||||
public static void checkIfBothStartAndEndAreEmpty(String start, String end) {
|
||||
|
||||
checkIfBothStartAndEndAreEmpty(List.of(start), List.of(end));
|
||||
@@ -56,8 +57,7 @@ public class EntityCreationUtility {
|
||||
|
||||
public static void addEntityToNodeEntitySets(TextEntity entity) {
|
||||
|
||||
entity.getIntersectingNodes()
|
||||
.forEach(node -> node.getEntities().add(entity));
|
||||
entity.getIntersectingNodes().forEach(node -> node.getEntities().add(entity));
|
||||
}
|
||||
|
||||
|
||||
|
||||
+1
-3
@@ -59,9 +59,7 @@ public class EntityEnrichmentService {
|
||||
|
||||
private static List<String> splitToWordsAndRemoveEmptyWords(String textAfter) {
|
||||
|
||||
return Arrays.stream(textAfter.split(" "))
|
||||
.filter(word -> !Objects.equals("", word))
|
||||
.toList();
|
||||
return Arrays.stream(textAfter.split(" ")).filter(word -> !Objects.equals("", word)).toList();
|
||||
}
|
||||
|
||||
|
||||
|
||||
+24
-49
@@ -47,9 +47,7 @@ public class EntityFindingUtility {
|
||||
}
|
||||
|
||||
|
||||
public Optional<TextEntity> findClosestEntityAndReturnEmptyIfNotFound(PrecursorEntity precursorEntity,
|
||||
Map<String, List<TextEntity>> entitiesWithSameValue,
|
||||
double matchThreshold) {
|
||||
public Optional<TextEntity> findClosestEntityAndReturnEmptyIfNotFound(PrecursorEntity precursorEntity, Map<String, List<TextEntity>> entitiesWithSameValue, double matchThreshold) {
|
||||
|
||||
if (precursorEntity.getValue() == null) {
|
||||
return Optional.empty();
|
||||
@@ -58,7 +56,7 @@ public class EntityFindingUtility {
|
||||
List<TextEntity> possibleEntities = entitiesWithSameValue.get(precursorEntity.getValue().toLowerCase(Locale.ENGLISH));
|
||||
|
||||
if (entityIdentifierValueNotFound(possibleEntities)) {
|
||||
log.info("Entity could not be created with precursorEntity: {}, due to the value {} not being found anywhere.", precursorEntity, precursorEntity.getValue());
|
||||
log.warn("Entity could not be created with precursorEntity: {}, due to the value {} not being found anywhere.", precursorEntity, precursorEntity.getValue());
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
@@ -68,22 +66,18 @@ public class EntityFindingUtility {
|
||||
.min(Comparator.comparingDouble(ClosestEntity::getDistance));
|
||||
|
||||
if (optionalClosestEntity.isEmpty()) {
|
||||
log.info("No Entity with value {} found on page {}", precursorEntity.getValue(), precursorEntity.getEntityPosition());
|
||||
log.warn("No Entity with value {} found on page {}", precursorEntity.getValue(), precursorEntity.getEntityPosition());
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
ClosestEntity closestEntity = optionalClosestEntity.get();
|
||||
if (closestEntity.getDistance() > matchThreshold) {
|
||||
log.info("For entity {} on page {} with positions {} distance to closest found entity is {} and therefore higher than the threshold of {}",
|
||||
precursorEntity.getValue(),
|
||||
precursorEntity.getEntityPosition()
|
||||
.get(0).pageNumber(),
|
||||
precursorEntity.getEntityPosition()
|
||||
.stream()
|
||||
.map(RectangleWithPage::rectangle2D)
|
||||
.toList(),
|
||||
closestEntity.getDistance(),
|
||||
matchThreshold);
|
||||
log.warn("For entity {} on page {} with positions {} distance to closest found entity is {} and therefore higher than the threshold of {}",
|
||||
precursorEntity.getValue(),
|
||||
precursorEntity.getEntityPosition().get(0).pageNumber(),
|
||||
precursorEntity.getEntityPosition().stream().map(RectangleWithPage::rectangle2D).toList(),
|
||||
closestEntity.getDistance(),
|
||||
matchThreshold);
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
@@ -99,14 +93,8 @@ public class EntityFindingUtility {
|
||||
|
||||
private static boolean pagesMatch(TextEntity entity, List<RectangleWithPage> originalPositions) {
|
||||
|
||||
Set<Integer> entityPageNumbers = entity.getPositionsOnPagePerPage()
|
||||
.stream()
|
||||
.map(PositionOnPage::getPage)
|
||||
.map(Page::getNumber)
|
||||
.collect(Collectors.toSet());
|
||||
Set<Integer> originalPageNumbers = originalPositions.stream()
|
||||
.map(RectangleWithPage::pageNumber)
|
||||
.collect(Collectors.toSet());
|
||||
Set<Integer> entityPageNumbers = entity.getPositionsOnPagePerPage().stream().map(PositionOnPage::getPage).map(Page::getNumber).collect(Collectors.toSet());
|
||||
Set<Integer> originalPageNumbers = originalPositions.stream().map(RectangleWithPage::pageNumber).collect(Collectors.toSet());
|
||||
return entityPageNumbers.containsAll(originalPageNumbers);
|
||||
}
|
||||
|
||||
@@ -117,16 +105,15 @@ public class EntityFindingUtility {
|
||||
return Double.MAX_VALUE;
|
||||
}
|
||||
return originalPositions.stream()
|
||||
.mapToDouble(rectangleWithPage -> calculateMinDistancePerRectangle(entity, rectangleWithPage.pageNumber(), rectangleWithPage.rectangle2D())).average()
|
||||
.mapToDouble(rectangleWithPage -> calculateMinDistancePerRectangle(entity, rectangleWithPage.pageNumber(), rectangleWithPage.rectangle2D()))
|
||||
.average()
|
||||
.orElse(Double.MAX_VALUE);
|
||||
}
|
||||
|
||||
|
||||
private static long countRectangles(TextEntity entity) {
|
||||
|
||||
return entity.getPositionsOnPagePerPage()
|
||||
.stream()
|
||||
.mapToLong(redactionPosition -> redactionPosition.getRectanglePerLine().size()).sum();
|
||||
return entity.getPositionsOnPagePerPage().stream().mapToLong(redactionPosition -> redactionPosition.getRectanglePerLine().size()).sum();
|
||||
}
|
||||
|
||||
|
||||
@@ -157,36 +144,24 @@ public class EntityFindingUtility {
|
||||
double maxY2 = Math.max(rectangle2.getMinY(), rectangle2.getMaxY());
|
||||
|
||||
return Math.abs(minX1 - minX2) //
|
||||
+ Math.abs(minY1 - minY2) //
|
||||
+ Math.abs(maxX1 - maxX2) //
|
||||
+ Math.abs(maxY1 - maxY2);
|
||||
+ Math.abs(minY1 - minY2) //
|
||||
+ Math.abs(maxX1 - maxX2) //
|
||||
+ Math.abs(maxY1 - maxY2);
|
||||
}
|
||||
|
||||
|
||||
public Map<String, List<TextEntity>> findAllPossibleEntitiesAndGroupByValue(SemanticNode node, List<PrecursorEntity> manualEntities) {
|
||||
|
||||
Set<Integer> pageNumbers = manualEntities.stream()
|
||||
.flatMap(entry -> entry.getEntityPosition()
|
||||
.stream()
|
||||
.map(RectangleWithPage::pageNumber))
|
||||
.collect(Collectors.toSet());
|
||||
Set<String> entryValues = manualEntities.stream()
|
||||
.map(PrecursorEntity::getValue)
|
||||
.filter(Objects::nonNull)
|
||||
.map(String::toLowerCase)
|
||||
.collect(Collectors.toSet());
|
||||
Set<Integer> pageNumbers = manualEntities.stream().flatMap(entry -> entry.getEntityPosition().stream().map(RectangleWithPage::pageNumber)).collect(Collectors.toSet());
|
||||
Set<String> entryValues = manualEntities.stream().map(PrecursorEntity::getValue).filter(Objects::nonNull).map(String::toLowerCase).collect(Collectors.toSet());
|
||||
|
||||
if (!pageNumbers.stream()
|
||||
.allMatch(node::onPage)) {
|
||||
if (!pageNumbers.stream().allMatch(node::onPage)) {
|
||||
throw new IllegalArgumentException(format("SemanticNode \"%s\" does not contain these pages %s, it has pages: %s",
|
||||
node,
|
||||
pageNumbers.stream()
|
||||
.filter(pageNumber -> !node.onPage(pageNumber))
|
||||
.toList(),
|
||||
node.getPages()));
|
||||
node,
|
||||
pageNumbers.stream().filter(pageNumber -> !node.onPage(pageNumber)).toList(),
|
||||
node.getPages()));
|
||||
}
|
||||
|
||||
SearchImplementation searchImplementation = new SearchImplementation(entryValues.stream().map(String::trim).collect(Collectors.toSet()), true);
|
||||
SearchImplementation searchImplementation = new SearchImplementation(entryValues, true);
|
||||
|
||||
return searchImplementation.getBoundaries(node.getTextBlock(), node.getTextRange())
|
||||
.stream()
|
||||
|
||||
+11
-2
@@ -9,6 +9,7 @@ import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
|
||||
@@ -22,21 +23,29 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNo
|
||||
import com.iqser.red.service.redaction.v1.server.service.DictionaryService;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class EntityFromPrecursorCreationService {
|
||||
|
||||
static double MATCH_THRESHOLD = 10; // Is compared to the average sum of distances in pdf coordinates for each corner of the bounding box of the entities
|
||||
EntityFindingUtility entityFindingUtility;
|
||||
EntityCreationService entityCreationService;
|
||||
DictionaryService dictionaryService;
|
||||
|
||||
|
||||
@Autowired
|
||||
public EntityFromPrecursorCreationService(EntityEnrichmentService entityEnrichmentService, DictionaryService dictionaryService, EntityFindingUtility entityFindingUtility) {
|
||||
|
||||
this.entityFindingUtility = entityFindingUtility;
|
||||
entityCreationService = new EntityCreationService(entityEnrichmentService);
|
||||
this.dictionaryService = dictionaryService;
|
||||
}
|
||||
|
||||
|
||||
public List<PrecursorEntity> createEntitiesIfFoundAndReturnNotFoundEntries(ManualRedactions manualRedactions, SemanticNode node, String dossierTemplateId) {
|
||||
|
||||
Set<IdRemoval> idRemovals = manualRedactions.getIdsToRemove();
|
||||
|
||||
+4
-8
@@ -52,13 +52,9 @@ public class ImportedRedactionEntryService {
|
||||
private List<BaseAnnotation> allManualChangesExceptAdd(ManualRedactions manualRedactions) {
|
||||
|
||||
return Stream.of(manualRedactions.getForceRedactions(),
|
||||
manualRedactions.getResizeRedactions(),
|
||||
manualRedactions.getRecategorizations(),
|
||||
manualRedactions.getIdsToRemove(),
|
||||
manualRedactions.getLegalBasisChanges())
|
||||
.flatMap(Collection::stream)
|
||||
.map(baseAnnotation -> (BaseAnnotation) baseAnnotation)
|
||||
.toList();
|
||||
manualRedactions.getResizeRedactions(),
|
||||
manualRedactions.getRecategorizations(),
|
||||
manualRedactions.getIdsToRemove(),
|
||||
manualRedactions.getLegalBasisChanges()).flatMap(Collection::stream).map(baseAnnotation -> (BaseAnnotation) baseAnnotation).toList();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-2
@@ -14,14 +14,12 @@ public class IntersectingNodeVisitor implements NodeVisitor {
|
||||
private Set<SemanticNode> intersectingNodes;
|
||||
private final TextRange textRange;
|
||||
|
||||
|
||||
public IntersectingNodeVisitor(TextRange textRange) {
|
||||
|
||||
this.textRange = textRange;
|
||||
this.intersectingNodes = new HashSet<>();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public void visit(SemanticNode node) {
|
||||
|
||||
|
||||
+5
-9
@@ -31,8 +31,7 @@ public class ManualRedactionEntryService {
|
||||
List<PrecursorEntity> notFoundManualRedactionEntries = Collections.emptyList();
|
||||
if (analyzeRequest.getManualRedactions() != null) {
|
||||
notFoundManualRedactionEntries = entityFromPrecursorCreationService.createEntitiesIfFoundAndReturnNotFoundEntries(analyzeRequest.getManualRedactions(),
|
||||
document,
|
||||
dossierTemplateId);
|
||||
document, dossierTemplateId);
|
||||
log.info("Added Manual redaction entries for file {} in dossier {}", analyzeRequest.getFileId(), analyzeRequest.getDossierId());
|
||||
}
|
||||
|
||||
@@ -52,13 +51,10 @@ public class ManualRedactionEntryService {
|
||||
private List<BaseAnnotation> allManualChangesExceptAdd(ManualRedactions manualRedactions) {
|
||||
|
||||
return Stream.of(manualRedactions.getForceRedactions(),
|
||||
manualRedactions.getResizeRedactions(),
|
||||
manualRedactions.getRecategorizations(),
|
||||
manualRedactions.getIdsToRemove(),
|
||||
manualRedactions.getLegalBasisChanges())
|
||||
.flatMap(Collection::stream)
|
||||
.map(baseAnnotation -> (BaseAnnotation) baseAnnotation)
|
||||
.toList();
|
||||
manualRedactions.getResizeRedactions(),
|
||||
manualRedactions.getRecategorizations(),
|
||||
manualRedactions.getIdsToRemove(),
|
||||
manualRedactions.getLegalBasisChanges()).flatMap(Collection::stream).map(baseAnnotation -> (BaseAnnotation) baseAnnotation).toList();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+19
-37
@@ -44,11 +44,9 @@ public class NerEntitiesAdapter {
|
||||
public NerEntities toNerEntities(NerEntitiesModel nerEntitiesModel, Document document) {
|
||||
|
||||
return new NerEntities(addOffsetsAndFlatten(getStringStartOffsetsForMainSections(document),
|
||||
nerEntitiesModel).map(nerEntityModel -> new NerEntities.NerEntity(nerEntityModel.getValue(),
|
||||
new TextRange(nerEntityModel.getStartOffset(),
|
||||
nerEntityModel.getEndOffset()),
|
||||
nerEntityModel.getType()))
|
||||
.toList());
|
||||
nerEntitiesModel).map(nerEntityModel -> new NerEntities.NerEntity(nerEntityModel.getValue(),
|
||||
new TextRange(nerEntityModel.getStartOffset(), nerEntityModel.getEndOffset()),
|
||||
nerEntityModel.getType())).toList());
|
||||
}
|
||||
|
||||
|
||||
@@ -85,9 +83,7 @@ public class NerEntitiesAdapter {
|
||||
|
||||
List<List<NerEntities.NerEntity>> entityClusters = new LinkedList<>();
|
||||
|
||||
List<NerEntities.NerEntity> startEntitiesOfEssentialType = sortedEntities.stream()
|
||||
.filter(e -> essentialTypes.contains(e.type()))
|
||||
.toList();
|
||||
List<NerEntities.NerEntity> startEntitiesOfEssentialType = sortedEntities.stream().filter(e -> essentialTypes.contains(e.type())).toList();
|
||||
for (NerEntities.NerEntity startEntity : startEntitiesOfEssentialType) {
|
||||
List<NerEntities.NerEntity> currentCluster = new LinkedList<>();
|
||||
entityClusters.add(currentCluster);
|
||||
@@ -109,10 +105,7 @@ public class NerEntitiesAdapter {
|
||||
}
|
||||
}
|
||||
|
||||
return entityClusters.stream()
|
||||
.filter(cluster -> cluster.size() >= minPartsToCombine)
|
||||
.map(NerEntitiesAdapter::toContainingBoundary)
|
||||
.distinct();
|
||||
return entityClusters.stream().filter(cluster -> cluster.size() >= minPartsToCombine).map(NerEntitiesAdapter::toContainingBoundary).distinct();
|
||||
}
|
||||
|
||||
|
||||
@@ -131,18 +124,17 @@ public class NerEntitiesAdapter {
|
||||
public Stream<TextRange> combineNerEntitiesToCbiAddressDefaults(NerEntities entityRecognitionEntities) {
|
||||
|
||||
return combineNerEntities(entityRecognitionEntities,
|
||||
CBI_ADDRESS_ESSENTIAL_TYPES,
|
||||
CBI_ADDRESS_TYPES_TO_COMBINE,
|
||||
MAX_DISTANCE_BETWEEN_PARTS,
|
||||
MIN_PARTS_TO_COMBINE,
|
||||
ALLOW_DUPLICATES);
|
||||
CBI_ADDRESS_ESSENTIAL_TYPES,
|
||||
CBI_ADDRESS_TYPES_TO_COMBINE,
|
||||
MAX_DISTANCE_BETWEEN_PARTS,
|
||||
MIN_PARTS_TO_COMBINE,
|
||||
ALLOW_DUPLICATES);
|
||||
}
|
||||
|
||||
|
||||
private static boolean isDuplicate(List<NerEntities.NerEntity> currentCluster, NerEntities.NerEntity entity, boolean allowDuplicates) {
|
||||
|
||||
return allowDuplicates || currentCluster.stream()
|
||||
.anyMatch(e -> e.type().equals(entity.type()));
|
||||
return allowDuplicates || currentCluster.stream().anyMatch(e -> e.type().equals(entity.type()));
|
||||
}
|
||||
|
||||
|
||||
@@ -154,34 +146,24 @@ public class NerEntitiesAdapter {
|
||||
|
||||
private static TextRange toContainingBoundary(List<NerEntities.NerEntity> nerEntities) {
|
||||
|
||||
return TextRange.merge(nerEntities.stream()
|
||||
.map(NerEntities.NerEntity::textRange)
|
||||
.toList());
|
||||
return TextRange.merge(nerEntities.stream().map(NerEntities.NerEntity::textRange).toList());
|
||||
}
|
||||
|
||||
|
||||
private static Stream<EntityRecognitionEntity> addOffsetsAndFlatten(List<Integer> stringOffsetsForMainSections, NerEntitiesModel nerEntitiesModel) {
|
||||
|
||||
nerEntitiesModel.getData()
|
||||
.forEach((sectionNumber, listOfNerEntities) -> listOfNerEntities.forEach(entityRecognitionEntity -> {
|
||||
int newStartOffset = entityRecognitionEntity.getStartOffset() + stringOffsetsForMainSections.get(sectionNumber);
|
||||
entityRecognitionEntity.setStartOffset(newStartOffset);
|
||||
entityRecognitionEntity.setEndOffset(newStartOffset + entityRecognitionEntity.getValue().length());
|
||||
}));
|
||||
return nerEntitiesModel.getData().values()
|
||||
.stream()
|
||||
.flatMap(Collection::stream);
|
||||
nerEntitiesModel.getData().forEach((sectionNumber, listOfNerEntities) -> listOfNerEntities.forEach(entityRecognitionEntity -> {
|
||||
int newStartOffset = entityRecognitionEntity.getStartOffset() + stringOffsetsForMainSections.get(sectionNumber);
|
||||
entityRecognitionEntity.setStartOffset(newStartOffset);
|
||||
entityRecognitionEntity.setEndOffset(newStartOffset + entityRecognitionEntity.getValue().length());
|
||||
}));
|
||||
return nerEntitiesModel.getData().values().stream().flatMap(Collection::stream);
|
||||
}
|
||||
|
||||
|
||||
private static List<Integer> getStringStartOffsetsForMainSections(Document document) {
|
||||
|
||||
return document.getMainSections()
|
||||
.stream()
|
||||
.map(Section::getTextBlock)
|
||||
.map(TextBlock::getTextRange)
|
||||
.map(TextRange::start)
|
||||
.toList();
|
||||
return document.getMainSections().stream().map(Section::getTextBlock).map(TextBlock::getTextRange).map(TextRange::start).toList();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-1
@@ -5,5 +5,4 @@ import com.iqser.red.service.redaction.v1.server.model.document.nodes.SemanticNo
|
||||
public interface NodeVisitor {
|
||||
|
||||
void visit(SemanticNode node);
|
||||
|
||||
}
|
||||
|
||||
+1
-22
@@ -43,29 +43,8 @@ public class PropertiesMapper {
|
||||
|
||||
private Rectangle2D parseRectangle2D(String bBox) {
|
||||
|
||||
List<Float> floats = Arrays.stream(bBox.split(DocumentStructure.RECTANGLE_DELIMITER))
|
||||
.map(Float::parseFloat)
|
||||
.toList();
|
||||
List<Float> floats = Arrays.stream(bBox.split(DocumentStructure.RECTANGLE_DELIMITER)).map(Float::parseFloat).toList();
|
||||
return new Rectangle2D.Float(floats.get(0), floats.get(1), floats.get(2), floats.get(3));
|
||||
}
|
||||
|
||||
|
||||
public static boolean isDuplicateParagraph(Map<String, String> properties) {
|
||||
|
||||
return properties.containsKey(DocumentStructure.DuplicateParagraphProperties.UNSORTED_TEXTBLOCK_ID);
|
||||
}
|
||||
|
||||
|
||||
public static Long[] getUnsortedTextblockIds(Map<String, String> properties) {
|
||||
|
||||
return toLongArray(properties.get(DocumentStructure.DuplicateParagraphProperties.UNSORTED_TEXTBLOCK_ID));
|
||||
}
|
||||
|
||||
|
||||
public static Long[] toLongArray(String ids) {
|
||||
|
||||
return Arrays.stream(ids.substring(1, ids.length() - 1).trim().split(",")).map(Long::valueOf).toArray(Long[]::new);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
+14
-4
@@ -11,6 +11,7 @@ import java.util.stream.Stream;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.AnalyzeRequest;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.Position;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedaction;
|
||||
@@ -43,14 +44,23 @@ public class SectionFinderService {
|
||||
|
||||
@Timed("redactmanager_findSectionsToReanalyse")
|
||||
public Set<Integer> findSectionsToReanalyse(DictionaryIncrement dictionaryIncrement,
|
||||
EntityLog entityLog,
|
||||
Document document,
|
||||
AnalyzeRequest analyzeRequest,
|
||||
ImportedRedactions importedRedactions,
|
||||
Set<String> relevantManuallyModifiedAnnotationIds) {
|
||||
ImportedRedactions importedRedactions) {
|
||||
|
||||
long start = System.currentTimeMillis();
|
||||
Set<String> relevantManuallyModifiedAnnotationIds = getRelevantManuallyModifiedAnnotationIds(analyzeRequest.getManualRedactions());
|
||||
Set<Integer> sectionsToReanalyse = new HashSet<>();
|
||||
|
||||
for (EntityLogEntry entry : entityLog.getEntityLogEntry()) {
|
||||
if (relevantManuallyModifiedAnnotationIds.contains(entry.getId())) {
|
||||
if (entry.getContainingNodeId().isEmpty()) {
|
||||
continue; // Empty list means either Entity has not been found or it is between main sections. Thus, this might lead to wrong reanalysis.
|
||||
}
|
||||
sectionsToReanalyse.add(entry.getContainingNodeId()
|
||||
.get(0));
|
||||
}
|
||||
}
|
||||
|
||||
var dictionaryIncrementsSearch = new SearchImplementation(dictionaryIncrement.getValues()
|
||||
.stream()
|
||||
@@ -123,7 +133,7 @@ public class SectionFinderService {
|
||||
}
|
||||
|
||||
|
||||
public static Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
|
||||
private static Set<String> getRelevantManuallyModifiedAnnotationIds(ManualRedactions manualRedactions) {
|
||||
|
||||
if (manualRedactions == null) {
|
||||
return new HashSet<>();
|
||||
|
||||
-1
@@ -12,7 +12,6 @@ public abstract class SemanticNodeComparators implements Comparator<SemanticNode
|
||||
return new FirstSemanticNode();
|
||||
}
|
||||
|
||||
|
||||
public static class FirstSemanticNode extends SemanticNodeComparators {
|
||||
|
||||
@Override
|
||||
|
||||
+3
-8
@@ -50,9 +50,7 @@ public class ComponentDroolsExecutionService {
|
||||
.filter(entityLogEntry -> entityLogEntry.getState().equals(EntryState.APPLIED))
|
||||
.map(entry -> Entity.fromEntityLogEntry(entry, document))
|
||||
.forEach(kieSession::insert);
|
||||
fileAttributes.stream()
|
||||
.filter(f -> f.getValue() != null)
|
||||
.forEach(kieSession::insert);
|
||||
fileAttributes.stream().filter(f -> f.getValue() != null).forEach(kieSession::insert);
|
||||
|
||||
CompletableFuture<Void> completableFuture = CompletableFuture.supplyAsync(() -> {
|
||||
kieSession.fireAllRules();
|
||||
@@ -60,8 +58,7 @@ public class ComponentDroolsExecutionService {
|
||||
});
|
||||
|
||||
try {
|
||||
completableFuture.orTimeout(settings.getDroolsExecutionTimeoutSecs(), TimeUnit.SECONDS)
|
||||
.get();
|
||||
completableFuture.orTimeout(settings.getDroolsExecutionTimeoutSecs(), TimeUnit.SECONDS).get();
|
||||
} catch (ExecutionException e) {
|
||||
kieSession.dispose();
|
||||
if (e.getCause() instanceof TimeoutException) {
|
||||
@@ -74,9 +71,7 @@ public class ComponentDroolsExecutionService {
|
||||
}
|
||||
|
||||
List<FileAttribute> resultingFileAttributes = getFileAttributes(kieSession);
|
||||
List<Component> components = getComponents(kieSession).stream()
|
||||
.sorted(ComponentComparator.first())
|
||||
.toList();
|
||||
List<Component> components = getComponents(kieSession).stream().sorted(ComponentComparator.first()).toList();
|
||||
kieSession.dispose();
|
||||
return components;
|
||||
}
|
||||
|
||||
+43
-111
@@ -7,7 +7,6 @@ import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.drools.drl.parser.DroolsParserException;
|
||||
import org.kie.api.builder.KieBuilder;
|
||||
@@ -16,13 +15,11 @@ import org.springframework.stereotype.Service;
|
||||
|
||||
import com.google.common.collect.Sets;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.RuleFileType;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsBlacklistErrorMessage;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxDeprecatedWarnings;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxErrorMessage;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.RuleValidationModel;
|
||||
import com.iqser.red.service.redaction.v1.server.DeprecatedElementsFinder;
|
||||
import com.iqser.red.service.redaction.v1.server.RedactionServiceSettings;
|
||||
import com.iqser.red.service.redaction.v1.server.model.dictionary.SearchImplementation;
|
||||
import com.iqser.red.service.redaction.v1.server.model.drools.BasicQuery;
|
||||
import com.iqser.red.service.redaction.v1.server.model.drools.BasicRule;
|
||||
@@ -38,61 +35,31 @@ import lombok.extern.slf4j.Slf4j;
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@Slf4j
|
||||
public class DroolsValidationService {
|
||||
public class DroolsSyntaxValidationService {
|
||||
|
||||
private final RedactionServiceSettings redactionServiceSettings;
|
||||
private final KieContainerCreationService kieContainerCreationService;
|
||||
private final DeprecatedElementsFinder deprecatedElementsFinder;
|
||||
private static final Pattern allowedImportsPattern = Pattern.compile("^(?:import\\s+static\\s+)?(?:import\\s+)?(?:com\\.knecon\\.fforesight|com\\.iqser\\.red)\\..*;$");
|
||||
public static final String LINEBREAK_MATCHER = "\\R";
|
||||
|
||||
private final KieContainerCreationService kieContainerCreationService;
|
||||
|
||||
private final DeprecatedElementsFinder deprecatedElementsFinder;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public DroolsValidation testRules(RuleValidationModel rules) {
|
||||
public DroolsSyntaxValidation testRules(RuleValidationModel rules) {
|
||||
|
||||
DroolsValidation customDroolsValidation;
|
||||
DroolsSyntaxValidation customDroolsSyntaxValidation;
|
||||
try {
|
||||
customDroolsValidation = buildCustomDroolsValidation(rules.getRulesString(), RuleFileType.valueOf(rules.getRuleFileType()));
|
||||
customDroolsSyntaxValidation = buildCustomDroolsSyntaxValidation(rules.getRulesString(), RuleFileType.valueOf(rules.getRuleFileType()));
|
||||
} catch (DroolsParserException e) {
|
||||
// this means the parser could not parse the file at all. In this case use drools compiler only as it will return useful error messages.
|
||||
customDroolsValidation = new DroolsValidation();
|
||||
customDroolsSyntaxValidation = new DroolsSyntaxValidation();
|
||||
}
|
||||
DroolsValidation droolsCompilerValidation = buildDroolsCompilerValidation(rules);
|
||||
droolsCompilerValidation.getSyntaxErrorMessages().addAll(customDroolsValidation.getSyntaxErrorMessages());
|
||||
droolsCompilerValidation.getDeprecatedWarnings().addAll(customDroolsValidation.getDeprecatedWarnings());
|
||||
droolsCompilerValidation.getBlacklistErrorMessages().addAll(customDroolsValidation.getBlacklistErrorMessages());
|
||||
return droolsCompilerValidation;
|
||||
DroolsSyntaxValidation droolsCompilerSyntaxValidation = buildDroolsCompilerSyntaxValidation(rules);
|
||||
droolsCompilerSyntaxValidation.getDroolsSyntaxErrorMessages().addAll(customDroolsSyntaxValidation.getDroolsSyntaxErrorMessages());
|
||||
droolsCompilerSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().addAll(customDroolsSyntaxValidation.getDroolsSyntaxDeprecatedWarnings());
|
||||
return droolsCompilerSyntaxValidation;
|
||||
}
|
||||
|
||||
|
||||
private DroolsValidation buildCustomDroolsValidation(String ruleString, RuleFileType ruleFileType) throws DroolsParserException {
|
||||
|
||||
RuleFileBluePrint ruleFileBluePrint = RuleFileParser.buildBluePrintFromRulesString(ruleString);
|
||||
|
||||
DroolsValidation customValidation = ruleFileBluePrint.getDroolsValidation();
|
||||
|
||||
addSyntaxDeprecatedWarnings(ruleFileBluePrint, customValidation);
|
||||
|
||||
addSyntaxErrorMessages(ruleFileType, ruleFileBluePrint, customValidation);
|
||||
|
||||
if (redactionServiceSettings.isRuleExecutionSecured()) {
|
||||
addBlacklistErrorMessages(ruleFileBluePrint, customValidation);
|
||||
}
|
||||
|
||||
return customValidation;
|
||||
}
|
||||
|
||||
|
||||
private void addSyntaxDeprecatedWarnings(RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
|
||||
// find deprecated elements in the ruleFileBluePrint
|
||||
DroolsSyntaxDeprecatedWarnings warningMessageForImports = getWarningsForDeprecatedImports(ruleFileBluePrint);
|
||||
if (warningMessageForImports != null) {
|
||||
customValidation.getDeprecatedWarnings().add(warningMessageForImports);
|
||||
}
|
||||
customValidation.getDeprecatedWarnings().addAll(getWarningsForDeprecatedRules(ruleFileBluePrint));
|
||||
}
|
||||
|
||||
|
||||
private DroolsSyntaxDeprecatedWarnings getWarningsForDeprecatedImports(RuleFileBluePrint ruleFileBluePrint) {
|
||||
|
||||
if (!deprecatedElementsFinder.getDeprecatedClasses().isEmpty()) {
|
||||
@@ -103,13 +70,13 @@ public class DroolsValidationService {
|
||||
String sb = "Following imports are deprecated: \n" + matches.stream()
|
||||
.map(m -> imports.substring(m.startIndex(), m.endIndex()))
|
||||
.collect(Collectors.joining("\n"));
|
||||
return DroolsSyntaxDeprecatedWarnings.builder().line(ruleFileBluePrint.getImportLine()).column(0).message(sb).build();
|
||||
return DroolsSyntaxDeprecatedWarnings.builder().line(ruleFileBluePrint.getImportLine()).column(0).message(sb)
|
||||
.build();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
private List<DroolsSyntaxDeprecatedWarnings> getWarningsForDeprecatedRules(RuleFileBluePrint ruleFileBluePrint) {
|
||||
|
||||
List<DroolsSyntaxDeprecatedWarnings> warningMessages = new ArrayList<>();
|
||||
@@ -129,7 +96,8 @@ public class DroolsValidationService {
|
||||
.distinct()
|
||||
.map(dm -> String.format("Method %s might be deprecated because of \n %s \n", dm, deprecatedMethodsSignatureMap.get(dm)))
|
||||
.collect(Collectors.joining("\n"));
|
||||
warningMessages.add(DroolsSyntaxDeprecatedWarnings.builder().line(basicRule.getLine()).column(0).message(warningMessage).build());
|
||||
warningMessages.add(DroolsSyntaxDeprecatedWarnings.builder().line(basicRule.getLine()).column(0).message(warningMessage)
|
||||
.build());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -143,7 +111,18 @@ public class DroolsValidationService {
|
||||
}
|
||||
|
||||
|
||||
private void addSyntaxErrorMessages(RuleFileType ruleFileType, RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
|
||||
private DroolsSyntaxValidation buildCustomDroolsSyntaxValidation(String ruleString, RuleFileType ruleFileType) throws DroolsParserException {
|
||||
|
||||
RuleFileBluePrint ruleFileBluePrint = RuleFileParser.buildBluePrintFromRulesString(ruleString);
|
||||
|
||||
DroolsSyntaxValidation customSyntaxValidation = ruleFileBluePrint.getDroolsSyntaxValidation();
|
||||
|
||||
// find deprecated elements in the ruleFileBluePrint
|
||||
DroolsSyntaxDeprecatedWarnings warningMessageForImports = getWarningsForDeprecatedImports(ruleFileBluePrint);
|
||||
if (warningMessageForImports != null) {
|
||||
customSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().add(warningMessageForImports);
|
||||
}
|
||||
customSyntaxValidation.getDroolsSyntaxDeprecatedWarnings().addAll(getWarningsForDeprecatedRules(ruleFileBluePrint));
|
||||
|
||||
RuleFileBluePrint baseRuleFileBluePrint = switch (ruleFileType) {
|
||||
case ENTITY -> RuleFileParser.buildBluePrintFromRulesString(RuleManagementResources.getBaseRuleFileString());
|
||||
@@ -151,7 +130,7 @@ public class DroolsValidationService {
|
||||
};
|
||||
|
||||
if (!importsAreValid(baseRuleFileBluePrint, ruleFileBluePrint)) {
|
||||
customValidation.getSyntaxErrorMessages()
|
||||
customSyntaxValidation.getDroolsSyntaxErrorMessages()
|
||||
.add(DroolsSyntaxErrorMessage.builder()
|
||||
.line(ruleFileBluePrint.getImportLine())
|
||||
.column(0)
|
||||
@@ -159,7 +138,7 @@ public class DroolsValidationService {
|
||||
.build());
|
||||
}
|
||||
if (!ruleFileBluePrint.getGlobals().equals(baseRuleFileBluePrint.getGlobals())) {
|
||||
customValidation.getSyntaxErrorMessages()
|
||||
customSyntaxValidation.getDroolsSyntaxErrorMessages()
|
||||
.add(DroolsSyntaxErrorMessage.builder()
|
||||
.line(ruleFileBluePrint.getGlobalsLine())
|
||||
.column(0)
|
||||
@@ -169,7 +148,7 @@ public class DroolsValidationService {
|
||||
baseRuleFileBluePrint.getQueries()
|
||||
.forEach(basicQuery -> {
|
||||
if (!validateQueryIsPresent(basicQuery, ruleFileBluePrint)) {
|
||||
customValidation.getSyntaxErrorMessages()
|
||||
customSyntaxValidation.getDroolsSyntaxErrorMessages()
|
||||
.add(DroolsSyntaxErrorMessage.builder()
|
||||
.line(basicQuery.getLine())
|
||||
.column(0)
|
||||
@@ -180,14 +159,12 @@ public class DroolsValidationService {
|
||||
if (ruleFileType.equals(RuleFileType.ENTITY)) {
|
||||
String requiredAgendaGroup = "LOCAL_DICTIONARY_ADDS";
|
||||
if (!validateAgendaGroupIsPresent(ruleFileBluePrint, requiredAgendaGroup)) {
|
||||
customValidation.getSyntaxErrorMessages()
|
||||
.add(DroolsSyntaxErrorMessage.builder()
|
||||
.line(0)
|
||||
.column(0)
|
||||
.message(String.format("At least one rule with Agenda-Group '%s' required!", requiredAgendaGroup))
|
||||
customSyntaxValidation.getDroolsSyntaxErrorMessages()
|
||||
.add(DroolsSyntaxErrorMessage.builder().line(0).column(0).message(String.format("At least one rule with Agenda-Group '%s' required!", requiredAgendaGroup))
|
||||
.build());
|
||||
}
|
||||
}
|
||||
return customSyntaxValidation;
|
||||
}
|
||||
|
||||
|
||||
@@ -219,54 +196,7 @@ public class DroolsValidationService {
|
||||
}
|
||||
|
||||
|
||||
private void addBlacklistErrorMessages(RuleFileBluePrint ruleFileBluePrint, DroolsValidation customValidation) {
|
||||
|
||||
List<DroolsBlacklistErrorMessage> blacklistErrorMessages = new ArrayList<>();
|
||||
|
||||
List<String> blacklistedKeywords = parseBlacklistFile(RuleManagementResources.getBlacklistFileString());
|
||||
|
||||
// checks the rules for occurrence of blacklisted keyword
|
||||
if (!blacklistedKeywords.isEmpty()) {
|
||||
SearchImplementation blacklistedKeywordSearchImplementation = new SearchImplementation(blacklistedKeywords, false);
|
||||
|
||||
for (RuleClass ruleClass : ruleFileBluePrint.getRuleClasses()) {
|
||||
for (RuleUnit ruleUnit : ruleClass.ruleUnits()) {
|
||||
for (BasicRule basicRule : ruleUnit.rules()) {
|
||||
List<SearchImplementation.MatchPosition> matches = blacklistedKeywordSearchImplementation.getMatches(basicRule.getCode());
|
||||
|
||||
if (!matches.isEmpty()) {
|
||||
List<String> foundBlacklistedKeywords = matches.stream()
|
||||
.map(m -> basicRule.getCode().substring(m.startIndex(), m.endIndex()))
|
||||
.distinct()
|
||||
.toList();
|
||||
blacklistErrorMessages.add(DroolsBlacklistErrorMessage.builder()
|
||||
.line(basicRule.getLine())
|
||||
.column(0)
|
||||
.blacklistedKeywords(foundBlacklistedKeywords)
|
||||
.build());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
customValidation.getBlacklistErrorMessages()
|
||||
.addAll(blacklistErrorMessages.stream()
|
||||
.sorted(Comparator.comparingInt(DroolsBlacklistErrorMessage::getLine))
|
||||
.toList());
|
||||
}
|
||||
|
||||
|
||||
private List<String> parseBlacklistFile(String blacklistFileString) {
|
||||
|
||||
return Stream.of(blacklistFileString.split(LINEBREAK_MATCHER))
|
||||
.distinct()
|
||||
.filter(s -> !s.isBlank())
|
||||
.toList();
|
||||
}
|
||||
|
||||
|
||||
private DroolsValidation buildDroolsCompilerValidation(RuleValidationModel rules) {
|
||||
private DroolsSyntaxValidation buildDroolsCompilerSyntaxValidation(RuleValidationModel rules) {
|
||||
|
||||
var versionId = System.currentTimeMillis();
|
||||
var testRules = "test-rules";
|
||||
@@ -274,23 +204,25 @@ public class DroolsValidationService {
|
||||
versionId,
|
||||
rules.getRulesString(),
|
||||
RuleFileType.valueOf(rules.getRuleFileType()));
|
||||
return buildDroolsCompilerValidation(kieBuilder);
|
||||
return buildDroolsCompilerSyntaxValidation(kieBuilder);
|
||||
}
|
||||
|
||||
|
||||
private DroolsValidation buildDroolsCompilerValidation(KieBuilder kieBuilder) {
|
||||
private DroolsSyntaxValidation buildDroolsCompilerSyntaxValidation(KieBuilder kieBuilder) {
|
||||
|
||||
List<Message> errorMessages = kieBuilder.getResults().getMessages(Message.Level.ERROR);
|
||||
List<DroolsSyntaxErrorMessage> droolsSyntaxErrorMessages = errorMessages.stream()
|
||||
.map(this::buildDroolsSyntaxErrorMessage)
|
||||
.collect(Collectors.toList());
|
||||
return DroolsValidation.builder().syntaxErrorMessages(droolsSyntaxErrorMessages).build();
|
||||
return DroolsSyntaxValidation.builder().droolsSyntaxErrorMessages(droolsSyntaxErrorMessages)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private DroolsSyntaxErrorMessage buildDroolsSyntaxErrorMessage(Message message) {
|
||||
|
||||
return DroolsSyntaxErrorMessage.builder().line(message.getLine()).column(message.getColumn()).message(message.getText()).build();
|
||||
return DroolsSyntaxErrorMessage.builder().line(message.getLine()).column(message.getColumn()).message(message.getText())
|
||||
.build();
|
||||
}
|
||||
|
||||
}
|
||||
+2
-1
@@ -30,7 +30,8 @@ public class KieContainerCreationService {
|
||||
private final RulesClient rulesClient;
|
||||
|
||||
|
||||
@Observed(name = "KieContainerCreationService", contextualName = "get-kie-container")
|
||||
@Observed(name = "KieContainerCreationService",
|
||||
contextualName = "get-kie-container")
|
||||
public KieWrapper getLatestKieContainer(String dossierTemplateId, RuleFileType ruleFileType) {
|
||||
|
||||
try {
|
||||
|
||||
+27
-53
@@ -16,7 +16,7 @@ import org.drools.drl.ast.descr.RuleDescr;
|
||||
import org.drools.drl.parser.DrlParser;
|
||||
import org.kie.internal.builder.conf.LanguageLevelOption;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsValidation;
|
||||
import com.iqser.red.service.redaction.v1.model.DroolsSyntaxValidation;
|
||||
import com.iqser.red.service.redaction.v1.server.model.drools.BasicQuery;
|
||||
import com.iqser.red.service.redaction.v1.server.model.drools.BasicRule;
|
||||
import com.iqser.red.service.redaction.v1.server.model.drools.RuleClass;
|
||||
@@ -38,7 +38,7 @@ public class RuleFileParser {
|
||||
@SneakyThrows
|
||||
public RuleFileBluePrint buildBluePrintFromRulesString(String ruleString) {
|
||||
|
||||
DroolsValidation customDroolsValidation = DroolsValidation.builder().build();
|
||||
DroolsSyntaxValidation customDroolsSyntaxValidation = DroolsSyntaxValidation.builder().build();
|
||||
DrlParser parser = new DrlParser(LanguageLevelOption.DRL6);
|
||||
PackageDescr packageDescr = parser.parse(false, ruleString);
|
||||
List<BasicRule> allRules = new LinkedList<>();
|
||||
@@ -48,16 +48,11 @@ public class RuleFileParser {
|
||||
if (rule.isQuery()) {
|
||||
allQueries.add(new BasicQuery(rule.getName(), rule.getLine(), ruleString.substring(rule.getStartCharacter(), rule.getEndCharacter())));
|
||||
} else {
|
||||
validateRule(ruleString, rule, customDroolsValidation, allRules);
|
||||
validateRule(ruleString, rule, customDroolsSyntaxValidation, allRules);
|
||||
}
|
||||
}
|
||||
|
||||
String imports = ruleString.substring(0,
|
||||
packageDescr.getImports()
|
||||
.stream()
|
||||
.mapToInt(ImportDescr::getEndCharacter)
|
||||
.max()
|
||||
.orElseThrow() + 1);
|
||||
String imports = ruleString.substring(0, packageDescr.getImports().stream().mapToInt(ImportDescr::getEndCharacter).max().orElseThrow() + 1);
|
||||
String globals = packageDescr.getGlobals()
|
||||
.stream()
|
||||
.map(globalDescr -> ruleString.substring(globalDescr.getStartCharacter(), globalDescr.getEndCharacter()))
|
||||
@@ -66,87 +61,66 @@ public class RuleFileParser {
|
||||
List<RuleClass> ruleClasses = buildRuleClasses(allRules);
|
||||
|
||||
return new RuleFileBluePrint(imports.trim(),
|
||||
packageDescr.getImports()
|
||||
.stream()
|
||||
.findFirst()
|
||||
.map(ImportDescr::getLine)
|
||||
.orElse(0),
|
||||
globals.trim(),
|
||||
packageDescr.getGlobals()
|
||||
.stream()
|
||||
.findFirst()
|
||||
.map(GlobalDescr::getLine)
|
||||
.orElse(0),
|
||||
allQueries,
|
||||
ruleClasses, customDroolsValidation);
|
||||
packageDescr.getImports().stream().findFirst().map(ImportDescr::getLine).orElse(0),
|
||||
globals.trim(),
|
||||
packageDescr.getGlobals().stream().findFirst().map(GlobalDescr::getLine).orElse(0), allQueries,
|
||||
ruleClasses,
|
||||
customDroolsSyntaxValidation);
|
||||
}
|
||||
|
||||
|
||||
private static void validateRule(String ruleString, RuleDescr rule, DroolsValidation customDroolsValidation, List<BasicRule> allRules) {
|
||||
private static void validateRule(String ruleString, RuleDescr rule, DroolsSyntaxValidation customDroolsSyntaxValidation, List<BasicRule> allRules) {
|
||||
|
||||
BasicRule basicRule;
|
||||
try {
|
||||
basicRule = BasicRule.fromRuleDescr(rule, ruleString);
|
||||
} catch (Exception e) {
|
||||
customDroolsValidation.addErrorMessage(rule.getLine(), rule.getColumn(), "Malformed rule name, correct format is \"\\w+.\\d+.\\d+: <rule description>\"");
|
||||
customDroolsSyntaxValidation.addErrorMessage(rule.getLine(), rule.getColumn(), "Malformed rule name, correct format is \"\\w+.\\d+.\\d+: <rule description>\"");
|
||||
return;
|
||||
}
|
||||
if (allRules.contains(basicRule)) {
|
||||
addDuplicateRuleIdentifierErrorMessage(rule, basicRule, customDroolsValidation);
|
||||
addDuplicateRuleIdentifierErrorMessage(rule, basicRule, customDroolsSyntaxValidation);
|
||||
}
|
||||
validateRuleIdentifierInCodeIsSame(basicRule.getCode(), basicRule.getIdentifier().toString(), rule.getLine(), customDroolsValidation);
|
||||
validateRuleIdentifierInCodeIsSame(basicRule.getCode(), basicRule.getIdentifier().toString(), rule.getLine(), customDroolsSyntaxValidation);
|
||||
allRules.add(BasicRule.fromRuleDescr(rule, ruleString));
|
||||
}
|
||||
|
||||
|
||||
private static void validateRuleIdentifierInCodeIsSame(String code, String identifier, int lineOffset, DroolsValidation customDroolsValidation) {
|
||||
private static void validateRuleIdentifierInCodeIsSame(String code, String identifier, int lineOffset, DroolsSyntaxValidation customDroolsSyntaxValidation) {
|
||||
|
||||
Matcher matcher = ruleIdentifierInCodeFinder.matcher(code);
|
||||
while (matcher.find()) {
|
||||
String identifierInCode = code.substring(matcher.start(1), matcher.end(1));
|
||||
long line = code.substring(0, matcher.start(1)).lines()
|
||||
.count() + lineOffset - 1;
|
||||
long line = code.substring(0, matcher.start(1)).lines().count() + lineOffset - 1;
|
||||
if (!identifier.equals(identifierInCode)) {
|
||||
customDroolsValidation.addErrorMessage((int) line,
|
||||
0,
|
||||
String.format("Rule identifier %s is not equal to rule identifier %s in rule name!", identifierInCode, identifier));
|
||||
customDroolsSyntaxValidation.addErrorMessage((int) line,
|
||||
0,
|
||||
String.format("Rule identifier %s is not equal to rule identifier %s in rule name!", identifierInCode, identifier));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addDuplicateRuleIdentifierErrorMessage(RuleDescr rule, BasicRule basicRule, DroolsValidation customDroolsValidation) {
|
||||
private void addDuplicateRuleIdentifierErrorMessage(RuleDescr rule, BasicRule basicRule, DroolsSyntaxValidation customDroolsSyntaxValidation) {
|
||||
|
||||
customDroolsValidation.addErrorMessage(rule.getLine(),
|
||||
rule.getColumn(),
|
||||
String.format("RuleIdentifier: %s is a duplicate, duplicates are not allowed!", basicRule.getIdentifier()));
|
||||
customDroolsSyntaxValidation.addErrorMessage(rule.getLine(),
|
||||
rule.getColumn(),
|
||||
String.format("RuleIdentifier: %s is a duplicate, duplicates are not allowed!", basicRule.getIdentifier()));
|
||||
}
|
||||
|
||||
|
||||
private List<RuleClass> buildRuleClasses(List<BasicRule> allRules) {
|
||||
|
||||
List<RuleType> ruleTypeOrder = allRules.stream()
|
||||
.map(BasicRule::getIdentifier)
|
||||
.map(RuleIdentifier::type)
|
||||
.distinct()
|
||||
.toList();
|
||||
Map<RuleType, List<BasicRule>> rulesPerType = allRules.stream()
|
||||
.collect(groupingBy(rule -> rule.getIdentifier().type()));
|
||||
return ruleTypeOrder.stream()
|
||||
.map(type -> new RuleClass(type, groupingByGroup(rulesPerType.get(type))))
|
||||
.collect(Collectors.toList());
|
||||
List<RuleType> ruleTypeOrder = allRules.stream().map(BasicRule::getIdentifier).map(RuleIdentifier::type).distinct().toList();
|
||||
Map<RuleType, List<BasicRule>> rulesPerType = allRules.stream().collect(groupingBy(rule -> rule.getIdentifier().type()));
|
||||
return ruleTypeOrder.stream().map(type -> new RuleClass(type, groupingByGroup(rulesPerType.get(type)))).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
|
||||
private List<RuleUnit> groupingByGroup(List<BasicRule> rules) {
|
||||
|
||||
Map<Integer, List<BasicRule>> rulesPerUnit = rules.stream()
|
||||
.collect(groupingBy(rule -> rule.getIdentifier().unit()));
|
||||
return rulesPerUnit.keySet()
|
||||
.stream()
|
||||
.sorted()
|
||||
.map(unit -> new RuleUnit(unit, rulesPerUnit.get(unit)))
|
||||
.collect(Collectors.toList());
|
||||
Map<Integer, List<BasicRule>> rulesPerUnit = rules.stream().collect(groupingBy(rule -> rule.getIdentifier().unit()));
|
||||
return rulesPerUnit.keySet().stream().sorted().map(unit -> new RuleUnit(unit, rulesPerUnit.get(unit))).collect(Collectors.toList());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-1
@@ -16,7 +16,6 @@ public class ObservedStorageService {
|
||||
|
||||
@Observed(name = "RedactionStorageService", contextualName = "get-document-data")
|
||||
public DocumentData getDocumentData(String dossierId, String fileId) {
|
||||
|
||||
return redactionStorageService.getDocumentData(dossierId, fileId);
|
||||
}
|
||||
|
||||
|
||||
+25
-89
@@ -3,24 +3,20 @@ package com.iqser.red.service.redaction.v1.server.storage;
|
||||
import java.io.File;
|
||||
import java.io.FileInputStream;
|
||||
import java.io.InputStream;
|
||||
import java.util.Collection;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.springframework.cache.annotation.Cacheable;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.componentlog.ComponentLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLogEntry;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.imported.ImportedRedactionsPerPage;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.redactionlog.RedactionLog;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.mongo.service.EntityLogMongoService;
|
||||
import com.iqser.red.service.redaction.v1.server.client.model.NerEntitiesModel;
|
||||
import com.iqser.red.service.redaction.v1.server.model.document.DocumentData;
|
||||
import com.iqser.red.service.persistence.service.v1.api.shared.model.analysislog.entitylog.EntityLog;
|
||||
import com.iqser.red.service.redaction.v1.server.utils.exception.NotFoundException;
|
||||
import com.iqser.red.storage.commons.exception.StorageObjectDoesNotExist;
|
||||
import com.iqser.red.storage.commons.service.StorageService;
|
||||
@@ -43,8 +39,6 @@ public class RedactionStorageService {
|
||||
|
||||
private final StorageService storageService;
|
||||
|
||||
private final EntityLogMongoService entityLogMongoService;
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public InputStream getStoredObject(String storageId) {
|
||||
@@ -81,45 +75,14 @@ public class RedactionStorageService {
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void updateEntityLogWithoutEntries(String dossierId, String fileId, EntityLog entityLog) {
|
||||
|
||||
entityLogMongoService.saveEntityLogWithoutEntries(dossierId, fileId, entityLog);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void saveEntityLog(String dossierId, String fileId, EntityLog entityLog) {
|
||||
|
||||
entityLogMongoService.saveEntityLog(dossierId, fileId, entityLog);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void saveEntityLogEntries(String dossierId, String fileId, List<EntityLogEntry> entityLogEntries) {
|
||||
|
||||
entityLogMongoService.saveEntityLogEntries(dossierId, fileId, entityLogEntries);
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public void updateEntityLogEntries(String dossierId, String fileId, List<EntityLogEntry> entityLogEntries) {
|
||||
|
||||
entityLogMongoService.updateEntityLogEntries(dossierId, fileId, entityLogEntries);
|
||||
}
|
||||
|
||||
|
||||
@Timed("redactmanager_getImportedRedactions")
|
||||
public ImportedRedactions getImportedRedactions(String dossierId, String fileId) {
|
||||
|
||||
try {
|
||||
ImportedRedactionsPerPage importedRedactionsPerPage = storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
|
||||
ImportedRedactionsPerPage.class);
|
||||
return new ImportedRedactions(importedRedactionsPerPage.getImportedRedactions().values()
|
||||
.stream()
|
||||
.flatMap(List::stream)
|
||||
.collect(Collectors.toList()));
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
|
||||
ImportedRedactionsPerPage.class);
|
||||
return new ImportedRedactions(importedRedactionsPerPage.getImportedRedactions().values().stream().flatMap(List::stream).collect(Collectors.toList()));
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("Imported redactions not available.");
|
||||
return new ImportedRedactions();
|
||||
@@ -127,13 +90,14 @@ public class RedactionStorageService {
|
||||
}
|
||||
|
||||
|
||||
|
||||
@Timed("redactmanager_getImportedRedactions")
|
||||
public ImportedRedactionsPerPage getImportedRedactionsPerPage(String dossierId, String fileId) {
|
||||
|
||||
try {
|
||||
return storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
|
||||
ImportedRedactionsPerPage.class);
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.IMPORTED_REDACTIONS),
|
||||
ImportedRedactionsPerPage.class);
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("Imported redactions not available.");
|
||||
return null;
|
||||
@@ -147,12 +111,12 @@ public class RedactionStorageService {
|
||||
|
||||
try {
|
||||
RedactionLog redactionLog = storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.REDACTION_LOG),
|
||||
RedactionLog.class);
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.REDACTION_LOG),
|
||||
RedactionLog.class);
|
||||
redactionLog.setRedactionLogEntry(redactionLog.getRedactionLogEntry()
|
||||
.stream()
|
||||
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
|
||||
.collect(Collectors.toList()));
|
||||
.stream()
|
||||
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
|
||||
.collect(Collectors.toList()));
|
||||
return redactionLog;
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("RedactionLog not available.");
|
||||
@@ -166,12 +130,11 @@ public class RedactionStorageService {
|
||||
public EntityLog getEntityLog(String dossierId, String fileId) {
|
||||
|
||||
try {
|
||||
EntityLog entityLog = entityLogMongoService.findEntityLogByDossierIdAndFileId(dossierId, fileId)
|
||||
.orElseThrow(() -> new StorageObjectDoesNotExist(""));
|
||||
EntityLog entityLog = storageService.readJSONObject(TenantContext.getTenantId(), StorageIdUtils.getStorageId(dossierId, fileId, FileType.ENTITY_LOG), EntityLog.class);
|
||||
entityLog.setEntityLogEntry(entityLog.getEntityLogEntry()
|
||||
.stream()
|
||||
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
|
||||
.collect(Collectors.toList()));
|
||||
.stream()
|
||||
.filter(entry -> !(entry.getPositions() == null || entry.getPositions().isEmpty()))
|
||||
.collect(Collectors.toList()));
|
||||
return entityLog;
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("EntityLog not available.");
|
||||
@@ -181,33 +144,6 @@ public class RedactionStorageService {
|
||||
}
|
||||
|
||||
|
||||
@Timed("redactmanager_getRedactionLog")
|
||||
public EntityLog getEntityLogWithoutEntries(String dossierId, String fileId) {
|
||||
|
||||
try {
|
||||
return entityLogMongoService.findEntityLogWithoutEntries(dossierId, fileId)
|
||||
.orElseThrow(() -> new StorageObjectDoesNotExist(""));
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("EntityLog not available.");
|
||||
return null;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
public Set<Integer> findIdsOfSectionsToReanalyse(String dossierId, String fileId, Collection<String> entryIds) {
|
||||
|
||||
return entityLogMongoService.findFirstContainingNodeIdForEachEntry(dossierId, fileId, entryIds);
|
||||
}
|
||||
|
||||
|
||||
public List<EntityLogEntry> findEntriesContainedBySectionsOrNotContained(String dossierId, String fileId, Collection<Integer> sectionIds) {
|
||||
|
||||
return entityLogMongoService.findEntityLogEntriesNotContainedOrFirstContainedByElementInList(dossierId, fileId, sectionIds);
|
||||
}
|
||||
|
||||
|
||||
|
||||
// !Warning! before activating redis cache you need to set
|
||||
// -Dio.netty.noPreferDirect=true -XX:MaxDirectMemorySize=1000M
|
||||
// Jvm args to the largest document data size we want to process. for 4443 pages file that was 500mb.
|
||||
@@ -220,17 +156,17 @@ public class RedactionStorageService {
|
||||
try {
|
||||
return DocumentData.builder()
|
||||
.documentStructure(storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_STRUCTURE),
|
||||
DocumentStructure.class))
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_STRUCTURE),
|
||||
DocumentStructure.class))
|
||||
.documentTextData(storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_TEXT),
|
||||
DocumentTextData[].class))
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_TEXT),
|
||||
DocumentTextData[].class))
|
||||
.documentPositionData(storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_POSITION),
|
||||
DocumentPositionData[].class))
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_POSITION),
|
||||
DocumentPositionData[].class))
|
||||
.documentPages(storageService.readJSONObject(TenantContext.getTenantId(),
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_PAGES),
|
||||
DocumentPage[].class))
|
||||
StorageIdUtils.getStorageId(dossierId, fileId, FileType.DOCUMENT_PAGES),
|
||||
DocumentPage[].class))
|
||||
.build();
|
||||
} catch (StorageObjectDoesNotExist e) {
|
||||
log.debug("DocumentData not available.");
|
||||
@@ -262,7 +198,7 @@ public class RedactionStorageService {
|
||||
|
||||
public boolean entityLogExists(String dossierId, String fileId) {
|
||||
|
||||
return entityLogMongoService.entityLogDocumentExists(dossierId, fileId);
|
||||
return storageService.objectExists(TenantContext.getTenantId(), StorageIdUtils.getStorageId(dossierId, fileId, FileType.ENTITY_LOG));
|
||||
}
|
||||
|
||||
|
||||
|
||||
-18
@@ -11,7 +11,6 @@ public class RuleManagementResources {
|
||||
|
||||
private static final String folderPrefix = "drools";
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public static InputStream getBaseRuleFileInputStream() {
|
||||
|
||||
@@ -27,7 +26,6 @@ public class RuleManagementResources {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public static InputStream getBaseComponentRuleFileInputStream() {
|
||||
|
||||
@@ -43,20 +41,4 @@ public class RuleManagementResources {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public static InputStream getBlacklistFileInputStream() {
|
||||
|
||||
return new ClassPathResource(Path.of(folderPrefix, "blacklist.txt").toString()).getInputStream();
|
||||
}
|
||||
|
||||
|
||||
@SneakyThrows
|
||||
public static String getBlacklistFileString() {
|
||||
|
||||
try (var in = getBlacklistFileInputStream()) {
|
||||
return new String(in.readAllBytes());
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+21
-56
@@ -1,19 +1,10 @@
|
||||
package com.iqser.red.service.redaction.v1.server.utils;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.text.DateFormat;
|
||||
import java.text.SimpleDateFormat;
|
||||
import java.time.LocalDate;
|
||||
import java.time.ZoneId;
|
||||
import java.time.format.DateTimeFormatter;
|
||||
import java.time.format.DateTimeFormatterBuilder;
|
||||
import java.time.format.DateTimeParseException;
|
||||
import java.time.format.ResolverStyle;
|
||||
import java.util.Date;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
@@ -26,65 +17,39 @@ import lombok.extern.slf4j.Slf4j;
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class DateConverter {
|
||||
|
||||
private static DateTimeFormatter DATE_TIME_FORMATTER;
|
||||
static List<SimpleDateFormat> formats = List.of(new SimpleDateFormat("dd MMM yy", Locale.ENGLISH),
|
||||
new SimpleDateFormat("dd MM yyyy", Locale.ENGLISH),
|
||||
new SimpleDateFormat("dd MM yyyy.", Locale.ENGLISH),
|
||||
new SimpleDateFormat("dd MMMM yyyy", Locale.ENGLISH),
|
||||
new SimpleDateFormat("MMMM dd, yyyy", Locale.ENGLISH),
|
||||
new SimpleDateFormat("dd-MMM-yyyy", Locale.ENGLISH));
|
||||
|
||||
|
||||
public Optional<Date> parseDate(String dateAsString) {
|
||||
|
||||
DateTimeFormatter formatter = getDateTimeFormatter();
|
||||
String cleanDate = dateAsString.trim();
|
||||
cleanDate = removeTrailingDot(cleanDate);
|
||||
try {
|
||||
LocalDate localDate = LocalDate.parse(cleanDate, formatter);
|
||||
Date date = Date.from(localDate.atStartOfDay(ZoneId.systemDefault()).toInstant());
|
||||
return Optional.of(date);
|
||||
} catch (DateTimeParseException e) {
|
||||
log.warn("Failed to parse date: {}", cleanDate);
|
||||
Date date = null;
|
||||
for (SimpleDateFormat format : formats) {
|
||||
|
||||
try {
|
||||
date = format.parse(dateAsString);
|
||||
break;
|
||||
} catch (Exception e) {
|
||||
log.warn("Failed to parse date from string {}. \n{}", dateAsString, e.getMessage());
|
||||
// ignore, try next...
|
||||
}
|
||||
}
|
||||
if (date == null) {
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
return Optional.of(date);
|
||||
}
|
||||
|
||||
|
||||
public String convertDate(Date date, String resultFormat) {
|
||||
|
||||
DateFormat resultDateFormat = new SimpleDateFormat(resultFormat, Locale.ENGLISH);
|
||||
|
||||
return resultDateFormat.format(date);
|
||||
}
|
||||
|
||||
|
||||
private DateTimeFormatter getDateTimeFormatter() {
|
||||
|
||||
if (DATE_TIME_FORMATTER == null) {
|
||||
DATE_TIME_FORMATTER = createFormatterFromResource();
|
||||
}
|
||||
return DATE_TIME_FORMATTER;
|
||||
}
|
||||
|
||||
|
||||
private DateTimeFormatter createFormatterFromResource() {
|
||||
|
||||
DateTimeFormatterBuilder builder = new DateTimeFormatterBuilder();
|
||||
try (BufferedReader reader = new BufferedReader(new InputStreamReader(Objects.requireNonNull(DateConverter.class.getResourceAsStream("/date_formats.txt"))))) {
|
||||
String line;
|
||||
while ((line = reader.readLine()) != null) {
|
||||
builder.appendOptional(DateTimeFormatter.ofPattern(line.trim(), Locale.ENGLISH));
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new RuntimeException("Error reading date format file: " + e.getMessage());
|
||||
}
|
||||
return builder.toFormatter().withResolverStyle(ResolverStyle.SMART).withLocale(Locale.ENGLISH);
|
||||
}
|
||||
|
||||
|
||||
private String removeTrailingDot(String dateAsString) {
|
||||
|
||||
String str = dateAsString;
|
||||
if (str != null && !str.isEmpty() && str.charAt(str.length() - 1) == '.') {
|
||||
str = str.substring(0, str.length() - 1);
|
||||
}
|
||||
|
||||
return str;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+2
-6
@@ -21,9 +21,7 @@ public final class IdBuilder {
|
||||
|
||||
public String buildId(Set<Page> pages, List<Rectangle2D> rectanglesPerLine, String type, String entityType) {
|
||||
|
||||
return buildId(pages.stream()
|
||||
.map(Page::getNumber)
|
||||
.collect(Collectors.toList()), rectanglesPerLine, type, entityType);
|
||||
return buildId(pages.stream().map(Page::getNumber).collect(Collectors.toList()), rectanglesPerLine, type, entityType);
|
||||
}
|
||||
|
||||
|
||||
@@ -31,9 +29,7 @@ public final class IdBuilder {
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append(type).append(entityType);
|
||||
List<Integer> sortedPageNumbers = pageNumbers.stream()
|
||||
.sorted(Comparator.comparingInt(Integer::intValue))
|
||||
.toList();
|
||||
List<Integer> sortedPageNumbers = pageNumbers.stream().sorted(Comparator.comparingInt(Integer::intValue)).toList();
|
||||
sortedPageNumbers.forEach(sb::append);
|
||||
rectanglesPerLine.forEach(rectangle2D -> sb.append(Math.round(rectangle2D.getX()))
|
||||
.append(Math.round(rectangle2D.getY()))
|
||||
|
||||
+8
-18
@@ -22,25 +22,19 @@ public class RectangleTransformations {
|
||||
|
||||
public static Rectangle2D atomicTextBlockBBox(List<AtomicTextBlock> atomicTextBlocks) {
|
||||
|
||||
return atomicTextBlocks.stream()
|
||||
.flatMap(atomicTextBlock -> atomicTextBlock.getPositions()
|
||||
.stream())
|
||||
.collect(new Rectangle2DBBoxCollector());
|
||||
return atomicTextBlocks.stream().flatMap(atomicTextBlock -> atomicTextBlock.getPositions().stream()).collect(new Rectangle2DBBoxCollector());
|
||||
}
|
||||
|
||||
|
||||
public static Rectangle2D rectangleBBox(List<Position> positions) {
|
||||
|
||||
return positions.stream()
|
||||
.map(Position::toRectangle2D)
|
||||
.collect(new Rectangle2DBBoxCollector());
|
||||
return positions.stream().map(Position::toRectangle2D).collect(new Rectangle2DBBoxCollector());
|
||||
}
|
||||
|
||||
|
||||
public static Rectangle2D rectangle2DBBox(List<Rectangle2D> rectangle2DList) {
|
||||
|
||||
return rectangle2DList.stream()
|
||||
.collect(new Rectangle2DBBoxCollector());
|
||||
return rectangle2DList.stream().collect(new Rectangle2DBBoxCollector());
|
||||
}
|
||||
|
||||
|
||||
@@ -55,9 +49,7 @@ public class RectangleTransformations {
|
||||
if (rectangle2DList.isEmpty()) {
|
||||
return Collections.emptyList();
|
||||
}
|
||||
double splitThreshold = rectangle2DList.stream()
|
||||
.mapToDouble(RectangularShape::getWidth).average()
|
||||
.orElse(5) * 5.0;
|
||||
double splitThreshold = rectangle2DList.stream().mapToDouble(RectangularShape::getWidth).average().orElse(5) * 5.0;
|
||||
|
||||
List<List<Rectangle2D>> rectangleListsWithGaps = new LinkedList<>();
|
||||
List<Rectangle2D> rectangleListWithoutGaps = new LinkedList<>();
|
||||
@@ -74,9 +66,7 @@ public class RectangleTransformations {
|
||||
previousRectangle = currentRectangle;
|
||||
}
|
||||
}
|
||||
return rectangleListsWithGaps.stream()
|
||||
.map(RectangleTransformations::rectangle2DBBox)
|
||||
.toList();
|
||||
return rectangleListsWithGaps.stream().map(RectangleTransformations::rectangle2DBBox).toList();
|
||||
}
|
||||
|
||||
|
||||
@@ -106,9 +96,9 @@ public class RectangleTransformations {
|
||||
public BinaryOperator<BBox> combiner() {
|
||||
|
||||
return (b1, b2) -> new BBox(Math.min(b1.lowerLeftX, b2.lowerLeftX),
|
||||
Math.min(b1.lowerLeftY, b2.lowerLeftY),
|
||||
Math.max(b1.upperRightX, b2.upperRightX),
|
||||
Math.max(b1.upperRightY, b2.upperRightY));
|
||||
Math.min(b1.lowerLeftY, b2.lowerLeftY),
|
||||
Math.max(b1.upperRightX, b2.upperRightX),
|
||||
Math.max(b1.upperRightY, b2.upperRightY));
|
||||
}
|
||||
|
||||
|
||||
|
||||
+3
-147
@@ -18,13 +18,6 @@ import lombok.experimental.UtilityClass;
|
||||
@UtilityClass
|
||||
public class RedactionSearchUtility {
|
||||
|
||||
/**
|
||||
* Checks if any part of a CharSequence matches a given regex pattern.
|
||||
*
|
||||
* @param charSequence The CharSequence to be searched.
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @return true if any part of the CharSequence matches the regex pattern.
|
||||
*/
|
||||
public static boolean anyMatch(CharSequence charSequence, String regexPattern) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
@@ -32,13 +25,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if any part of a CharSequence matches a given regex pattern, case-insensitive.
|
||||
*
|
||||
* @param charSequence The CharSequence to be searched.
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @return true if any part of the CharSequence matches the regex pattern.
|
||||
*/
|
||||
public static boolean anyMatchIgnoreCase(CharSequence charSequence, String regexPattern) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, true);
|
||||
@@ -46,53 +32,24 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if the entirety of a CharSequence exactly matches a given regex pattern.
|
||||
*
|
||||
* @param charSequence The CharSequence to be matched.
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @return true if the CharSequence exactly matches the regex pattern.
|
||||
*/
|
||||
public static boolean exactMatch(CharSequence charSequence, String regexPattern) {
|
||||
|
||||
return charSequence.toString().matches(regexPattern);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if any part of a TextBlock matches a given regex pattern, case-insensitive.
|
||||
*
|
||||
* @param textBlock The TextBlock to be searched.
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @return true if any part of the TextBlock matches the regex pattern.
|
||||
*/
|
||||
public static boolean anyMatchIgnoreCase(TextBlock textBlock, String regexPattern) {
|
||||
|
||||
return anyMatchIgnoreCase(textBlock.getSearchText(), regexPattern);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Checks if any part of a TextBlock matches a given regex pattern.
|
||||
*
|
||||
* @param textBlock The TextBlock to be searched.
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @return true if any part of the TextBlock matches the regex pattern.
|
||||
*/
|
||||
public static boolean anyMatch(TextBlock textBlock, String regexPattern) {
|
||||
|
||||
return anyMatch(textBlock.getSearchText(), regexPattern);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds the first TextRange in a given CharSequence that matches a regex pattern.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @param searchText The CharSequence to be searched.
|
||||
* @return The first TextRange that matches the pattern.
|
||||
* @throws IllegalArgumentException If no match is found.
|
||||
*/
|
||||
public static TextRange findFirstTextRange(String regexPattern, CharSequence searchText) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
@@ -104,13 +61,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Expands the end boundary of a TextEntity based on a subsequent regex match.
|
||||
*
|
||||
* @param entity The entity to expand.
|
||||
* @param regexPattern The regex pattern used for expansion.
|
||||
* @return The new end boundary index.
|
||||
*/
|
||||
public static int getExpandedEndByRegex(TextEntity entity, String regexPattern) {
|
||||
|
||||
int expandedEnd;
|
||||
@@ -124,13 +74,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Expands the start boundary of a TextEntity based on a subsequent regex match.
|
||||
*
|
||||
* @param entity The entity to expand.
|
||||
* @param regexPattern The regex pattern used for expansion.
|
||||
* @return The new end boundary index.
|
||||
*/
|
||||
public static int getExpandedStartByRegex(TextEntity entity, String regexPattern) {
|
||||
|
||||
int expandedStart;
|
||||
@@ -144,20 +87,9 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Identifies all lines within a text block that fall within a specified vertical range.
|
||||
*
|
||||
* @param maxY The maximum Y-coordinate of the vertical range.
|
||||
* @param minY The minimum Y-coordinate of the vertical range.
|
||||
* @param textBlock The text block containing the lines to be checked.
|
||||
* @return A {@link TextRange} encompassing all lines within the specified Y-coordinate range.
|
||||
*/
|
||||
public static TextRange findTextRangesOfAllLinesInYRange(double maxY, double minY, TextBlock textBlock) {
|
||||
|
||||
List<TextRange> lineBoundaries = IntStream.range(0, textBlock.numberOfLines()).boxed()
|
||||
.map(textBlock::getLineTextRange)
|
||||
.filter(lineBoundary -> isWithinYRange(maxY, minY, textBlock, lineBoundary))
|
||||
.toList();
|
||||
List<TextRange> lineBoundaries = IntStream.range(0, textBlock.numberOfLines()).boxed().map(textBlock::getLineTextRange).filter(lineBoundary -> isWithinYRange(maxY, minY, textBlock, lineBoundary)).toList();
|
||||
if (lineBoundaries.isEmpty()) {
|
||||
return new TextRange(textBlock.getTextRange().start(), textBlock.getTextRange().start());
|
||||
}
|
||||
@@ -172,13 +104,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds TextRanges matching a regex pattern within a TextBlock.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @param textBlock The TextBlock to search within.
|
||||
* @return A list of TextRanges corresponding to regex matches.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegex(String regexPattern, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
@@ -187,14 +112,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds TextRanges matching a regex pattern within a TextBlock capturing groups.
|
||||
*
|
||||
* @param regexPattern The regex pattern to match against.
|
||||
* @param group The group to capture within the regex pattern.
|
||||
* @param textBlock The TextBlock to search within.
|
||||
* @return A list of TextRanges corresponding to regex matches.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegex(String regexPattern, int group, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
@@ -202,14 +119,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds text ranges that match a regex pattern with consideration for line breaks within a text block.
|
||||
*
|
||||
* @param regexPattern The regex pattern to search for, allowing for multiline matches.
|
||||
* @param group The regex pattern group to extract from matches.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects corresponding to the matches found.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegexWithLineBreaks(String regexPattern, int group, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledMultilinePattern(regexPattern, false);
|
||||
@@ -217,27 +126,12 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds text ranges within a text block that match a given regex pattern, case-insensitive.
|
||||
*
|
||||
* @param regexPattern The regex pattern to search for, with case-insensitive matching.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects corresponding to the matches found.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegexWithLineBreaksIgnoreCase(String regexPattern, int group, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledMultilinePattern(regexPattern, true);
|
||||
return getTextRangesByPatternWithLineBreaks(textBlock, group, pattern);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds text ranges within a text block that match a given regex pattern, and case-insensitive.
|
||||
*
|
||||
* @param regexPattern The regex pattern to search for.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects corresponding to the group matches found, with case-insensitive matching.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegexIgnoreCase(String regexPattern, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, true);
|
||||
@@ -245,14 +139,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds text ranges within a text block that match a given regex pattern, capturing a specific group, and case-insensitive.
|
||||
*
|
||||
* @param regexPattern The regex pattern to search for.
|
||||
* @param group The group within the regex pattern to capture.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects corresponding to the group matches found, with case-insensitive matching.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByRegexIgnoreCase(String regexPattern, int group, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Patterns.getCompiledPattern(regexPattern, true);
|
||||
@@ -283,13 +169,6 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds all occurrences of a specified string within a text block and returns their positions as text ranges.
|
||||
*
|
||||
* @param searchString The string to search for within the text block.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects representing the start and end positions of each occurrence of the search string.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByString(String searchString, TextBlock textBlock) {
|
||||
|
||||
List<TextRange> boundaries = new LinkedList<>();
|
||||
@@ -300,48 +179,25 @@ public class RedactionSearchUtility {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Finds all occurrences of a specified string within a text block, case-insensitive, and returns their positions as text ranges.
|
||||
*
|
||||
* @param searchString The string to search for within the text block, case-insensitively.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects representing the start and end positions of each occurrence of the search string, case-insensitive.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByStringIgnoreCase(String searchString, TextBlock textBlock) {
|
||||
|
||||
Pattern pattern = Pattern.compile(Pattern.quote(searchString), Pattern.CASE_INSENSITIVE);
|
||||
return getTextRangesByPattern(textBlock, 0, pattern);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Searches a text block for all occurrences of each string in a list and returns their positions as text ranges.
|
||||
*
|
||||
* @param searchList A list of strings to search for within the text block.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects representing the start and end positions of occurrences of each string in the list.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByList(List<String> searchList, TextBlock textBlock) {
|
||||
|
||||
List<TextRange> boundaries = new LinkedList<>();
|
||||
for (var searchString : searchList) {
|
||||
for (var searchString: searchList) {
|
||||
boundaries.addAll(findTextRangesByString(searchString, textBlock));
|
||||
}
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Searches a text block for all occurrences of each string in a list, case-insensitive, and returns their positions as text ranges.
|
||||
*
|
||||
* @param searchList A list of strings to search for within the text block, case-insensitively.
|
||||
* @param textBlock The text block to search within.
|
||||
* @return A list of {@link TextRange} objects representing the start and end positions of occurrences of each string in the list, case-insensitive.
|
||||
*/
|
||||
public static List<TextRange> findTextRangesByListIgnoreCase(List<String> searchList, TextBlock textBlock) {
|
||||
|
||||
List<TextRange> boundaries = new LinkedList<>();
|
||||
for (var searchString : searchList) {
|
||||
for (var searchString: searchList) {
|
||||
boundaries.addAll(findTextRangesByStringIgnoreCase(searchString, textBlock));
|
||||
}
|
||||
return boundaries;
|
||||
|
||||
+1
-2
@@ -20,8 +20,7 @@ public final class ResourceLoader {
|
||||
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath);
|
||||
}
|
||||
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
|
||||
return br.lines()
|
||||
.collect(Collectors.toSet());
|
||||
return br.lines().collect(Collectors.toSet());
|
||||
} catch (IOException e) {
|
||||
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath, e);
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user