Compare commits

...
Author SHA1 Message Date
deiflaender 872c384dc6 Use default color from configuration-service on unknown type 2020-07-28 12:38:59 +02:00
Dominique Eiflaender 2aca35e5a0 Pull request #15: Let Tables know its headlines
Merge in RED/redaction-service from DEV5 to master

* commit '88e1c5c58ea44dfab15f086ef58f23df897777d6':
  Let Tables know its headlines
2020-07-27 15:45:43 +02:00
deiflaender 88e1c5c58e Let Tables know its headlines 2020-07-27 15:33:25 +02:00
Dominique Eiflaender b7ee62f44d Pull request #14: RED-207: Match caseInsensitive dictionaries caseInSensitive
Merge in RED/redaction-service from RED-207 to master

* commit '135a715e22e6c2536b268db29161552cfd7a6c1c':
  Fixed style in EnityRedactionService
  Fixed wrong naming of caseInsensitive
  RED-207: Match caseInsensitive dictionaries caseInSensitive
2020-07-27 13:52:40 +02:00
deiflaender 135a715e22 Fixed style in EnityRedactionService 2020-07-27 13:39:31 +02:00
deiflaender c953f161b2 Fixed wrong naming of caseInsensitive 2020-07-27 13:38:13 +02:00
deiflaender f0e48087ff RED-207: Match caseInsensitive dictionaries caseInSensitive 2020-07-27 13:20:00 +02:00
Cheng Zhu d282680cc8 Pull request #13: Use Hint in IntegrationTest
Merge in RED/redaction-service from DEV3 to master

* commit 'b57a4a2db3fb8090004dd5f3babd0d7786312730':
  Use Hint in IntegrationTest
2020-07-24 11:03:58 +02:00
deiflaender b57a4a2db3 Use Hint in IntegrationTest 2020-07-24 10:45:04 +02:00
Thierry Goeckel ca439d821d Pull request #12: Bugfix/RED-183
Merge in RED/redaction-service from bugfix/RED-183 to master

* commit '7dbe03483b7a8a3c391e9d0791717b9fab5fc5db':
  RED-183: Fix catching validation errors
  Fix style.
2020-07-23 14:48:11 +02:00
Thierry Göckel 7dbe03483b RED-183: Fix catching validation errors 2020-07-23 13:36:02 +02:00
Thierry Göckel 33ab09d7fc Fix style. 2020-07-23 13:35:58 +02:00
Thierry Goeckel 9a425a8594 Pull request #11: Feature/RED-161: replace hard-coded hint-type with dictionary-service query
Merge in RED/redaction-service from feature/RED-161 to master

* commit '3eb37fa6db0d9f1a4e6b93b2047b3b96c94cd9ff':
  RED-161: reverse caseSensitive
  RED-161: replace hard-coded hint-type with dictionary-service query
2020-07-23 10:59:37 +02:00
cheng 3eb37fa6db RED-161: reverse caseSensitive 2020-07-23 10:51:30 +02:00
cheng c8375a1ad9 RED-161: replace hard-coded hint-type with dictionary-service query 2020-07-22 22:54:05 +02:00
Dominique Eiflaender 94827a417f Pull request #10: Do not return redacted=true for vertebrates in RedactionLog, return isHint=true and redacted=false
Merge in RED/redaction-service from RedactionLog1 to master

* commit 'e7ea12a1b03f7f79e0f4fa12671f91ee326eef73':
  CodeStyle
  Do not return redacted=true for vertebrates in RedactionLog, return isHint=true and redacted=false
2020-07-22 14:15:56 +02:00
deiflaender e7ea12a1b0 CodeStyle 2020-07-22 14:14:10 +02:00
deiflaender 6a9df397eb Do not return redacted=true for vertebrates in RedactionLog, return isHint=true and redacted=false 2020-07-22 14:06:45 +02:00
Cheng Zhu 41c61dd214 Pull request #9: DEV: Fixed Nullpointer on undefined color for type, adapted rule to Type naming.
Merge in RED/redaction-service from Fix1 to master

* commit '49fe418a60f424e373038b1f7b9d43ff4afde37c':
  DEV: Fixed Nullpointer on undefined color for type, adapted rule to Type naming.
2020-07-21 16:20:22 +02:00
deiflaender 49fe418a60 DEV: Fixed Nullpointer on undefined color for type, adapted rule to Type naming. 2020-07-21 16:17:52 +02:00
Dominique Eiflaender c98f1d5e37 Pull request #8: RED-169: Fixed startup problem
Merge in RED/redaction-service from RED-169 to master

* commit 'c0882346a9d1554c51ccc526d1f6d7c277ffbc7c':
  RED-169: Fixed startup problem
2020-07-21 14:53:55 +02:00
deiflaender c0882346a9 RED-169: Fixed startup problem 2020-07-21 14:49:00 +02:00
Dominique Eiflaender 302daff526 Pull request #6: RED-106: replace the local dictionary preload with remove dictionary service.
Merge in RED/redaction-service from feature/RED-106i to master

* commit 'd607ac567d1e07cfc4c7dd4fa581d1881a874537':
  RED-106: integrationTest fixed.
  RED-106: integrationTest fixed.
  RED-106: integrationTest fixed.
  REd-106: rebase auf master
  REd-106: rebase auf master
  DEV: bugfix missing bean
  REd-106: enable dictionary version
  RED-106: replace the local dictionary preload with remove dictionary service.
  RED-106: replace the local dictionary preload with remove dictionary service.
2020-07-21 13:19:27 +02:00
cheng d607ac567d RED-106: integrationTest fixed. 2020-07-21 13:15:58 +02:00
cheng 1b40931c87 RED-106: integrationTest fixed. 2020-07-21 13:14:17 +02:00
cheng 75c6433eef RED-106: integrationTest fixed. 2020-07-21 13:14:10 +02:00
cheng 9c14f331fd REd-106: rebase auf master 2020-07-20 10:13:35 +02:00
cheng 141e4b9371 REd-106: rebase auf master 2020-07-20 09:19:54 +02:00
cheng 902d5cddf7 DEV: bugfix missing bean 2020-07-20 09:18:42 +02:00
cheng b3841b86b1 REd-106: enable dictionary version 2020-07-20 09:18:42 +02:00
cheng 4e3a6af3f8 RED-106: replace the local dictionary preload with remove dictionary service. 2020-07-20 09:18:42 +02:00
cheng 8ed548f1a8 RED-106: replace the local dictionary preload with remove dictionary service. 2020-07-20 09:18:42 +02:00
Thierry Goeckel fe5c20e1a0 Pull request #7: Fix application startup and remove redundant code.
Merge in RED/redaction-service from bugfix/application-startup to master

* commit '3e18fe19f31081df98670130512ab3933f7015af':
  Fix application startup and remove redundant code.
2020-07-20 09:18:21 +02:00
Thierry Göckel 3e18fe19f3 Fix application startup and remove redundant code. 2020-07-20 09:14:25 +02:00
Thierry Goeckel 5e48450669 Pull request #4: RED-125: Section must know its headlines, RED-156: Return RedactionLog
Merge in RED/redaction-service from RED-125,RED-156 to master

* commit '4b818dba54eea8eb33d5507e024487d5be3ca996':
  DEV: Fixed build problem
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket
  RED-125: Section must know its headlines RED-156: Return RedactionLog
2020-07-17 15:16:48 +02:00
deiflaender 4b818dba54 DEV: Fixed build problem 2020-07-17 15:08:47 +02:00
Dominique Eiflaender aed59b4578 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:03:01 +02:00
Dominique Eiflaender 53d36f1579 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:02:46 +02:00
Dominique Eiflaender 4f88ef0662 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:02:34 +02:00
Dominique Eiflaender e447772a23 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:02:26 +02:00
Dominique Eiflaender d656a29352 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:02:19 +02:00
Dominique Eiflaender dd350befd1 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/segmentation/SectionsBuilderService.java online editiert mit Bitbucket 2020-07-17 15:02:08 +02:00
deiflaender 8389a92820 RED-125: Section must know its headlines
RED-156: Return RedactionLog
2020-07-17 14:16:53 +02:00
Cheng Zhu fa7693e88b Pull request #5: RED-155: Bump config service version to make sure, dictionaries and
Merge in RED/redaction-service from feature/RED-155 to master

* commit 'c80cae3fc361a651e120626daf26d39de1dc656f':
  No need to write or add rules to file in classpath.
  RED-155: Bump config service version to make sure, dictionaries and rules are pulled correctly.
2020-07-17 13:01:32 +02:00
Thierry Göckel c80cae3fc3 No need to write or add rules to file in classpath. 2020-07-17 12:55:49 +02:00
Thierry Göckel 4832555343 RED-155: Bump config service version to make sure, dictionaries and
rules are pulled correctly.
2020-07-17 12:40:49 +02:00
Thierry Goeckel 0ed8530cb5 Pull request #3: Load initial rules from configuration service.
Merge in RED/redaction-service from feature/load-initial-rules-from-configuration-service to master

* commit '01d08fb1913d748fe04fcea78b8a405a92bd1a49':
  Removed code is debugged code.
  Add real test.
  Remove unused import.
  Make testing possible again.
  Move update of rules out of controller.
  Load initial rules from configuration service.
2020-07-17 11:24:07 +02:00
Thierry Göckel 01d08fb191 Removed code is debugged code. 2020-07-17 11:13:31 +02:00
Thierry Göckel 705c499911 Add real test. 2020-07-16 15:57:22 +02:00
Thierry Göckel 7435c1eb87 Remove unused import. 2020-07-16 14:18:18 +02:00
Thierry Göckel 531e34c6d0 Make testing possible again. 2020-07-16 13:37:43 +02:00
Thierry Göckel cc0d585c0b Move update of rules out of controller. 2020-07-16 10:21:23 +02:00
Thierry Göckel 74e5bc0635 Load initial rules from configuration service. 2020-07-14 17:12:27 +02:00
Cheng Zhu fbeaebab7d Pull request #2: DEV: Update rules on each redaction request.
Merge in RED/redaction-service from dev/update-rules-on-redaction-requests to master

* commit 'bf89e42fcf597033a17acc63919f62aed6640285':
  DEV: Update rules on each redaction request.
2020-07-09 22:50:14 +02:00
Thierry Göckel bf89e42fcf DEV: Update rules on each redaction request. 2020-07-09 13:47:31 +02:00
Thierry Goeckel a3d471e940 Pull request #1: DEV: Handle rules validation exception when updating.
Merge in RED/redaction-service from dev/rules-validation to master

* commit '23741deff713d5c865c13aa3a8cac73795d969e8':
  DEV: Use appropriate HTTP request type.
  DEV: Handle rules validation exception when updating.
2020-07-09 10:25:26 +02:00
Thierry Göckel 23741deff7 DEV: Use appropriate HTTP request type. 2020-07-09 10:06:17 +02:00
36 changed files with 890 additions and 361 deletions
@@ -0,0 +1,15 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class Point {
private float x;
private float y;
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class Rectangle {
private Point topLeft;
private float width;
private float height;
private int page;
}
@@ -0,0 +1,16 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class RedactionLog {
private List<RedactionLogEntry> redactionLogEntry;
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.Data;
@Data
public class RedactionLogEntry {
private String id;
private String type;
private String value;
private String reason;
private boolean redacted;
private boolean isHint;
private String section;
private float[] color;
private List<Rectangle> positions = new ArrayList<>();
}
@@ -13,4 +13,6 @@ public class RedactionResult {
private byte[] document;
private int numberOfPages;
private RedactionLog redactionLog;
}
@@ -23,9 +23,6 @@ public interface RedactionResource {
@PostMapping(value = "/debug/htmlTables", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/rules", produces = MediaType.APPLICATION_JSON_VALUE)
String getRules();
@PostMapping(value = "/rules/update", consumes = MediaType.APPLICATION_JSON_VALUE)
void updateRules(@RequestBody String rules);
@@ -31,6 +31,16 @@
</dependencyManagement>
<dependencies>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>redaction-service-api-v1</artifactId>
<version>${project.version}</version>
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>configuration-service-api-v1</artifactId>
<version>1.0.12</version>
</dependency>
<dependency>
<groupId>org.drools</groupId>
<artifactId>drools-core</artifactId>
@@ -46,11 +56,6 @@
<artifactId>jts-core</artifactId>
<version>1.16.1</version>
</dependency>
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>redaction-service-api-v1</artifactId>
<version>${project.version}</version>
</dependency>
<!-- commons -->
<dependency>
<groupId>com.iqser.gin4.commons</groupId>
@@ -1,45 +1,69 @@
package com.iqser.red.service.redaction.v1.server;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.kie.internal.io.ResourceFactory;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
import org.springframework.boot.context.properties.EnableConfigurationProperties;
import org.springframework.cloud.openfeign.EnableFeignClients;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Import;
import com.iqser.gin4.commons.spring.DefaultWebMvcConfiguration;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
@Import({DefaultWebMvcConfiguration.class})
@EnableFeignClients(basePackageClasses = RulesClient.class)
@EnableConfigurationProperties(RedactionServiceSettings.class)
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
public class Application {
@Autowired
private RulesClient rulesClient;
public static void main(String[] args) {
SpringApplication.run(Application.class, args);
}
private static final String drlFile = "drools/rules.drl";
@Bean
public KieContainer kieContainer() {
KieServices kieServices = KieServices.Factory.get();
try {
KieServices kieServices = KieServices.Factory.get();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
kieFileSystem.write(ResourceFactory.newClassPathResource(drlFile));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
RulesResponse rules = rulesClient.getRules();
if (StringUtils.isEmpty(rules.getRules())) {
throw new RuntimeException("Rules cannot be empty.");
}
InputStream input = new ByteArrayInputStream(rules.getRules().getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/main/resources/drools/rules.drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
return kieServices.newKieContainer(kieModule.getReleaseId());
return kieServices.newKieContainer(kieModule.getReleaseId());
} catch (Exception e) {
throw new RulesValidationException("Could not update rules: " + e.getMessage(), e);
}
}
@@ -6,6 +6,7 @@ import java.util.List;
import java.util.Map;
import java.util.Set;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import lombok.Data;
@@ -23,4 +24,6 @@ public class Document {
private StringFrequencyCounter fontCounter= new StringFrequencyCounter();
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
private boolean headlines;
private List<RedactionLogEntry> redactionLogEntities = new ArrayList<>();
}
@@ -16,6 +16,7 @@ import lombok.NoArgsConstructor;
public class Paragraph {
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
private String headline;
public SearchableText getSearchableText(){
SearchableText searchableText = new SearchableText();
@@ -0,0 +1,10 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.configuration.v1.api.resource.DictionaryResource;
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
@FeignClient(name = "DictionaryResource", url = "http://" + RulesResource.SERVICE_NAME + ":8080")
public interface DictionaryClient extends DictionaryResource {
}
@@ -0,0 +1,9 @@
package com.iqser.red.service.redaction.v1.server.client;
import org.springframework.cloud.openfeign.FeignClient;
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
@FeignClient(name = RulesResource.SERVICE_NAME, url = "http://" + RulesResource.SERVICE_NAME + ":8080")
public interface RulesClient extends RulesResource {
}
@@ -8,6 +8,7 @@ import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
@@ -24,9 +25,7 @@ import com.iqser.red.service.redaction.v1.server.visualization.service.PdfFlatte
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@RestController
@RequiredArgsConstructor
public class RedactionController implements RedactionResource {
@@ -38,7 +37,7 @@ public class RedactionController implements RedactionResource {
private final PdfFlattenService pdfFlattenService;
private final DroolsExecutionService droolsExecutionService;
@Override
public RedactionResult redact(@RequestBody RedactionRequest redactionRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
@@ -50,17 +49,18 @@ public class RedactionController implements RedactionResource {
if (redactionRequest.isFlatRedaction()) {
PDDocument flatDocument = pdfFlattenService.flattenPDF(pdDocument);
return convert(flatDocument, classifiedDoc.getPages().size());
return convert(flatDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
}
return convert(pdDocument, classifiedDoc.getPages().size());
return convert(pdDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult classify(@RequestBody RedactionRequest pdfSegmentationRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(pdfSegmentationRequest.getDocument()))) {
@@ -74,9 +74,10 @@ public class RedactionController implements RedactionResource {
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
@@ -90,10 +91,12 @@ public class RedactionController implements RedactionResource {
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
pdDocument.setAllSecurityToBeRemoved(true);
@@ -114,24 +117,30 @@ public class RedactionController implements RedactionResource {
} catch (IOException e) {
throw new RedactionException(e);
}
}
public String getRules() {
return droolsExecutionService.getRules();
}
@Override
public void updateRules(@RequestBody String rules) {
droolsExecutionService.updateRules(rules);
}
private RedactionResult convert(PDDocument document, int numberOfPages) throws IOException {
return convert(document, numberOfPages, null);
}
private RedactionResult convert(PDDocument document, int numberOfPages, RedactionLog redactionLog) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
return RedactionResult.builder()
.document(byteArrayOutputStream.toByteArray())
.numberOfPages(numberOfPages)
.redactionLog(redactionLog)
.build();
}
}
}
}
@@ -39,6 +39,12 @@ public class TextPositionSequence implements CharSequence {
return text.charAt(0);
}
public char charAt(int index, boolean caseInSensitive) {
TextPosition textPosition = textPositionAt(index);
String text = textPosition.getUnicode();
return caseInSensitive ? text.toLowerCase().charAt(0) : text.charAt(0);
}
@Override
public TextPositionSequence subSequence(int start, int end) {
return new TextPositionSequence(textPositions.subList(start, end), page);
@@ -16,19 +16,24 @@ public class Entity {
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
private Integer start;
private Integer end;
private String headline;
private int matchedRule;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences) {
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule) {
this.word = word;
this.type = type;
this.redaction = redaction;
this.redactionReason = redactionReason;
this.positionSequences = positionSequences;
this.headline = headline;
this.matchedRule = matchedRule;
}
public Entity(String word, String type, Integer start, Integer end) {
public Entity(String word, String type, Integer start, Integer end, String headline) {
this.word = word;
this.type = type;
this.start = start;
this.end = end;
this.headline = headline;
}
}
@@ -13,45 +13,65 @@ public class SearchableText {
private List<TextPositionSequence> sequences = new ArrayList<>();
public void add(TextPositionSequence textPositionSequence) {
sequences.add(textPositionSequence);
}
public void addAll(List<TextPositionSequence> textPositionSequences) {
sequences.addAll(textPositionSequences);
}
public List<EntityPositionSequence> getSequences(String searchString) {
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive) {
char[] searchChars = searchString.replaceAll("\\n", " ").toCharArray();
String normalizedSearchString;
if (caseInsensitive) {
normalizedSearchString = searchString.toLowerCase();
} else {
normalizedSearchString = searchString;
}
char[] searchChars = normalizedSearchString.replaceAll("\\n", " ").toCharArray();
int counter = 0;
List<TextPositionSequence> crossSequenceParts = new ArrayList<>();
List<EntityPositionSequence> finalMatches = new ArrayList<>();
for (int i = 0; i < sequences.size(); i++) {
TextPositionSequence partMatch = new TextPositionSequence(sequences.get(i).getPage());
for (int j = 0; j < sequences.get(i).length(); j++) {
if(i > 0 && j == 0 && sequences.get(i).charAt(0) == ' ' && sequences.get(i - 1).charAt(sequences.get(i - 1).length() - 1) == ' '
|| j > 0 && sequences.get(i).charAt(j) == ' ' && sequences.get(i).charAt(j - 1) == ' '){
if(j == sequences.get(i).length() -1 && counter != 0 && !partMatch.getTextPositions().isEmpty()){
if (i > 0 && j == 0 && sequences.get(i).charAt(0, caseInsensitive) == ' ' && sequences.get(i - 1)
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && sequences.get(i)
.charAt(j, caseInsensitive) == ' ' && sequences.get(i).charAt(j - 1, caseInsensitive) == ' ') {
if (j == sequences.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
crossSequenceParts.add(partMatch);
}
continue;
}
if(j == 0 && sequences.get(i).charAt(j) != ' ' && i != 0 && sequences.get(i - 1).charAt(sequences.get(i - 1).length() -1) != ' ' && searchChars[counter] == ' '){
if (j == 0 && sequences.get(i).charAt(j, caseInsensitive) != ' ' && i != 0 && sequences.get(i - 1)
.charAt(sequences.get(i - 1)
.length() - 1, caseInsensitive) != ' ' && searchChars[counter] == ' ') {
counter++;
}
if (sequences.get(i).charAt(j) == searchChars[counter] || counter != 0 && sequences.get(i).charAt(j) == '-') {
if (sequences.get(i)
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && sequences.get(i)
.charAt(j, caseInsensitive) == '-') {
if(counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(sequences.get(i).charAt(j - 1)) || j == 0 && i != 0 && isSeparator(sequences.get(i - 1).charAt(sequences.get(i - 1).length() -1))
|| j == 0 && i != 0 && sequences.get(i - 1).charAt(sequences.get(i - 1).length() -1) != ' ' && sequences.get(i).charAt(j) != ' ') {
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(sequences.get(i)
.charAt(j - 1, caseInsensitive)) || j == 0 && i != 0 && isSeparator(sequences.get(i - 1)
.charAt(sequences.get(i - 1)
.length() - 1, caseInsensitive)) || j == 0 && i != 0 && sequences.get(i - 1)
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) != ' ' && sequences.get(i)
.charAt(j, caseInsensitive) != ' ') {
partMatch.add(sequences.get(i).textPositionAt(j));
if (!(j == sequences.get(i).length() -1 && sequences.get(i).charAt(j) == '-' && searchChars[counter] != '-')) {
if (!(j == sequences.get(i).length() - 1 && sequences.get(i)
.charAt(j, caseInsensitive) == '-' && searchChars[counter] != '-')) {
counter++;
}
}
@@ -59,10 +79,13 @@ public class SearchableText {
if (counter == searchString.length()) {
crossSequenceParts.add(partMatch);
if(i == sequences.size() - 1 && j == sequences.get(i).length() -1
|| j != sequences.get(i).length() -1 && isSeparator(sequences.get(i).charAt(j +1))
|| j == sequences.get(i).length() -1 && isSeparator(sequences.get(i + 1).charAt(0))
|| j == sequences.get(i).length() -1 && sequences.get(i).charAt(j) != ' ' && sequences.get(i + 1).charAt(0) != ' ') {
if (i == sequences.size() - 1 && j == sequences.get(i).length() - 1 || j != sequences.get(i)
.length() - 1 && isSeparator(sequences.get(i)
.charAt(j + 1, caseInsensitive)) || j == sequences.get(i)
.length() - 1 && isSeparator(sequences.get(i + 1)
.charAt(0, caseInsensitive)) || j == sequences.get(i).length() - 1 && sequences.get(i)
.charAt(j, caseInsensitive) != ' ' && sequences.get(i + 1)
.charAt(0, caseInsensitive) != ' ') {
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts));
}
@@ -72,14 +95,14 @@ public class SearchableText {
}
} else {
counter = 0;
if(!crossSequenceParts.isEmpty()){
if (!crossSequenceParts.isEmpty()) {
j--;
}
crossSequenceParts = new ArrayList<>();
partMatch = new TextPositionSequence(sequences.get(i).getPage());
}
if(j == sequences.get(i).length() -1 && counter != 0){
if (j == sequences.get(i).length() - 1 && counter != 0) {
crossSequenceParts.add(partMatch);
}
}
@@ -89,18 +112,18 @@ public class SearchableText {
}
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts){
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
UUID id = UUID.randomUUID();
List<EntityPositionSequence> result = new ArrayList<>();
int currentPage = -1;
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
for (TextPositionSequence textPositionSequence :crossSequenceParts){
if(currentPage == -1){
for (TextPositionSequence textPositionSequence : crossSequenceParts) {
if (currentPage == -1) {
currentPage = textPositionSequence.getPage();
entityPositionSequence.setPageNumber(currentPage);
entityPositionSequence.getSequences().add(textPositionSequence);
} else if(currentPage == textPositionSequence.getPage()){
} else if (currentPage == textPositionSequence.getPage()) {
entityPositionSequence.getSequences().add(textPositionSequence);
} else {
result.add(entityPositionSequence);
@@ -114,13 +137,14 @@ public class SearchableText {
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
@Override
public String toString() {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
@@ -137,10 +161,14 @@ public class SearchableText {
previous = word;
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString()).replaceAll("\n", " ").replaceAll(" ", " ");
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" ", " ");
}
public String getAsStringWithLinebreaks(){
public String getAsStringWithLinebreaks() {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
@@ -23,96 +23,95 @@ public class Section {
//This does not contain linebreaks and must always be used for correct offsets.
private String searchText;
private String headline;
public boolean contains(String type) {
return entities.stream().anyMatch(entity -> entity.getType().equals(type));
}
public void redact(String type, int ruleNumber, String reason){
public void redact(String type, int ruleNumber, String reason) {
entities.forEach(entity -> {
if(entity.getType().equals(type)){
if (entity.getType().equals(type)) {
entity.setRedaction(true);
entity.setRedactionReason("\nRule " + ruleNumber + " matched\n\n" +reason);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
}
public void redactNot(String type, int ruleNumber, String reason){
public void redactNot(String type, int ruleNumber, String reason) {
entities.forEach(entity -> {
if(entity.getType().equals(type)){
if (entity.getType().equals(type)) {
entity.setRedaction(false);
entity.setRedactionReason("\nRule " + ruleNumber + " matched\n\n" +reason);
}
});
}
public void highlightAll(String type){
entities.forEach(entity -> {
if(entity.getType().equals(type)){
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
}
public void redactLineAfter(String start, String asType, int ruleNumber, String reason){
public void redactLineAfter(String start, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(text, start, "\n");
if(value != null){
Set<Entity> found = findEntity(value.trim(), asType);
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
}
// TODO No need to iterate
entities.forEach(entity -> {
if(entity.getType().equals(asType)){
if (entity.getType().equals(asType)) {
entity.setRedaction(true);
entity.setRedactionReason("\nRule " + ruleNumber + " matched\n\n" +reason);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
}
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason){
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(searchText, start, stop);
if(value != null){
Set<Entity> found = findEntity(value.trim(), asType);
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
}
// TODO No need to iterate
entities.forEach(entity -> {
if(entity.getType().equals(asType)){
if (entity.getType().equals(asType)) {
entity.setRedaction(true);
entity.setRedactionReason("\nRule " + ruleNumber + " matched\n\n" +reason);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
}
private Set<Entity> findEntity(String value, String asType) {
Set<Entity> found = new HashSet<>();
int startIndex;
int stopIndex = 0;
do {
startIndex = searchText.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 &&
(startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText.charAt(startIndex - 1))) &&
(stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex));
}
} while (startIndex > -1);
int startIndex;
int stopIndex = 0;
do {
startIndex = searchText.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText
.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline));
}
} while (startIndex > -1);
removeEntitiesContainedInLarger(found);
@@ -121,14 +120,18 @@ public class Section {
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
public void removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
for (Entity inner : entities) {
if (inner.getWord().length() < word.getWord().length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
if (inner.getWord().length() < word.getWord()
.length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
wordsToRemove.add(inner);
}
}
@@ -1,58 +1,97 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import javax.annotation.PostConstruct;
import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import feign.FeignException;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@Slf4j
public class DictionaryService {
public static final String VERTEBRATES_CODE = "VERTEBRATE";
public static final String ADDRESS_CODE = "ADDRESS";
public static final String NAME_CODE = "NAME";
public static final String NO_REDACTION_INDICATOR = "NO_REDACTION_INDICATOR";
private final DictionaryClient dictionaryClient;
private long dictionaryVersion = -1;
@Getter
private Map<String, Set<String>> dictionary = new HashMap<>();
@Getter
private long generation;
private Map<String, float[]> entryColors = new HashMap<>();
@PostConstruct
public void init() {
loadFromResourceFiles();
}
@Getter
private List<String> hintTypes = new ArrayList<>();
@Getter
private List<String> caseInsensitiveTypes = new ArrayList<>();
@Getter
private float[] defaultColor;
public void updateDictionary() {
//TODO
long version = dictionaryClient.getVersion();
if (version > dictionaryVersion) {
dictionaryVersion = version;
updateDictionaryEntry();
}
}
public void loadFromResourceFiles() {
dictionary.computeIfAbsent(NAME_CODE, v -> new HashSet<>()).addAll(ResourceLoader.load("dictionaries/names.txt").stream().map(this::cleanDictionaryEntry).collect(Collectors.toList()));
dictionary.computeIfAbsent(VERTEBRATES_CODE, v -> new HashSet<>()).addAll(ResourceLoader.load("dictionaries/vertebrates.txt").stream().map(this::cleanDictionaryEntry).collect(Collectors.toList()));
dictionary.computeIfAbsent(ADDRESS_CODE, v -> new HashSet<>()).addAll(ResourceLoader.load("dictionaries/addresses.txt").stream().map(this::cleanDictionaryEntry).collect(Collectors.toList()));
dictionary.computeIfAbsent(NO_REDACTION_INDICATOR, v -> new HashSet<>()).addAll(ResourceLoader.load("dictionaries/NoRedactionIndicator.txt").stream().map(this::cleanDictionaryEntry).collect(Collectors.toList()));
private void updateDictionaryEntry() {
try {
TypeResponse typeResponse = dictionaryClient.getAllTypes();
if (typeResponse != null && CollectionUtils.isNotEmpty(typeResponse.getTypes())) {
entryColors = typeResponse.getTypes()
.stream()
.collect(Collectors.toMap(TypeResult::getType, TypeResult::getColor));
hintTypes = typeResponse.getTypes()
.stream()
.filter(TypeResult::isHint)
.map(TypeResult::getType)
.collect(Collectors.toList());
caseInsensitiveTypes = typeResponse.getTypes()
.stream()
.filter(TypeResult::isCaseInsensitive)
.map(TypeResult::getType)
.collect(Collectors.toList());
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, s -> convertEntries(s)));
defaultColor = dictionaryClient.getDefaultColor().getColor();
}
} catch (FeignException e) {
log.warn("Got some unknown feignException", e);
throw e;
}
}
private String cleanDictionaryEntry(String entry) {
return TextNormalizationUtilities.removeHyphenLineBreaks(entry).replaceAll("\\n", " ");
private Set<String> convertEntries(String s) {
if (caseInsensitiveTypes.contains(s)) {
return dictionaryClient.getDictionaryForType(s)
.getEntries()
.stream()
.map(String::toLowerCase)
.collect(Collectors.toSet());
}
return new HashSet<>(dictionaryClient.getDictionaryForType(s).getEntries());
}
}
}
@@ -4,8 +4,6 @@ import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import javax.annotation.PostConstruct;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
@@ -16,30 +14,43 @@ import org.kie.api.runtime.KieSession;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class DroolsExecutionService {
private final RulesClient rulesClient;
@Autowired
private KieContainer kieContainer;
private String currentDrlRules;
@PostConstruct
public void init() {
currentDrlRules = ResourceLoader.loadAsString("drools/rules.drl");
}
private long rulesVersion = -1;
public Section executeRules(Section section) {
KieSession kieSession = kieContainer.newKieSession();
kieSession.setGlobal("section", section);
kieSession.insert(section);
kieSession.fireAllRules();
kieSession.dispose();
return section;
}
public void updateRules() {
long version = rulesClient.getVersion();
if (version > rulesVersion) {
rulesVersion = version;
updateRules(rulesClient.getRules().getRules());
}
}
public void updateRules(String drlAsString) {
@@ -56,15 +67,10 @@ public class DroolsExecutionService {
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
kieContainer.updateToVersion(kieModule.getReleaseId());
currentDrlRules = drlAsString;
} catch (Exception e) {
throw new RulesValidationException("Could not update rules", e);
throw new RulesValidationException("Could not update rules: " + e.getMessage(), e);
}
}
public String getRules() {
return currentDrlRules;
}
}
@@ -31,6 +31,7 @@ public class EntityRedactionService {
public void processDocument(Document classifiedDoc) {
dictionaryService.updateDictionary();
droolsExecutionService.updateRules();
Set<Entity> documentEntities = new HashSet<>();
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
@@ -39,7 +40,6 @@ public class EntityRedactionService {
List<Table> tables = paragraph.getTables();
List<SearchableText> searchableRows = new ArrayList<>();
for (Table table : tables) {
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
@@ -51,70 +51,68 @@ public class EntityRedactionService {
searchableRow.addAll(textBlock.getSequences());
}
}
searchableRows.add(searchableRow);
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline());
Section analysedRowSection = droolsExecutionService.executeRules(Section.builder()
.entities(rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.headline(table.getHeadline())
.build());
for (Entity entity : analysedRowSection.getEntities()) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), false));
}
}
documentEntities.addAll(analysedRowSection.getEntities());
}
}
Set<Entity> entities = findEntities(searchableText);
Section analysedSection = droolsExecutionService.executeRules(Section
.builder()
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline());
Section analysedSection = droolsExecutionService.executeRules(Section.builder()
.entities(entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(paragraph.getHeadline())
.build());
for (Entity entity : analysedSection.getEntities()) {
entity.setPositionSequences(searchableText.getSequences(entity.getWord()));
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), false));
}
}
documentEntities.addAll(analysedSection.getEntities());
for (SearchableText searchableRow : searchableRows) {
Set<Entity> rowEntities = findEntities(searchableRow);
Section analysedRowSection = droolsExecutionService.executeRules(Section
.builder()
.entities(rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.build());
for (Entity entity : analysedRowSection.getEntities()) {
entity.setPositionSequences(searchableRow.getSequences(entity.getWord()));
}
documentEntities.addAll(analysedRowSection.getEntities());
}
}
documentEntities.forEach(entity -> {
entity.getPositionSequences().forEach(sequence -> {
classifiedDoc.getEntities().computeIfAbsent(sequence.getPageNumber(), (x) -> new HashSet<>()).add(
new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), List.of(sequence))
);
classifiedDoc.getEntities()
.computeIfAbsent(sequence.getPageNumber(), (x) -> new HashSet<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), List
.of(sequence), entity.getHeadline(), entity.getMatchedRule()));
});
});
}
private Set<Entity> findEntities(SearchableText searchableText) {
private Set<Entity> findEntities(SearchableText searchableText, String headline) {
String normalizedInputString = searchableText.toString();
String inputString = searchableText.toString();
String lowercaseInputString = inputString.toLowerCase();
Set<Entity> found = new HashSet<>();
for (Map.Entry<String, Set<String>> entry : dictionaryService.getDictionary().entrySet()) {
for (String value : entry.getValue()) {
int startIndex;
int stopIndex = 0;
do {
startIndex = normalizedInputString.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 &&
(startIndex == 0 || Character.isWhitespace(normalizedInputString.charAt(startIndex - 1)) || isSeparator(normalizedInputString.charAt(startIndex - 1))) &&
(stopIndex == normalizedInputString.length() || isSeparator(normalizedInputString.charAt(stopIndex)))) {
found.add(new Entity(normalizedInputString.substring(startIndex, stopIndex), entry.getKey(), startIndex, stopIndex));
}
} while (startIndex > -1);
if (dictionaryService.getCaseInsensitiveTypes().contains(entry.getKey())) {
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline));
} else {
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline));
}
}
@@ -123,16 +121,40 @@ public class EntityRedactionService {
return found;
}
private Set<Entity> find(String inputString, Set<String> values, String type, String headline) {
Set<Entity> found = new HashSet<>();
for (String value : values) {
int startIndex;
int stopIndex = 0;
do {
startIndex = inputString.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(inputString.charAt(startIndex - 1)) || isSeparator(inputString
.charAt(startIndex - 1))) && (stopIndex == inputString.length() || isSeparator(inputString.charAt(stopIndex)))) {
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline));
}
} while (startIndex > -1);
}
return found;
}
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
public void removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
for (Entity inner : entities) {
if (inner.getWord().length() < word.getWord().length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
if (inner.getWord().length() < word.getWord()
.length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
wordsToRemove.add(inner);
}
}
@@ -140,5 +162,4 @@ public class EntityRedactionService {
entities.removeAll(wordsToRemove);
}
}
@@ -2,7 +2,6 @@ package com.iqser.red.service.redaction.v1.server.redaction.utils;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
@@ -20,38 +19,12 @@ public class ResourceLoader {
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath);
}
try (InputStream is = resource.openStream();
InputStreamReader isr = new InputStreamReader(is, StandardCharsets.UTF_8);
BufferedReader br = new BufferedReader(isr)) {
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
return br.lines().collect(Collectors.toSet());
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath, e);
}
}
public String loadAsString(String classpathPath) {
URL resource = ResourceLoader.class.getClassLoader().getResource(classpathPath);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath);
}
try (InputStream is = resource.openStream();
InputStreamReader isr = new InputStreamReader(is, StandardCharsets.UTF_8);
BufferedReader br = new BufferedReader(isr)) {
StringBuffer sb = new StringBuffer();
String str;
while ((str = br.readLine()) != null) {
sb.append(str).append("\n");
}
return sb.toString();
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath, e);
}
}
}
}
@@ -17,18 +17,19 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@SuppressWarnings("all")
public class SectionsBuilderService {
public void buildSections(Document document) {
List<AbstractTextContainer> chunkWords = new ArrayList<>();
List<Paragraph> chunkBlockList1 = new ArrayList<>();
List<Paragraph> chunkBlockList = new ArrayList<>();
AbstractTextContainer prev = null;
String lastHeadline = "";
for (Page page : document.getPages()) {
for (AbstractTextContainer current : page.getTextBlocks()) {
if (current.getClassification() == null || current.getClassification().equals("Header") || current.getClassification().equals("Footer")) {
if (current.getClassification() == null || current.getClassification()
.equals("Header") || current.getClassification().equals("Footer")) {
continue;
}
@@ -36,8 +37,10 @@ public class SectionsBuilderService {
if (prev != null && current.getClassification().startsWith("H ") || !document.isHeadlines()) {
Paragraph cb1 = buildTextBlock(chunkWords);
chunkBlockList1.add(cb1);
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
lastHeadline = current.getText();
chunkBlockList.add(chunkBlock);
chunkWords = new ArrayList<>();
}
@@ -48,16 +51,17 @@ public class SectionsBuilderService {
}
}
Paragraph cb1 = buildTextBlock(chunkWords);
if (cb1 != null) {
chunkBlockList1.add(cb1);
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
if (chunkBlock != null) {
chunkBlockList.add(chunkBlock);
chunkBlock.setHeadline(lastHeadline);
}
document.setParagraphs(chunkBlockList1);
document.setParagraphs(chunkBlockList);
}
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList) {
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline) {
Paragraph paragraph = new Paragraph();
TextBlock textBlock = null;
@@ -66,17 +70,23 @@ public class SectionsBuilderService {
boolean splitByTable = false;
Iterator<AbstractTextContainer> itty = wordBlockList.iterator();
boolean alreadyAdded= false;
boolean alreadyAdded = false;
AbstractTextContainer previous = null;
while (itty.hasNext()) {
AbstractTextContainer container = itty.next();
if (container instanceof Table) {
splitByTable = true;
if (previous != null && previous instanceof TextBlock && previous.getText().startsWith("Table ")) {
((Table) container).setHeadline(previous.getText());
} else {
((Table) container).setHeadline("Table in: " + lastHeadline);
}
if (textBlock != null && !alreadyAdded) {
paragraph.getPageBlocks().add(textBlock);
alreadyAdded =true;
alreadyAdded = true;
}
paragraph.getPageBlocks().add(container);
continue;
@@ -85,24 +95,28 @@ public class SectionsBuilderService {
TextBlock wordBlock = (TextBlock) container;
if (textBlock == null) {
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock.getSequences(), wordBlock.getRotation());
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock
.getSequences(), wordBlock.getRotation());
textBlock.setPage(wordBlock.getPage());
} else if (splitByTable) {
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock.getSequences(), wordBlock.getRotation());
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock
.getSequences(), wordBlock.getRotation());
textBlock.setPage(wordBlock.getPage());
alreadyAdded = false;
} else if (pageBefore != -1 && wordBlock.getPage() != pageBefore) {
textBlock.setPage(pageBefore);
paragraph.getPageBlocks().add(textBlock);
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock.getSequences(), wordBlock.getRotation());
textBlock = new TextBlock(wordBlock.getMinX(), wordBlock.getMaxX(), wordBlock.getMinY(), wordBlock.getMaxY(), wordBlock
.getSequences(), wordBlock.getRotation());
textBlock.setPage(wordBlock.getPage());
} else {
TextBlock spatialEntity = textBlock.union(wordBlock);
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(),
spatialEntity.getWidth(), spatialEntity.getHeight());
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
.getHeight());
}
pageBefore = wordBlock.getPage();
splitByTable = false;
previous = container;
}
if (textBlock != null && !alreadyAdded) {
@@ -111,5 +125,4 @@ public class SectionsBuilderService {
return paragraph;
}
}
@@ -13,6 +13,7 @@ import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.Getter;
import lombok.Setter;
@SuppressWarnings("all")
public class Table extends AbstractTextContainer {
@@ -21,6 +22,10 @@ public class Table extends AbstractTextContainer {
private RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
@Getter
@Setter
private String headline;
@Getter
private int rowCount = 0;
@Getter
@@ -1,14 +1,10 @@
package com.iqser.red.service.redaction.v1.server.visualization.service;
import static com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService.ADDRESS_CODE;
import static com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService.NAME_CODE;
import static com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService.NO_REDACTION_INDICATOR;
import static com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService.VERTEBRATES_CODE;
import java.awt.Color;
import java.io.IOException;
import java.util.List;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
@@ -20,12 +16,16 @@ import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@@ -38,6 +38,8 @@ import lombok.extern.slf4j.Slf4j;
@RequiredArgsConstructor
public class AnnotationHighlightService {
private final DictionaryService dictionaryService;
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction) throws IOException {
@@ -77,6 +79,8 @@ public class AnnotationHighlightService {
for (Entity entity : classifiedDoc.getEntities().get(page)) {
RedactionLogEntry redactionLogEntry = new RedactionLogEntry();
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
if (flatRedaction && !isRedactionType(entity)) {
@@ -91,47 +95,54 @@ public class AnnotationHighlightService {
float posXEnd;
float posYInit;
float posYEnd;
float[] quadPoints;
if (textPositions.getTextPositions().get(0).getRotation() == 90) {
posXEnd = textPositions.getTextPositions().get(0).getYDirAdj() + 2;
posXInit = textPositions.getTextPositions().get(0).getYDirAdj() - height;
posYInit = textPositions.getTextPositions().get(0).getXDirAdj();
posYEnd = textPositions.getTextPositions().get(textPositions.getTextPositions().size() - 1).getXDirAdj() - height + 2;
quadPoints = new float[]{posXInit, posYInit, posXInit, posYEnd + height + 2, posXEnd, posYInit, posXEnd, posYEnd + height + 2};
posYEnd = textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getXDirAdj() - height + 4;
} else {
posXInit = textPositions.getTextPositions().get(0).getXDirAdj();
posXEnd = textPositions.getTextPositions().get(textPositions.getTextPositions().size() - 1).getXDirAdj() + textPositions.getTextPositions().get(textPositions.getTextPositions().size() - 1).getWidth() + 1;
posYInit = textPositions.getTextPositions().get(0).getPageHeight() - textPositions.getTextPositions().get(0).getYDirAdj();
posYEnd = textPositions.getTextPositions().get(0).getPageHeight() - textPositions.getTextPositions().get(textPositions.getTextPositions().size() - 1).getYDirAdj();
quadPoints = new float[]{posXInit, posYEnd + height + 2, posXEnd, posYEnd + height + 2, posXInit, posYInit - 2, posXEnd, posYEnd - 2};
posXEnd = textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getXDirAdj() + textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getWidth() + 1;
posYInit = textPositions.getTextPositions()
.get(0)
.getPageHeight() - textPositions.getTextPositions().get(0).getYDirAdj() - 2;
posYEnd = textPositions.getTextPositions()
.get(0)
.getPageHeight() - textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getYDirAdj() + 2;
}
Rectangle textHighlightRectangle = new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
List<PDAnnotation> annotations = pdPage.getAnnotations();
PDAnnotationTextMarkup highlight = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
highlight.constructAppearances();
PDRectangle position = new PDRectangle();
position.setLowerLeftX(posXInit);
position.setLowerLeftY(posYEnd);
position.setUpperRightX(posXEnd);
position.setUpperRightY(posYEnd + height);
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(posXInit);
annotationPosition.setLowerLeftY(posYEnd);
annotationPosition.setUpperRightX(posXEnd);
annotationPosition.setUpperRightY(posYEnd + height);
highlight.setRectangle(position);
if (!flatRedaction) {
highlight.setRectangle(annotationPosition);
if (!flatRedaction && !isHint(entity)) {
highlight.setAnnotationName(entityPositionSequence.getId().toString());
highlight.setTitlePopup(entityPositionSequence.getId().toString());
highlight.setContents(entity.getRedactionReason());
highlight.setContents("\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\n" + "In Section : \"" + entity
.getHeadline() + "\"");
}
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
highlight.setQuadPoints(quadPoints);
highlight.setQuadPoints(toQuadPoints(textHighlightRectangle));
PDColor color;
if (flatRedaction) {
@@ -142,47 +153,69 @@ public class AnnotationHighlightService {
highlight.setColor(color);
annotations.add(highlight);
redactionLogEntry.getPositions().add(textHighlightRectangle);
}
redactionLogEntry.setId(entityPositionSequence.getId().toString());
}
redactionLogEntry.setColor(getColor(entity));
redactionLogEntry.setReason(entity.getRedactionReason());
redactionLogEntry.setValue(entity.getWord());
redactionLogEntry.setType(entity.getType());
redactionLogEntry.setRedacted(entity.isRedaction());
redactionLogEntry.setSection(entity.getHeadline());
redactionLogEntry.setHint(isHint(entity));
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
}
private float[] toQuadPoints(Rectangle rectangle) {
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
return new float[]{rectangle.getTopLeft().getX(), rectangle.getTopLeft().getY(), rectangle.getTopLeft()
.getX() + rectangle.getWidth(), rectangle.getTopLeft().getY(), rectangle.getTopLeft().getX(), rectangle.getTopLeft()
.getY() + rectangle.getHeight(), rectangle.getTopLeft()
.getX() + rectangle.getWidth(), rectangle.getTopLeft().getY() + rectangle.getHeight()};
}
private boolean isRedactionType(Entity entity) {
if (!entity.isRedaction()) {
return false;
}
if (entity.getType().equals(ADDRESS_CODE)) {
return true;
if (isHint(entity)) {
return false;
}
if (entity.getType().equals(NAME_CODE)) {
return true;
}
return false;
return true;
}
private float[] getColor(Entity entity) {
if (!entity.isRedaction()) {
if (!entity.isRedaction() && !isHint(entity)) {
return new float[]{0.627f, 0.627f, 0.627f};
}
if (entity.getType().equals(VERTEBRATES_CODE)) {
return new float[]{0, 1, 0};
if (!dictionaryService.getEntryColors().containsKey(entity.getType())) {
return dictionaryService.getDefaultColor();
}
if (entity.getType().equals(ADDRESS_CODE)) {
return new float[]{0, 1, 1};
}
if (entity.getType().equals(NAME_CODE)) {
return new float[]{1, 1, 0};
}
if (entity.getType().equals(NO_REDACTION_INDICATOR)) {
return new float[]{1, 0.502f, 0};
}
return null;
return dictionaryService.getEntryColors().get(entity.getType());
}
private boolean isHint(Entity entity) {
List<String> hintTypes = dictionaryService.getHintTypes();
if (CollectionUtils.isNotEmpty(hintTypes) && hintTypes.contains(entity.getType())) {
return true;
}
return false;
}
private void visualizeTextBlock(TextBlock textBlock, PDPageContentStream contentStream) throws IOException {
@@ -208,13 +241,15 @@ public class AnnotationHighlightService {
private void visualizeTable(Table table, PDPageContentStream contentStream) throws IOException {
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (cell != null) {
contentStream.setLineWidth(0.5f);
contentStream.setStrokingColor(Color.CYAN);
contentStream.addRect((float) cell.getX(), (float) cell.getY(), (float) cell.getWidth(), (float) cell.getHeight());
contentStream.addRect((float) cell.getX(), (float) cell.getY(), (float) cell.getWidth(), (float) cell
.getHeight());
contentStream.stroke();
// contentStream.setStrokingColor(Color.GREEN);
@@ -239,4 +274,5 @@ public class AnnotationHighlightService {
contentStream.endText();
}
}
}
@@ -1,66 +0,0 @@
package drools
import com.iqser.red.service.redaction.v1.server.redaction.model.Section
global Section section
rule "0: Highlight Indicators"
when
eval(section.getEntities().isEmpty()==false);
then
section.highlightAll("VERTEBRATE");
section.highlightAll("NO_REDACTION_INDICATOR");
end
rule "1: Redacted because Section contains Vertebrate"
when
eval(section.contains("VERTEBRATE")==true);
then
section.redact("NAME", 1, "Redacted because Section contains Vertebrate");
section.redact("ADDRESS", 1, "Redacted because Section contains Vertebrate");
end
rule "2: Not Redacted because Section contains no Vertebrate"
when
eval(section.contains("VERTEBRATE")==false);
then
section.redactNot("NAME", 2, "Not Redacted because Section contains no Vertebrate");
section.redactNot("ADDRESS", 2, "Not Redacted because Section contains no Vertebrate");
end
rule "3: Do not redact Names and Addresses if no redaction Indicator is contained"
when
eval(section.contains("VERTEBRATE")==true && section.contains("NO_REDACTION_INDICATOR")==true);
then
section.redactNot("NAME", 3, "Vertebrate was found, but also a no redaction indicator");
section.redactNot("ADDRESS", 3, "Vertebrate was found, but also a no redaction indicator");
end
rule "4: Redact contact information, if applicant is found"
when
eval(section.getText().toLowerCase().contains("applicant"));
then
section.redactLineAfter("Name:", "ADDRESS", 4, "Redacted because of Rule 4");
section.redactBetween("Address:", "Contact", "ADDRESS", 4, "Redacted because of Rule 4");
section.redactLineAfter("Contact point:", "ADDRESS", 4, "Redacted because of Rule 4");
section.redactLineAfter("Phone:", "ADDRESS", 4, "Redacted because of Rule 4");
section.redactLineAfter("Fax:", "ADDRESS", 4, "Redacted because of Rule 4");
section.redactLineAfter("E-mail:", "ADDRESS", 4, "Redacted because of Rule 4");
end
rule "5: Redact contact information, if 'Producer of the plant protection product' is found"
when
eval(section.getText().contains("Producer of the plant protection product"));
then
section.redactLineAfter("Name:", "ADDRESS", 5, "xxxx");
section.redactBetween("Address:", "Contact", "ADDRESS", 5, "xxxx");
section.redactBetween("Contact:", "Phone", "ADDRESS", 5, "xxxx");
section.redactLineAfter("Phone:", "ADDRESS", 5, "xxxx");
section.redactLineAfter("Fax:", "ADDRESS", 5, "xxxx");
section.redactLineAfter("E-mail:", "ADDRESS", 5, "xxxx");
end
@@ -1,14 +0,0 @@
package com.iqser.red.service.redaction.v1.server;
import org.junit.Test;
/**
*
*/
public class DummyTest {
@Test
public void dummy(){
System.out.println("Hello World");
}
}
@@ -1,32 +1,192 @@
package com.iqser.red.service.redaction.v1.server;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.DEFINED_PORT;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
import org.junit.Before;
import org.junit.Ignore;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.test.context.SpringBootTest;
import org.springframework.boot.test.context.TestConfiguration;
import org.springframework.boot.test.mock.mockito.MockBean;
import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
@Ignore
@RunWith(SpringRunner.class)
@SpringBootTest(webEnvironment = DEFINED_PORT)
public class RedactionIntegrationTest {
private static final String RULES = loadFromClassPath("drools/rules.drl");
private static final String VERTEBRATES_CODE = "vertebrate";
private static final String ADDRESS_CODE = "address";
private static final String NAME_CODE = "name";
private static final String NO_REDACTION_INDICATOR = "no_redaction_indicator";
@Autowired
private RedactionController redactionController;
@MockBean
private RulesClient rulesClient;
@MockBean
private DictionaryClient dictionaryClient;
private final Map<String, List<String>> dictionary = new HashMap<>();
private final Map<String, float[]> typeColorMap = new HashMap<>();
private final Map<String, Boolean> hintTypeMap = new HashMap<>();
private final Map<String, Boolean> caseInSensitiveMap = new HashMap<>();
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@Bean
public KieContainer kieContainer() {
KieServices kieServices = KieServices.Factory.get();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
InputStream input = new ByteArrayInputStream(RULES.getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/test/resources/drools/rules.drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
return kieServices.newKieContainer(kieModule.getReleaseId());
}
}
@Before
public void stubRulesClient() {
when(rulesClient.getVersion()).thenReturn(0L);
when(rulesClient.getRules()).thenReturn(new RulesResponse(RULES));
loadDictionaryForTest();
loadTypeForTest();
when(dictionaryClient.getVersion()).thenReturn(0L);
when(dictionaryClient.getAllTypes()).thenReturn(TypeResponse.builder().types(getTypeResponse()).build());
when(dictionaryClient.getDictionaryForType(VERTEBRATES_CODE)).thenReturn(getDictionaryResponse(VERTEBRATES_CODE));
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(getDictionaryResponse(ADDRESS_CODE));
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(getDictionaryResponse(NAME_CODE));
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor(new float[]{1f, 0.502f, 0f}));
}
private void loadDictionaryForTest() {
dictionary.computeIfAbsent(NAME_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/names.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(VERTEBRATES_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/vertebrates.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(ADDRESS_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/addresses.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(NO_REDACTION_INDICATOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/NoRedactionIndicator.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
}
private String cleanDictionaryEntry(String entry) {
return TextNormalizationUtilities.removeHyphenLineBreaks(entry).replaceAll("\\n", " ");
}
private void loadTypeForTest() {
typeColorMap.put(VERTEBRATES_CODE, new float[]{0, 1, 0});
typeColorMap.put(ADDRESS_CODE, new float[]{0, 1, 1});
typeColorMap.put(NAME_CODE, new float[]{1, 1, 0});
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{1, 0.502f, 0});
hintTypeMap.put(VERTEBRATES_CODE, true);
hintTypeMap.put(ADDRESS_CODE, false);
hintTypeMap.put(NAME_CODE, false);
hintTypeMap.put(NO_REDACTION_INDICATOR, true);
caseInSensitiveMap.put(VERTEBRATES_CODE, true);
caseInSensitiveMap.put(ADDRESS_CODE, false);
caseInSensitiveMap.put(NAME_CODE, false);
caseInSensitiveMap.put(NO_REDACTION_INDICATOR, true);
}
private List<TypeResult> getTypeResponse() {
return typeColorMap.entrySet()
.stream()
.map(typeColor -> TypeResult.builder()
.type(typeColor.getKey())
.color(typeColor.getValue())
.isHint(hintTypeMap.get(typeColor.getKey()))
.isCaseInsensitive(caseInSensitiveMap.get(typeColor.getKey()))
.build())
.collect(Collectors.toList());
}
private DictionaryResponse getDictionaryResponse(String type) {
return DictionaryResponse.builder()
.color(typeColorMap.get(type))
.entries(dictionary.get(type))
.isHint(hintTypeMap.get(type))
.isCaseInsensitive(caseInSensitiveMap.get(type))
.build();
}
@Test
@@ -35,7 +195,9 @@ public class RedactionIntegrationTest {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder().document(IOUtils.toByteArray(pdfFileResource.getInputStream())).build();
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
@@ -55,7 +217,9 @@ public class RedactionIntegrationTest {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder().document(IOUtils.toByteArray(pdfFileResource.getInputStream())).build();
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.classify(request);
@@ -70,7 +234,9 @@ public class RedactionIntegrationTest {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder().document(IOUtils.toByteArray(pdfFileResource.getInputStream())).build();
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.sections(request);
@@ -79,12 +245,15 @@ public class RedactionIntegrationTest {
}
}
@Test
public void htmlTablesTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder().document(IOUtils.toByteArray(pdfFileResource.getInputStream())).build();
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.htmlTables(request);
@@ -93,12 +262,15 @@ public class RedactionIntegrationTest {
}
}
@Test
public void htmlTableRotationTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder().document(IOUtils.toByteArray(pdfFileResource.getInputStream())).build();
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.htmlTables(request);
@@ -107,4 +279,23 @@ public class RedactionIntegrationTest {
}
}
private static String loadFromClassPath(String path) {
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
StringBuilder sb = new StringBuilder();
String str;
while ((str = br.readLine()) != null) {
sb.append(str).append("\n");
}
return sb.toString();
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + path, e);
}
}
}
@@ -0,0 +1,64 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
import lombok.experimental.UtilityClass;
@UtilityClass
public class ResourceLoader {
public Map<String, String> loadDictionaryFiles() {
String name = "dictionaries/";
List<String> files;
try {
files = IOUtils.readLines(ResourceLoader.class.getClassLoader().getResourceAsStream(name), "UTF-8");
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + name, e);
}
return files.stream().collect(Collectors.toMap(ResourceLoader::getFileName, s -> name + s));
}
private String getFileName(String filePath) {
return filePath.substring(0, filePath.indexOf(".txt"));
}
public Set<String> load(String classpathPath) {
URL resource = ResourceLoader.class.getClassLoader().getResource(classpathPath);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath);
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
return br.lines().collect(Collectors.toSet());
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath, e);
}
}
public String loadToString(String classpathPath) {
URL resource = ResourceLoader.class.getClassLoader().getResource(classpathPath);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath);
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
return br.lines().collect(Collectors.joining("\n"));
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + classpathPath, e);
}
}
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import lombok.experimental.UtilityClass;
@UtilityClass
public class TextNormalizationUtilities {
/**
* Revert hyphenation due to line breaks.
* @param text Text to be processed.
* @return Text without line-break hyphenation.
*/
public static String removeHyphenLineBreaks(String text) {
return text.replaceAll("\\s(\\S+)[\\-\\u00AD]\\R|\n\r(.+ )", "\n$1$2");
}
}
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import org.assertj.core.api.Assertions;
import org.junit.Test;
public class TextNormalizationUtilitiesTest {
@Test
public void testHyphenRemoval() {
String test = "Without these peo-\nple, this conference would not happen";
Assertions.assertThat(TextNormalizationUtilities.removeHyphenLineBreaks(test))
.contains("\npeople");
test = "Die\t\nFreiwillige\t Versicherung\t endet\t zudem\t für\t den\t ein\u00AD\nzelnen\tVersicherten\tmit\tder\tAufhebung\tdes\tVertra-\nges,\t seiner\t Unterstellung\t unter\t die\t obligatorische\t\nVersicherung\t oder\t seinem\t Ausschluss.";
Assertions.assertThat(TextNormalizationUtilities.removeHyphenLineBreaks(test))
.contains("\neinzelnen", "\nVertrages");
}
}
@@ -100,15 +100,11 @@ Pseudacris triseriata
poecilia reticulata
poultry
quail
rabbit
rabbits
rainbow trout
Rana limnocharis
rana
limnocharis
rana pipiens
rat
rats
reptile
reptiles
ricefish
@@ -0,0 +1,58 @@
package drools
import com.iqser.red.service.redaction.v1.server.redaction.model.Section
global Section section
rule "1: Redacted because Section contains Vertebrate"
when
eval(section.contains("vertebrate")==true);
then
section.redact("name", 1, "Redacted because Section contains Vertebrate");
section.redact("address", 1, "Redacted because Section contains Vertebrate");
end
rule "2: Not Redacted because Section contains no Vertebrate"
when
eval(section.contains("vertebrate")==false);
then
section.redactNot("name", 2, "Not Redacted because Section contains no Vertebrate");
section.redactNot("address", 2, "Not Redacted because Section contains no Vertebrate");
end
rule "3: Do not redact Names and Addresses if no redaction Indicator is contained"
when
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true);
then
section.redactNot("name", 3, "Vertebrate was found, but also a no redaction indicator");
section.redactNot("address", 3, "Vertebrate was found, but also a no redaction indicator");
end
rule "4: Redact contact information, if applicant is found"
when
eval(section.getText().toLowerCase().contains("applicant"));
then
section.redactLineAfter("Name:", "address", 4, "Redacted because of Rule 4");
section.redactBetween("Address:", "Contact", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Contact point:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Phone:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Fax:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("E-mail:", "address", 4, "Redacted because of Rule 4");
end
rule "5: Redact contact information, if 'Producer of the plant protection product' is found"
when
eval(section.getText().contains("Producer of the plant protection product"));
then
section.redactLineAfter("Name:", "address", 5, "xxxx");
section.redactBetween("Address:", "Contact", "address", 5, "xxxx");
section.redactBetween("Contact:", "Phone", "address", 5, "xxxx");
section.redactLineAfter("Phone:", "address", 5, "xxxx");
section.redactLineAfter("Fax:", "address", 5, "xxxx");
section.redactLineAfter("E-mail:", "address", 5, "xxxx");
end