Compare commits

...
Author SHA1 Message Date
Thierry Goeckel 954765759c Pull request #23: Log warning message if tabular data mismatches
Merge in RED/redaction-service from RED-101-quickfix to master

* commit '17aabcd09c1f76bdd467f66aac0cf243c0625734':
  Fix index out of bounds exception
  Remove redundant warn message
  Fix NPE for empty cells
  Add test redacting all files and expecting no exception
  Log warning message if tabular data mismatches
2020-08-13 11:33:17 +02:00
Thierry Göckel 17aabcd09c Fix index out of bounds exception 2020-08-13 11:13:01 +02:00
Thierry Göckel 32aa500983 Remove redundant warn message 2020-08-13 11:04:56 +02:00
Thierry Göckel a151a13b4c Fix NPE for empty cells 2020-08-13 11:02:09 +02:00
Thierry Göckel 5542a97a38 Add test redacting all files and expecting no exception 2020-08-13 10:14:49 +02:00
Thierry Göckel 8edaa93bda Log warning message if tabular data mismatches 2020-08-13 10:07:53 +02:00
Dominique Eiflaender 98e0bf5606 Pull request #22: RED-242: Return sectionNumber for tests
Merge in RED/redaction-service from RED-242 to master

* commit '8344b7ccafcf98e7376c53a0a79708c51daa6d0a':
  RED-242: Return sectionNumber for tests
2020-08-12 15:51:50 +02:00
deiflaender 8344b7ccaf RED-242: Return sectionNumber for tests 2020-08-12 15:46:33 +02:00
Dominique Eiflaender 96ba93f774 Pull request #20: RED-101
Merge in RED/redaction-service from RED-101 to master

* commit 'c93ca745fc61fc2d7f1a1f474a4e3c464091e70d':
  Normalize header information
  Fix test and suppress checkstyle warnings
  Fix PMD errors
  RED-101: Detect vertebrate study row value
  RED-101: Implement table cell and row redaction
  Fix style
2020-08-11 13:00:39 +02:00
Thierry Göckel c93ca745fc Normalize header information 2020-08-11 10:24:33 +02:00
Thierry Göckel a6415363cd Fix test and suppress checkstyle warnings 2020-08-10 19:04:49 +02:00
Thierry Göckel d97a7b629a Fix PMD errors 2020-08-10 19:04:49 +02:00
Thierry Göckel 00c96c6f57 RED-101: Detect vertebrate study row value 2020-08-10 19:04:49 +02:00
Thierry Göckel 06630b09d2 RED-101: Implement table cell and row redaction 2020-08-10 19:04:49 +02:00
Thierry Göckel 695564d162 Fix style
Fix style.

Fix style.

Fix style and naming

Fix style, naming and field modifier

Fix style and remove warning suppression
2020-08-10 19:04:49 +02:00
Cheng Zhu 81048dcc9f Pull request #21: Cleaned up dictionaries
Merge in RED/redaction-service from cleanDictionaries to master

* commit 'b5412dc9590d15dcae9f87f7fc9b7ea50e4c63ae':
  Cleaned up dictionaries
2020-08-10 15:46:37 +02:00
deiflaender b5412dc959 Cleaned up dictionaries 2020-08-10 15:16:05 +02:00
Dominique Eiflaender 40e40a01ad Pull request #19: Avoid duplicate redaction if type have same entries, made Applicant and Producer rules more specific
Merge in RED/redaction-service from ApplicantRule to master

* commit '99bac4550a9cade36de313735916687ea26d7b4d':
  Use @EqualsAndHashCode(onlyExplicitlyIncluded = true) in Entity.java
  Avoid duplicate redaction if type have same entries, made Applicant and Producer rules more specific
2020-08-07 12:31:26 +02:00
deiflaender 99bac4550a Use @EqualsAndHashCode(onlyExplicitlyIncluded = true) in Entity.java 2020-08-07 12:28:21 +02:00
deiflaender 7d0b0ed3d0 Avoid duplicate redaction if type have same entries, made Applicant and Producer rules more specific 2020-08-07 12:09:37 +02:00
Cheng Zhu d465a4ba5b Pull request #18: RED-149: Added must_redact dictionary and Rule, Adjusted rules for applicant and producer to work on all documents.
Merge in RED/redaction-service from RED-149 to master

* commit 'cce8200d433ec89160af3af32f40be57c0b67678':
  redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java online editiert mit Bitbucket
  RED-149: Added must_redact dictionary and Rule, Adjusted rules for applicant and producer to work on all documents. Fixed endless loop in rules. Detect multiple occurences in rules
2020-08-05 13:21:14 +02:00
Dominique Eiflaender cce8200d43 redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java online editiert mit Bitbucket 2020-08-05 13:15:02 +02:00
deiflaender 1f8d371a82 RED-149: Added must_redact dictionary and Rule, Adjusted rules for applicant and producer to work on all documents. Fixed endless loop in rules. Detect multiple occurences in rules 2020-08-05 13:10:31 +02:00
Thierry Goeckel 70804f111d Pull request #17: Duplicates
Merge in RED/redaction-service from duplicates to master

* commit '81723ce4022e1e006054eb743f2cc2c0b9faf14f':
  Use void method type
  Use EqualsAndHashcode annotation from Lombok
  Fixed duplicated redaction/RedactionLog entries
  RED-211, RED-215 Added dictionaries and rules for testing.
2020-08-04 10:59:38 +02:00
Thierry Göckel 81723ce402 Use void method type 2020-08-04 10:34:54 +02:00
deiflaender b1a266d4d4 Use EqualsAndHashcode annotation from Lombok 2020-08-04 09:53:58 +02:00
deiflaender d2d7f8c50c Fixed duplicated redaction/RedactionLog entries 2020-07-31 16:25:10 +02:00
deiflaender e2895a1c7a RED-211, RED-215 Added dictionaries and rules for testing. 2020-07-31 16:22:47 +02:00
Lena  Maldacker cd07dc6a44 Pull request #16: Use default color from configuration-service on unknown type
Merge in RED/redaction-service from defaultColor to master

* commit '872c384dc6da60f421e7aa21f58b574c49414c81':
  Use default color from configuration-service on unknown type
2020-07-28 12:41:35 +02:00
29 changed files with 10040 additions and 3749 deletions
@@ -17,5 +17,6 @@ public class RedactionLogEntry {
private String section;
private float[] color;
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
}
@@ -4,7 +4,6 @@ import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
@@ -18,7 +17,7 @@ public class Document {
private List<Page> pages = new ArrayList<>();
private List<Paragraph> paragraphs = new ArrayList<>();
private Map<Integer, Set<Entity>> entities = new HashMap<>();
private Map<Integer, List<Entity>> entities = new HashMap<>();
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
private FloatFrequencyCounter fontSizeCounter= new FloatFrequencyCounter();
private StringFrequencyCounter fontCounter= new StringFrequencyCounter();
@@ -10,7 +10,6 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
public class Paragraph {
@@ -18,10 +17,12 @@ public class Paragraph {
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
private String headline;
public SearchableText getSearchableText(){
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
pageBlocks.forEach(block -> {
if(block instanceof TextBlock){
if (block instanceof TextBlock) {
searchableText.addAll(((TextBlock) block).getSequences());
}
});
@@ -29,14 +30,15 @@ public class Paragraph {
}
public List<Table> getTables(){
public List<Table> getTables() {
List<Table> tables = new ArrayList<>();
pageBlocks.forEach(block -> {
if(block instanceof Table){
if (block instanceof Table) {
tables.add((Table) block);
}
});
return tables;
}
}
}
@@ -5,43 +5,45 @@ import java.util.Map;
import lombok.Getter;
/**
*
*/
public class StringFrequencyCounter {
@Getter
Map<String, Integer> countPerValue = new HashMap<>();
private final Map<String, Integer> countPerValue = new HashMap<>();
public void add(String value){
if(!countPerValue.containsKey(value)){
public void add(String value) {
if (!countPerValue.containsKey(value)) {
countPerValue.put(value, 1);
} else {
countPerValue.put(value, countPerValue.get(value) + 1);
}
}
public void addAll(Map<String, Integer> otherCounter){
for(Map.Entry<String, Integer> entry: otherCounter.entrySet()){
if(countPerValue.containsKey(entry.getKey())){
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey())+ entry.getValue());
public void addAll(Map<String, Integer> otherCounter) {
for (Map.Entry<String, Integer> entry : otherCounter.entrySet()) {
if (countPerValue.containsKey(entry.getKey())) {
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
} else {
countPerValue.put(entry.getKey(), entry.getValue());
}
}
}
public String getMostPopular(){
public String getMostPopular() {
Map.Entry<String, Integer> mostPopular = null;
for(Map.Entry<String, Integer> entry: countPerValue.entrySet()){
if(mostPopular == null){
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
mostPopular = entry;
} else if(entry.getValue() > mostPopular.getValue()){
} else if (entry.getValue() > mostPopular.getValue()) {
mostPopular = entry;
}
}
return mostPopular != null ? mostPopular.getKey() : null;
}
}
}
@@ -29,20 +29,16 @@ public class BlockificationService {
float minX = 1000, maxX = 0, minY = 1000, maxY = 0;
TextPositionSequence prev = null;
for (TextPositionSequence word : textPositions) {
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
boolean startFromTop = word.getY1() > maxY + word.getHeight();
if (prev != null &&
(lineSeparation
|| startFromTop
|| word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines)
|| word.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines)
|| word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines)
|| word.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines)
)) {
if (prev != null && (lineSeparation || startFromTop || word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word
.getX1(), word.getY1(), verticalRulingLines) || word.getRotation() == 0 && isSplittedByRuling(minX, minY, word
.getX1(), word.getY2(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word
.getX1(), word.getY1(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(minX, minY, word
.getX1(), word.getY2(), verticalRulingLines))) {
TextBlock cb1 = buildTextBlock(chunkWords);
chunkBlockList1.add(cb1);
@@ -100,11 +96,12 @@ public class BlockificationService {
styleFrequencyCounter.add(wordBlock.getFontStyle());
if (textBlock == null) {
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock.getRotation());
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
.getRotation());
} else {
TextBlock spatialEntity = textBlock.union(wordBlock);
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(),
spatialEntity.getWidth(), spatialEntity.getHeight());
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
.getHeight());
}
}
@@ -122,6 +119,7 @@ public class BlockificationService {
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1, List<Ruling> rulingLines) {
for (Ruling ruling : rulingLines) {
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
return true;
@@ -133,7 +131,6 @@ public class BlockificationService {
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape) {
float minX = 10000;
float maxX = -100;
float minY = 10000;
@@ -147,7 +144,6 @@ public class BlockificationService {
for (AbstractTextContainer container : page.getTextBlocks()) {
if (container instanceof TextBlock) {
TextBlock textBlock = (TextBlock) container;
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
@@ -179,16 +175,15 @@ public class BlockificationService {
}
}
if (container instanceof Table) {
Table table = (Table) container;
for (List<Cell> row : table.getRows()) {
for (Cell column : row) {
for (Cell cell : row) {
if (column == null || column.getTextBlocks() == null) {
if (cell == null || cell.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : column.getTextBlocks()) {
for (TextBlock textBlock : cell.getTextBlocks()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
@@ -211,5 +206,4 @@ public class BlockificationService {
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
}
}
@@ -1,14 +1,16 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.ArrayList;
import java.util.List;
import lombok.Data;
import lombok.EqualsAndHashCode;
@Data
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
public class Entity {
@EqualsAndHashCode.Include
private final String word;
private final String type;
private boolean redaction;
@@ -16,10 +18,17 @@ public class Entity {
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
private Integer start;
private Integer end;
@EqualsAndHashCode.Include
private String headline;
private int matchedRule;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule) {
@EqualsAndHashCode.Include
private int sectionNumber;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber) {
this.word = word;
this.type = type;
this.redaction = redaction;
@@ -27,13 +36,18 @@ public class Entity {
this.positionSequences = positionSequences;
this.headline = headline;
this.matchedRule = matchedRule;
this.sectionNumber = sectionNumber;
}
public Entity(String word, String type, Integer start, Integer end, String headline) {
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber) {
this.word = word;
this.type = type;
this.start = start;
this.end = end;
this.headline = headline;
this.sectionNumber = sectionNumber;
}
}
@@ -6,15 +6,20 @@ import java.util.UUID;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.RequiredArgsConstructor;
@Data
@RequiredArgsConstructor
@AllArgsConstructor
@EqualsAndHashCode
public class EntityPositionSequence {
@EqualsAndHashCode.Exclude
private List<TextPositionSequence> sequences = new ArrayList<>();
private int pageNumber;
private final UUID id;
}
@@ -8,10 +8,9 @@ import java.util.regex.Pattern;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
@SuppressWarnings("all")
public class SearchableText {
private List<TextPositionSequence> sequences = new ArrayList<>();
private final List<TextPositionSequence> sequences = new ArrayList<>();
public void add(TextPositionSequence textPositionSequence) {
@@ -26,6 +25,7 @@ public class SearchableText {
}
@SuppressWarnings("checkstyle:ModifiedControlVariable")
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive) {
String normalizedSearchString;
@@ -163,7 +163,7 @@ public class SearchableText {
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" ", " ");
.replaceAll(" {2}", " ");
}
@@ -187,4 +187,4 @@ public class SearchableText {
return sb.append("\n").toString();
}
}
}
@@ -3,15 +3,19 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.StringUtils;
import lombok.Builder;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
@Data
@Slf4j
@Builder
public class Section {
@@ -25,6 +29,10 @@ public class Section {
private String headline;
private int sectionNumber;
private Map<String, String> tabularData;
public boolean contains(String type) {
@@ -32,6 +40,12 @@ public class Section {
}
public boolean headlineContainsWord(String word) {
return StringUtils.containsIgnoreCase(headline, word);
}
public void redact(String type, int ruleNumber, String reason) {
entities.forEach(entity -> {
@@ -58,11 +72,15 @@ public class Section {
public void redactLineAfter(String start, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(text, start, "\n");
String[] values = StringUtils.substringsBetween(text, start, "\n");
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType);
entities.addAll(found);
}
}
}
// TODO No need to iterate
@@ -79,11 +97,15 @@ public class Section {
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(searchText, start, stop);
String[] values = StringUtils.substringsBetween(searchText, start, stop);
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType);
entities.addAll(found);
}
}
}
// TODO No need to iterate
@@ -97,7 +119,7 @@ public class Section {
}
private Set<Entity> findEntity(String value, String asType) {
private Set<Entity> findEntities(String value, String asType) {
Set<Entity> found = new HashSet<>();
@@ -109,13 +131,11 @@ public class Section {
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText
.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline));
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline, sectionNumber));
}
} while (startIndex > -1);
removeEntitiesContainedInLarger(found);
return found;
return removeEntitiesContainedInLarger(found);
}
@@ -125,7 +145,7 @@ public class Section {
}
public void removeEntitiesContainedInLarger(Set<Entity> entities) {
public Set<Entity> removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
@@ -137,6 +157,28 @@ public class Section {
}
}
entities.removeAll(wordsToRemove);
return entities;
}
}
public void highlightCell(String cellHeader, int ruleNumber) {
String value = tabularData.get(cellHeader);
if (value == null) {
log.warn("Could not find any data for {}.", cellHeader);
} else {
Set<Entity> found = findEntities(value, "must_redact");
if (CollectionUtils.isEmpty(found)) {
log.warn("Could not identify value {} in row.", value);
} else {
Entity entity = found.iterator().next();
entity.setRedaction(false);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(cellHeader);
entities.add(entity);
}
}
}
}
@@ -73,7 +73,7 @@ public class DictionaryService {
.filter(TypeResult::isCaseInsensitive)
.map(TypeResult::getType)
.collect(Collectors.toList());
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, s -> convertEntries(s)));
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, this::convertEntries));
defaultColor = dictionaryClient.getDefaultColor().getColor();
}
} catch (FeignException e) {
@@ -1,25 +1,31 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.StringUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class EntityRedactionService {
@@ -34,6 +40,7 @@ public class EntityRedactionService {
droolsExecutionService.updateRules();
Set<Entity> documentEntities = new HashSet<>();
int sectionNumber = 1;
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
SearchableText searchableText = paragraph.getSearchableText();
@@ -41,88 +48,131 @@ public class EntityRedactionService {
List<Table> tables = paragraph.getTables();
for (Table table : tables) {
List<String> metadata = table.getHeaders();
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
for (Cell column : row) {
if (column == null || column.getTextBlocks() == null) {
List<String> cellValues = new ArrayList<>();
for (Cell cell : row) {
if (cell == null || CollectionUtils.isEmpty(cell.getTextBlocks())) {
cellValues.add(null);
continue;
}
for (TextBlock textBlock : column.getTextBlocks()) {
cellValues.add(cell.getTextBlocks().get(0).getText());
for (TextBlock textBlock : cell.getTextBlocks()) {
searchableRow.addAll(textBlock.getSequences());
}
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline());
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber);
Map<String, String> tabularData = toMap(metadata, cellValues);
Section analysedRowSection = droolsExecutionService.executeRules(Section.builder()
.entities(rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber)
.tabularData(tabularData)
.build());
for (Entity entity : analysedRowSection.getEntities()) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), false));
}
}
documentEntities.addAll(analysedRowSection.getEntities());
documentEntities.addAll(clearAndFindPositions(analysedRowSection.getEntities(), searchableRow));
sectionNumber++;
}
sectionNumber++;
}
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline());
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline(), sectionNumber);
Section analysedSection = droolsExecutionService.executeRules(Section.builder()
.entities(entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(paragraph.getHeadline())
.sectionNumber(sectionNumber)
.build());
for (Entity entity : analysedSection.getEntities()) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), false));
}
}
documentEntities.addAll(analysedSection.getEntities());
documentEntities.addAll(clearAndFindPositions(analysedSection.getEntities(), searchableText));
sectionNumber++;
}
documentEntities.forEach(entity -> {
entity.getPositionSequences().forEach(sequence -> {
for (Entity entity : documentEntities) {
Map<Integer, List<EntityPositionSequence>> sequenceOnPage = new HashMap<>();
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
sequenceOnPage.computeIfAbsent(entityPositionSequence.getPageNumber(), (x) -> new ArrayList<>())
.add(entityPositionSequence);
}
for (Map.Entry<Integer, List<EntityPositionSequence>> entry : sequenceOnPage.entrySet()) {
classifiedDoc.getEntities()
.computeIfAbsent(sequence.getPageNumber(), (x) -> new HashSet<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), List
.of(sequence), entity.getHeadline(), entity.getMatchedRule()));
});
});
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(),
entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber()));
}
}
}
private Set<Entity> findEntities(SearchableText searchableText, String headline) {
private Map<String, String> toMap(List<String> keys, List<String> values) {
if (keys.size() != values.size()) {
log.warn("Cannot merge lists of unequal size, returning empty map.");
return new HashMap<>();
}
Map<String, String> result = new HashMap<>();
for (int i = 0; i < keys.size(); i++) {
String value = values.get(i);
if (value == null) {
continue;
}
result.put(keys.get(i), value);
}
return result;
}
private Set<Entity> clearAndFindPositions(Set<Entity> entities, SearchableText text) {
removeEntitiesContainedInLarger(entities);
for (Entity entity : entities) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(text.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(text.getSequences(entity.getWord(), false));
}
}
return entities;
}
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber) {
Set<Entity> found = new HashSet<>();
if (StringUtils.isEmpty(searchableText.toString()) && StringUtils.isEmpty(headline)) {
return found;
}
String inputString = searchableText.toString();
String lowercaseInputString = inputString.toLowerCase();
Set<Entity> found = new HashSet<>();
for (Map.Entry<String, Set<String>> entry : dictionaryService.getDictionary().entrySet()) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entry.getKey())) {
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline));
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
} else {
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline));
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
}
}
removeEntitiesContainedInLarger(found);
return found;
}
private Set<Entity> find(String inputString, Set<String> values, String type, String headline) {
private Set<Entity> find(String inputString, Set<String> values, String type, String headline, int sectionNumber) {
Set<Entity> found = new HashSet<>();
for (String value : values) {
@@ -134,7 +184,8 @@ public class EntityRedactionService {
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(inputString.charAt(startIndex - 1)) || isSeparator(inputString
.charAt(startIndex - 1))) && (stopIndex == inputString.length() || isSeparator(inputString.charAt(stopIndex)))) {
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline));
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex,
headline, sectionNumber));
}
} while (startIndex > -1);
}
@@ -29,7 +29,6 @@ import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@SuppressWarnings("PMD")
public class PdfSegmentationService {
private final RulingCleaningService rulingCleaningService;
@@ -4,6 +4,8 @@ import java.util.ArrayList;
import java.util.Iterator;
import java.util.List;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.StringUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
@@ -14,7 +16,6 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractT
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@Service
@SuppressWarnings("all")
public class SectionsBuilderService {
public void buildSections(Document document) {
@@ -25,6 +26,7 @@ public class SectionsBuilderService {
AbstractTextContainer prev = null;
String lastHeadline = "";
Table previousTable = null;
for (Page page : document.getPages()) {
for (AbstractTextContainer current : page.getTextBlocks()) {
@@ -36,32 +38,30 @@ public class SectionsBuilderService {
current.setPage(page.getPageNumber());
if (prev != null && current.getClassification().startsWith("H ") || !document.isHeadlines()) {
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
chunkBlock.setHeadline(lastHeadline);
lastHeadline = current.getText();
if (CollectionUtils.isNotEmpty(chunkBlock.getTables())) {
previousTable = chunkBlock.getTables().get(0);
}
chunkBlockList.add(chunkBlock);
chunkWords = new ArrayList<>();
}
chunkWords.add(current);
prev = current;
}
}
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
if (chunkBlock != null) {
chunkBlockList.add(chunkBlock);
chunkBlock.setHeadline(lastHeadline);
}
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
chunkBlock.setHeadline(lastHeadline);
chunkBlockList.add(chunkBlock);
document.setParagraphs(chunkBlockList);
}
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline) {
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline, Table previousTable) {
Paragraph paragraph = new Paragraph();
TextBlock textBlock = null;
@@ -76,19 +76,26 @@ public class SectionsBuilderService {
AbstractTextContainer container = itty.next();
if (container instanceof Table) {
Table table = (Table) container;
splitByTable = true;
if (previous != null && previous instanceof TextBlock && previous.getText().startsWith("Table ")) {
((Table) container).setHeadline(previous.getText());
if (previous != null && previous.getText().startsWith("Table ")) {
table.setHeadline(previous.getText());
} else {
((Table) container).setHeadline("Table in: " + lastHeadline);
table.setHeadline("Table in: " + lastHeadline);
}
// Distribute header information for subsequent tables
if (previousTable != null && hasInvalidHeaderInformation(table) && hasValidHeaderInformation(previousTable) &&
(previousTable.isVerticalHeader() && previousTable.getRowCount() == table.getRowCount() ||
previousTable.getColCount() == table.getColCount())) {
table.setHeaders(previousTable.getHeaders());
}
if (textBlock != null && !alreadyAdded) {
paragraph.getPageBlocks().add(textBlock);
alreadyAdded = true;
}
paragraph.getPageBlocks().add(container);
paragraph.getPageBlocks().add(table);
continue;
}
@@ -125,4 +132,24 @@ public class SectionsBuilderService {
return paragraph;
}
}
private boolean hasValidHeaderInformation(Table table) {
return !hasInvalidHeaderInformation(table);
}
private boolean hasInvalidHeaderInformation(Table table) {
if (CollectionUtils.isEmpty(table.getHeaders())) {
return true;
}
if (table.getHeaders().stream().anyMatch(StringUtils::isEmpty)) {
return true;
}
return false;
}
}
@@ -16,11 +16,17 @@ public class Cell extends Rectangle {
private List<TextBlock> textBlocks = new ArrayList<>();
public Cell(Point2D topLeft, Point2D bottomRight) {
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight.getY() - topLeft.getY()));
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight
.getY() - topLeft.getY()));
}
public void addTextBlock(TextBlock textBlock) {
textBlocks.add(textBlock);
}
}
@@ -8,25 +8,28 @@ import org.locationtech.jts.index.strtree.STRtree;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@SuppressWarnings("all")
public class RectangleSpatialIndex<T extends Rectangle> {
private final STRtree si = new STRtree();
private final List<T> rectangles = new ArrayList<>();
public void add(T te) {
rectangles.add(te);
si.insert(new Envelope(te.getLeft(), te.getRight(), te.getBottom(), te.getTop()), te);
}
public List<T> contains(Rectangle r) {
List<T> intersection = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
public List<T> contains(Rectangle rectangle) {
List<T> intersection = si.query(new Envelope(rectangle.getLeft(), rectangle.getRight(), rectangle.getTop(), rectangle
.getBottom()));
List<T> rv = new ArrayList<T>();
for (T ir: intersection) {
if (r.contains(ir)) {
for (T ir : intersection) {
if (rectangle.contains(ir)) {
rv.add(ir);
}
}
@@ -34,18 +37,22 @@ public class RectangleSpatialIndex<T extends Rectangle> {
Utils.sort(rv, Rectangle.ILL_DEFINED_ORDER);
return rv;
}
public List<T> intersects(Rectangle r) {
List rv = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
return rv;
}
/**
* Minimum bounding box of all the Rectangles contained on this RectangleSpatialIndex
*
*
* @return a Rectangle
*/
public Rectangle getBounds() {
return Rectangle.boundingBoxOf(rectangles);
}
@@ -8,32 +8,45 @@ import java.util.Iterator;
import java.util.List;
import java.util.Map;
import java.util.TreeMap;
import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.Getter;
import lombok.Setter;
import lombok.extern.slf4j.Slf4j;
@SuppressWarnings("all")
@Slf4j
public class Table extends AbstractTextContainer {
private final TreeMap<CellPosition, Cell> cells = new TreeMap<>();
private RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
private final RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
@Getter
@Setter
private String headline;
@Getter
private int rowCount = 0;
private int rowCount;
@Getter
private int colCount = 0;
private int colCount;
private int rotation = 0;
private final int rotation;
private List<List<Cell>> memoizedRows = null;
private List<List<Cell>> rows;
@Getter
@Setter
private List<String> headers;
@Getter
private boolean verticalHeader;
public Table(List<Cell> cells, Rectangle area, int rotation) {
@@ -47,16 +60,90 @@ public class Table extends AbstractTextContainer {
}
public List<List<Cell>> getRows() {
if (memoizedRows == null) {
memoizedRows = computeRows();
if (rows == null) {
rows = computeRows();
headers = computeHeaders();
}
return memoizedRows;
return rows;
}
/**
* Detect header cells (either first row or first column):
* Column is marked as header if cell text is bold and row cell text is not bold.
* Defaults to row.
*/
private List<String> computeHeaders() {
boolean allBold = true;
if (rows.isEmpty()) {
return Collections.emptyList();
}
List<Cell> rowCells = rows.get(0);
for (Cell cell : rowCells) {
if (cell == null || CollectionUtils.isEmpty(cell.getTextBlocks()) ||
!cell.getTextBlocks().get(0).getMostPopularWordStyle().equals("bold")) {
allBold = false;
break;
}
}
if (!allBold) {
allBold = true;
List<Cell> firstColCells = new ArrayList<>();
for (List<Cell> row : rows) {
Cell firstInRow = row.get(0);
if (firstInRow == null || CollectionUtils.isEmpty(firstInRow.getTextBlocks()) ||
!firstInRow.getTextBlocks().get(0).getMostPopularWordStyle().equals("bold")) {
allBold = false;
break;
}
firstColCells.add(firstInRow);
}
if (allBold) {
log.info("Headers are in first column");
verticalHeader = true;
return firstColCells.stream().map(cell -> {
if (CollectionUtils.isNotEmpty(cell.getTextBlocks())) {
return TextNormalizationUtilities.removeHyphenLineBreaks(cell.getTextBlocks().get(0).getText())
.replaceAll("\n", " ")
.replaceAll(" ", " ");
} else {
return null;
}
}).collect(Collectors.toList());
} else {
log.info("Headers are defaulted in first row.");
return rowCells.stream().map(cell -> {
if (cell != null && CollectionUtils.isNotEmpty(cell.getTextBlocks())) {
return TextNormalizationUtilities.removeHyphenLineBreaks(cell.getTextBlocks().get(0).getText())
.replaceAll("\n", " ")
.replaceAll(" ", " ");
} else {
return null;
}
}).collect(Collectors.toList());
}
} else {
log.info("Headers are in first row.");
return rowCells.stream().map(cell -> {
if (CollectionUtils.isNotEmpty(cell.getTextBlocks())) {
return TextNormalizationUtilities.removeHyphenLineBreaks(cell.getTextBlocks().get(0).getText())
.replaceAll("\n", " ")
.replaceAll(" ", " ");
} else {
return null;
}
}).collect(Collectors.toList());
}
}
private List<List<Cell>> computeRows() {
List<List<Cell>> rows = new ArrayList<>();
@@ -93,7 +180,8 @@ public class Table extends AbstractTextContainer {
}
public void add(Cell chunk, int row, int col) {
private void add(Cell chunk, int row, int col) {
rowCount = Math.max(rowCount, row + 1);
colCount = Math.max(colCount, col + 1);
@@ -103,6 +191,7 @@ public class Table extends AbstractTextContainer {
}
private void addCells(List<Cell> cells) {
if (cells.isEmpty()) {
@@ -131,14 +220,9 @@ public class Table extends AbstractTextContainer {
while (rowCells.hasNext()) {
Cell cell = rowCells.next();
if (i > 0) {
List<List<Cell>> others = rowsOfCells(
si.contains(
new Rectangle(cell.getBottom(),
si.getBounds().getLeft(),
cell.getLeft() - si.getBounds().getLeft() + 1,
si.getBounds().getBottom() - cell.getBottom()
)
));
List<List<Cell>> others = rowsOfCells(si.contains(new Rectangle(cell.getBottom(), si.getBounds()
.getLeft(), cell.getLeft() - si.getBounds().getLeft() + 1, si.getBounds().getBottom() - cell
.getBottom())));
for (List<Cell> r : others) {
jumpToColumn = Math.max(jumpToColumn, r.size());
@@ -158,7 +242,9 @@ public class Table extends AbstractTextContainer {
}
}
private static List<List<Cell>> rowsOfCells(List<Cell> cells) {
Cell c;
float lastTop;
List<List<Cell>> rv = new ArrayList<>();
@@ -168,19 +254,10 @@ public class Table extends AbstractTextContainer {
return rv;
}
Collections.sort(cells, new Comparator<Cell>() {
@Override
public int compare(Cell arg0, Cell arg1) {
return Double.compare(arg0.getLeft(), arg1.getLeft());
}
});
cells.sort(Comparator.comparingDouble(Rectangle::getLeft));
Collections.sort(cells, Collections.reverseOrder(new Comparator<Cell>() {
@Override
public int compare(Cell arg0, Cell arg1) {
return Float.compare(Utils.round(arg0.getBottom(), 2), Utils.round(arg1.getBottom(),2));
}
}));
cells.sort(Collections.reverseOrder((arg0, arg1) -> Float.compare(Utils.round(arg0.getBottom(), 2), Utils.round(arg1
.getBottom(), 2))));
Iterator<Cell> iter = cells.iterator();
c = iter.next();
@@ -201,6 +278,7 @@ public class Table extends AbstractTextContainer {
return rv;
}
@Override
public String getText() {
@@ -237,6 +315,7 @@ public class Table extends AbstractTextContainer {
return sb.toString();
}
public String getTextAsHtml() {
StringBuilder sb = new StringBuilder();
@@ -270,22 +349,30 @@ public class Table extends AbstractTextContainer {
return sb.toString();
}
class CellPosition implements Comparable<CellPosition> {
static class CellPosition implements Comparable<CellPosition> {
CellPosition(int row, int col) {
this.row = row;
this.col = col;
}
final int row, col;
final int row;
final int col;
@Override
public int hashCode() {
return row + 101 * col;
}
@Override
public boolean equals(Object obj) {
if (this == obj) {
return true;
}
@@ -299,10 +386,12 @@ public class Table extends AbstractTextContainer {
return row == other.row && col == other.col;
}
@Override
public int compareTo(CellPosition other) {
int rowdiff = row - other.row;
return rowdiff != 0 ? rowdiff : col - other.col;
int rowDiff = row - other.row;
return rowDiff != 0 ? rowDiff : col - other.col;
}
}
@@ -167,6 +167,7 @@ public class AnnotationHighlightService {
redactionLogEntry.setSection(entity.getHeadline());
redactionLogEntry.setHint(isHint(entity));
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
redactionLogEntry.setSectionNumber(entity.getSectionNumber());
}
}
@@ -5,6 +5,8 @@ import static org.springframework.boot.test.context.SpringBootTest.WebEnvironmen
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
@@ -19,7 +21,6 @@ import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
import org.junit.Before;
import org.junit.Ignore;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
@@ -48,7 +49,6 @@ import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
@Ignore
@RunWith(SpringRunner.class)
@SpringBootTest(webEnvironment = DEFINED_PORT)
public class RedactionIntegrationTest {
@@ -58,6 +58,9 @@ public class RedactionIntegrationTest {
private static final String ADDRESS_CODE = "address";
private static final String NAME_CODE = "name";
private static final String NO_REDACTION_INDICATOR = "no_redaction_indicator";
private static final String REDACTION_INDICATOR = "redaction_indicator";
private static final String HINT_ONLY = "hint_only";
private static final String MUST_REDACT = "must_redact";
@Autowired
private RedactionController redactionController;
@@ -96,7 +99,7 @@ public class RedactionIntegrationTest {
@Before
public void stubRulesClient() {
public void stubClients() {
when(rulesClient.getVersion()).thenReturn(0L);
when(rulesClient.getRules()).thenReturn(new RulesResponse(RULES));
@@ -109,6 +112,9 @@ public class RedactionIntegrationTest {
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(getDictionaryResponse(ADDRESS_CODE));
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(getDictionaryResponse(NAME_CODE));
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(HINT_ONLY)).thenReturn(getDictionaryResponse(HINT_ONLY));
when(dictionaryClient.getDictionaryForType(MUST_REDACT)).thenReturn(getDictionaryResponse(MUST_REDACT));
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor(new float[]{1f, 0.502f, 0f}));
}
@@ -131,7 +137,22 @@ public class RedactionIntegrationTest {
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(NO_REDACTION_INDICATOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/NoRedactionIndicator.txt")
.addAll(ResourceLoader.load("dictionaries/no_redaction_indicator.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(REDACTION_INDICATOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/redaction_indicator.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(HINT_ONLY, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/hint_only.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(MUST_REDACT, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/must_redact.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
@@ -149,17 +170,26 @@ public class RedactionIntegrationTest {
typeColorMap.put(VERTEBRATES_CODE, new float[]{0, 1, 0});
typeColorMap.put(ADDRESS_CODE, new float[]{0, 1, 1});
typeColorMap.put(NAME_CODE, new float[]{1, 1, 0});
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{1, 0.502f, 0});
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{0.8f, 0, 0.8f});
typeColorMap.put(REDACTION_INDICATOR, new float[]{1, 0.502f, 0.1f});
typeColorMap.put(HINT_ONLY, new float[]{0.8f, 1, 0.8f});
typeColorMap.put(MUST_REDACT, new float[]{1, 0, 0});
hintTypeMap.put(VERTEBRATES_CODE, true);
hintTypeMap.put(ADDRESS_CODE, false);
hintTypeMap.put(NAME_CODE, false);
hintTypeMap.put(NO_REDACTION_INDICATOR, true);
hintTypeMap.put(REDACTION_INDICATOR, true);
hintTypeMap.put(HINT_ONLY, true);
hintTypeMap.put(MUST_REDACT, true);
caseInSensitiveMap.put(VERTEBRATES_CODE, true);
caseInSensitiveMap.put(ADDRESS_CODE, false);
caseInSensitiveMap.put(NAME_CODE, false);
caseInSensitiveMap.put(NO_REDACTION_INDICATOR, true);
caseInSensitiveMap.put(REDACTION_INDICATOR, true);
caseInSensitiveMap.put(HINT_ONLY, true);
caseInSensitiveMap.put(MUST_REDACT, true);
}
@@ -189,11 +219,50 @@ public class RedactionIntegrationTest {
}
@Test
public void noExceptionShouldBeThrownForAnyFiles() throws IOException {
ClassLoader loader = getClass().getClassLoader();
URL url = loader.getResource("files");
File[] files = new File(url.getPath()).listFiles();
List<File> input = new ArrayList<>();
for (File file : files) {
input.addAll(getPathsRecursively(file));
}
for (File path : input) {
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(new FileInputStream(path)))
.build();
System.out.println("Redacting file : " + path.getName());
redactionController.redact(request);
}
}
private List<File> getPathsRecursively(File path) {
List<File> result = new ArrayList<>();
if (path == null || path.listFiles() == null) {
return result;
}
for (File f : path.listFiles()) {
if (f.isFile()) {
result.add(f);
} else {
result.addAll(getPathsRecursively(f));
}
}
return result;
}
@Test
public void redactionTest() throws IOException {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Trinexapac/96 Trinexapac-ethyl_RAR_09_Volume_3CA_B-7_2018-02-23.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -212,10 +281,33 @@ public class RedactionIntegrationTest {
}
@Test
public void testTableRedaction() throws IOException {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.redact(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
}
@Test
public void classificationTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -232,7 +324,8 @@ public class RedactionIntegrationTest {
@Test
public void sectionsTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -249,7 +342,8 @@ public class RedactionIntegrationTest {
@Test
public void htmlTablesTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -266,7 +360,8 @@ public class RedactionIntegrationTest {
@Test
public void htmlTableRotationTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S" +
"-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -286,7 +381,8 @@ public class RedactionIntegrationTest {
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(),
StandardCharsets.UTF_8))) {
StringBuilder sb = new StringBuilder();
String str;
while ((str = br.readLine()) != null) {
@@ -0,0 +1,175 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.util.Arrays;
import java.util.Collections;
import java.util.HashSet;
import java.util.Set;
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.test.context.SpringBootTest;
import org.springframework.boot.test.context.TestConfiguration;
import org.springframework.boot.test.mock.mockito.MockBean;
import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
@RunWith(SpringRunner.class)
@SpringBootTest
public class EntityRedactionServiceTest {
private static final String DEFAULT_RULES = loadFromClassPath("drools/rules.drl");
private static final String NAME_CODE = "name";
private static final String ADDRESS_CODE = "address";
@MockBean
private DictionaryClient dictionaryClient;
@MockBean
private RulesClient rulesClient;
@Autowired
private EntityRedactionService entityRedactionService;
@Autowired
private PdfSegmentationService pdfSegmentationService;
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@Bean
public KieContainer kieContainer() {
KieServices kieServices = KieServices.Factory.get();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
InputStream input = new ByteArrayInputStream(DEFAULT_RULES.getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/test/resources/drools/rules.drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
return kieServices.newKieContainer(kieModule.getReleaseId());
}
}
@Test
public void testNestedEntitiesRemoval() {
Set<Entity> entities = new HashSet<>();
Entity nested = new Entity("nested", "fake type", 10, 16, "fake headline", 0);
Entity nesting = new Entity("nesting nested", "fake type", 2, 16, "fake headline", 0);
entities.add(nested);
entities.add(nesting);
entityRedactionService.removeEntitiesContainedInLarger(entities);
assertThat(entities.size()).isEqualTo(1);
assertThat(entities).contains(nesting);
}
@Test
public void testTableRedaction() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
RedactionRequest redactionRequest = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" +
"rule \"9: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study\"\n" +
" when\n" +
" Section(tabularData != null && tabularData.size() > 0\n" +
" && tabularData.containsKey(\"Vertebrate study Y/N\")\n" +
" && tabularData.get(\"Vertebrate study Y/N\").equals(\"Y\")\n" +
" )\n" +
" then\n" +
" section.redact(\"name\", 9, \"Redacted because row is a vertebrate study\");\n" +
" section.redact(\"address\", 9, \"Redacted because rows is a vertebrate study\");\n" +
" section.highlightCell(\"Vertebrate study Y/N\", 9);\n" +
" end";
when(rulesClient.getVersion()).thenReturn(1L);
when(rulesClient.getRules()).thenReturn(new RulesResponse(tableRules));
TypeResponse typeResponse = TypeResponse.builder()
.types(Arrays.asList(
TypeResult.builder().type(NAME_CODE).color(new float[]{1, 1, 0}).build(),
TypeResult.builder().type(ADDRESS_CODE).color(new float[]{0, 1, 1}).build()))
.build();
when(dictionaryClient.getAllTypes()).thenReturn(typeResponse);
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Casey, H.W.", "O’Loughlin, C.K.", "Salamon, C.M.", "Smith, S.H."))
.build();
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Toxigenics, Inc., Decatur, IL 62526, USA"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor());
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1)).hasSize(5); // 4 out of 5 entities recognized on page 1
}
}
private static String loadFromClassPath(String path) {
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
StringBuilder sb = new StringBuilder();
String str;
while ((str = br.readLine()) != null) {
sb.append(str).append("\n");
}
return sb.toString();
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + path, e);
}
}
}
@@ -0,0 +1,2 @@
guideline
unpublished
@@ -0,0 +1,3 @@
Batches Produced at
CTL
for determination of residues
@@ -0,0 +1,3 @@
published paper
in vitro
in-vitro
@@ -0,0 +1,9 @@
in vivo
in-vivo
dermal penetration
oral toxicity
oral-toxicity
acute toxicity
acute-toxicity
eco toxicity
eco-toxicity
@@ -1,48 +1,63 @@
Vulpes vulpes
a. sylvaticus
african clawed frog
agalychnis callidryas
albino rat
american bullfrog tadpole
american toad
amphibian
amphibians
American bullfrog tadpole
american toad
anad platyrhynchos
Anas platyrhynchos
anas platyrhynchos
anuran
anurans
apodemus
apodemus flavicollis
apodemus syl vaticus
apodemus sylvaticus
arvicola terrestris
avian
bank vole
bird
birds
bluegill
bluegill sunfish
bobwhite
bobwhite quail
bullfrog
Bufo americanus
brachydanio rerio
brown hare
bufo americanus
bullfrog
canary
carassius carassius
carp
catesbeiana
catfish
cattle
cattles
channel catfish
Chinook
chicken
Colinus virginianus
chinese hamster
chinese hamsters
chinook
coho salmon
colinus virginianus
Common carp
columba palumbus
columbidae
common carp
common vole
coturnix japonica
Coturnix japonica
cow
cows
Crucian carp
crocidura russula
crucian carp
cyprinodon variegatus
cyprinus carpio
dog
dogs
duck
ducks
european brown hare
european rabbit
fathead minnow
fish
fishes
@@ -56,174 +71,121 @@ galaxias truttaceus
gasterosteus aculeatus
goat
goats
greater white-toothed shrew
guinea
guinea pig
guinea pigs
Guppy
guinea-pigs
guppy
hamster
hamsters
hen
hens
Hyla versicolor
house mouse
hyla versicolor
ictalurus melas
ictalurus punctatus
japanese quail
japonica
kisutch
lagomorph
lebistes reticulatus
leiostomus xanthurus
leisostomus xanthurus
lepomis macrochirus
lepus europaeus
limnocharis
limnodynastes
limnodynastes tasmaniensis
livestock
livestocks
mallard
mallard duck
mammal
mammalian
mammals
Mammalian
marten
martes
mice
microtus
microtus agrestis
microtus arvalis
microtus subterraneus
midwestern anurans
minnow
minnows
monkey
mouse
mus musculus
myodes glareolus
northern bobwhite
o. cuniculus
o. mykiss
Oncorhynchus mykiss
Oncorhynchus
O. mykiss
o. tshawytscha
oncorhynchus
oncorhynchus mykiss
oncorhynchus tshawytscha
oryctolagus cuniculus
oryzias melastigma
oryzias melastigma larvae
p. promelas
pagrus major
palumbus
pig
pigeon
pigeons
pigs
pimephales promela
pimephales promelas
Pseudacris triseriata
poecilia reticulata
poultry
pseudacris
pseudacris triseriata
quail
r. catesbeiana
rabbit
rabbits
rainbow trout
Rana limnocharis
rana
limnocharis
rana catesbeiana
rana limnocharis
rana pipiens
rat
rats
reptile
reptiles
ricefish
ruminant
ruminants
salmo gairdneri
salmon
serinus canaria
sheepshead minnow
sheepshead minnows
spea multiplicata
Salmo gairdneri
salmon
spotted march frog
tadpoles
treefrog
toad
terrestrial vertrebrates
Limnodynastes tasmaniensis
trout
Vulpes vulpes
wistar
xenopus laevis
xenpous leavis
zebra fish
zebrafish
Salmo gairdneri
minnow
minnows
Pimephales promela
Cyprinodon variegatus
limnodynastes
Rana catesbeiana
R. catesbeiana
coho salmon
Oncorhynchus tshawytscha
O. tshawytscha
tshawytscha
catesbeiana
kisutch
Pseudacris triseriata
Pseudacris
triseriata
Wood pigeon
Columba palumbus
palumbus
Columbidae
shrew
shrews
bank vole
common vole
sorex araneus
spea multiplicata
spotted march frog
tadpoles
terrestrial vertrebrates
toad
treefrog
triseriata
trout
tshawytscha
vole
voles
lagomorph
Wood mouse
Apodemus sylvaticus
A. sylvaticus
Apodemus flavicollis
Apodemus
mus musculus
Microtus arvalis
Microtus agrestis
Microtus
Arvicola terrestris
Sorex araneus
Myodes glareolus
yellow-necked mouse
house mouse
Oryctolagus cuniculus
marten
martes
vulpes vulpes
white rabbits
white-toothed shrew
greater white-toothed shrew
Lepus europaeus
brown hare
European brown hare
European rabbit
O. cuniculus
Crocidura russula
Chinese Hamster
Rat
Rats
Dog
Chinese hamsters
Chinese hamster
Mouse
Guinea pig
Wistar rats
Rabbit
mammalian
Japanese quail
Microtus subterraneus
Lepomis macrochirus
P. promelas
Cyprinus carpio
Fish
Ictalurus punctatus
Carassius carassius
Lepomis macrochirus
Poecilia reticulata
Lebistes reticulatus
Lepomis macrochirus
Leiostomus xanthurus
Pimephales promelas
Lepomis macrochirus
Albino rat
Hen
Goat
Livestock
Guinea Pigs
Hamster
wistar
wistar rats
wood mice
wood mouse
Rabbits
Mice
Rainbow trout
Canary
Serinus canaria
Guinea Pig
Cow
Pigs
Poultry
Guinea-pigs
White rabbits
Birds
Wood mice
wood pigeon
xenopus laevis
xenpous leavis
yellow-necked mouse
zebra fish
zebrafish
@@ -32,27 +32,81 @@ rule "3: Do not redact Names and Addresses if no redaction Indicator is containe
end
rule "4: Redact contact information, if applicant is found"
rule "4: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
when
eval(section.getText().toLowerCase().contains("applicant"));
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true && section.contains("redaction_indicator")==true);
then
section.redactLineAfter("Name:", "address", 4, "Redacted because of Rule 4");
section.redactBetween("Address:", "Contact", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Contact point:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Phone:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("Fax:", "address", 4, "Redacted because of Rule 4");
section.redactLineAfter("E-mail:", "address", 4, "Redacted because of Rule 4");
section.redact("name", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
section.redact("address", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
end
rule "5: Redact contact information, if 'Producer of the plant protection product' is found"
rule "5: Do not redact in guideline sections"
when
eval(section.getText().contains("Producer of the plant protection product"));
eval(section.headlineContainsWord("guideline") || section.headlineContainsWord("Guidance"));
then
section.redactLineAfter("Name:", "address", 5, "xxxx");
section.redactBetween("Address:", "Contact", "address", 5, "xxxx");
section.redactBetween("Contact:", "Phone", "address", 5, "xxxx");
section.redactLineAfter("Phone:", "address", 5, "xxxx");
section.redactLineAfter("Fax:", "address", 5, "xxxx");
section.redactLineAfter("E-mail:", "address", 5, "xxxx");
end
section.redactNot("name", 5, "Section is a guideline section.");
section.redactNot("address", 5, "Section is a guideline section.");
end
rule "6: Redact if must redact entry is found"
when
eval(section.contains("must_redact")==true);
then
section.redact("name", 6, "must_redact entry was found.");
section.redact("address", 6, "must_redact entry was found.");
end
rule "7: Redact contact information, if applicant is found"
when
eval(section.headlineContainsWord("applicant") || section.getText().contains("Applicant"));
then
section.redactLineAfter("Name:", "address", 7, "Applicant information was found");
section.redactBetween("Address:", "Contact", "address", 7, "Applicant information was found");
section.redactLineAfter("Contact point:", "address", 7, "Applicant information was found");
section.redactLineAfter("Phone:", "address", 7, "Applicant information was found");
section.redactLineAfter("Fax:", "address", 7, "Applicant information was found");
section.redactLineAfter("Tel.:", "address", 7, "Applicant information was found");
section.redactLineAfter("Tel:", "address", 7, "Applicant information was found");
section.redactLineAfter("E-mail:", "address", 7, "Applicant information was found");
section.redactLineAfter("Email:", "address", 7, "Applicant information was found");
section.redactLineAfter("Contact:", "address", 7, "Applicant information was found");
section.redactLineAfter("Telephone number:", "address", 7, "Applicant information was found");
section.redactLineAfter("Fax number:", "address", 7, "Applicant information was found");
section.redactLineAfter("Telephone:", "address", 7, "Applicant information was found");
section.redactBetween("No:", "Fax", "address", 7, "Applicant information was found");
section.redactBetween("Contact:", "Tel.:", "address", 7, "Applicant information was found");
end
rule "8: Redact contact information, if Producer is found"
when
eval(section.getText().toLowerCase().contains("producer of the plant protection") || section.getText().toLowerCase().contains("producer of the active substance") || section.getText().contains("Manufacturer of the active substance") || section.getText().contains("Manufacturer:") || section.getText().contains("Producer or producers of the active substance"));
then
section.redactLineAfter("Name:", "address", 8, "Producer was found");
section.redactBetween("Address:", "Contact", "address", 8, "Producer was found");
section.redactBetween("Contact:", "Phone", "address", 8, "Producer was found");
section.redactBetween("Contact:", "Telephone number:", "address", 8, "Producer was found");
section.redactBetween("Address:", "Manufacturing", "address", 8, "Producer was found");
section.redactLineAfter("Telephone:", "address", 8, "Producer was found");
section.redactLineAfter("Phone:", "address", 8, "Producer was found");
section.redactLineAfter("Fax:", "address", 8, "Producer was found");
section.redactLineAfter("E-mail:", "address", 8, "Producer was found");
section.redactLineAfter("Contact:", "address", 8, "Producer was found");
section.redactLineAfter("Fax number:", "address", 8, "Producer was found");
section.redactLineAfter("Telephone number:", "address", 8, "Producer was found");
section.redactLineAfter("Tel:", "address", 8, "Producer was found");
section.redactBetween("No:", "Fax", "address", 8, "Producer was found");
end
rule "9: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study"
when
Section(tabularData != null
&& tabularData.containsKey("Vertebrate study Y/N")
&& tabularData.get("Vertebrate study Y/N").equals("Y")
)
then
section.redact("name", 9, "Redacted because row is a vertebrate study");
section.redact("address", 9, "Redacted because rows is a vertebrate study");
section.highlightCell("Vertebrate study Y/N", 9);
end