Compare commits
48
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
afa26895c8 | ||
|
|
85667c91e4 | ||
|
|
8a471d0771 | ||
|
|
bb1112d0d7 | ||
|
|
00a960ee23 | ||
|
|
8c08bb3664 | ||
|
|
14b4f4ab8a | ||
|
|
4206e708a2 | ||
|
|
c4164b57fb | ||
|
|
76369f13f8 | ||
|
|
1fff4f7eb0 | ||
|
|
c7f5b4a280 | ||
|
|
89ef8234d2 | ||
|
|
4236fa05cc | ||
|
|
031f10435d | ||
|
|
5d286f58e9 | ||
|
|
5d18a46ec3 | ||
|
|
9b61971132 | ||
|
|
2b93ae57d5 | ||
|
|
954765759c | ||
|
|
17aabcd09c | ||
|
|
32aa500983 | ||
|
|
a151a13b4c | ||
|
|
5542a97a38 | ||
|
|
8edaa93bda | ||
|
|
98e0bf5606 | ||
|
|
8344b7ccaf | ||
|
|
96ba93f774 | ||
|
|
c93ca745fc | ||
|
|
a6415363cd | ||
|
|
d97a7b629a | ||
|
|
00c96c6f57 | ||
|
|
06630b09d2 | ||
|
|
695564d162 | ||
|
|
81048dcc9f | ||
|
|
b5412dc959 | ||
|
|
40e40a01ad | ||
|
|
99bac4550a | ||
|
|
7d0b0ed3d0 | ||
|
|
d465a4ba5b | ||
|
|
cce8200d43 | ||
|
|
1f8d371a82 | ||
|
|
70804f111d | ||
|
|
81723ce402 | ||
|
|
b1a266d4d4 | ||
|
|
d2d7f8c50c | ||
|
|
e2895a1c7a | ||
|
|
cd07dc6a44 |
@@ -5,7 +5,7 @@
|
||||
<parent>
|
||||
<artifactId>platform-dependency</artifactId>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<version>1.0.1</version>
|
||||
<version>1.0.2</version>
|
||||
</parent>
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
@@ -32,21 +32,21 @@
|
||||
<dependency>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<artifactId>platform-commons-dependency</artifactId>
|
||||
<version>1.0.0</version>
|
||||
<version>1.1.0</version>
|
||||
<scope>import</scope>
|
||||
<type>pom</type>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>pdfbox</artifactId>
|
||||
<version>${pdfbox.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>pdfbox-tools</artifactId>
|
||||
<version>${pdfbox.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>pdfbox</artifactId>
|
||||
<version>${pdfbox.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>pdfbox-tools</artifactId>
|
||||
<version>${pdfbox.version}</version>
|
||||
</dependency>
|
||||
|
||||
</dependencies>
|
||||
|
||||
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class ManualRedactionEntry {
|
||||
|
||||
private String type;
|
||||
private String value;
|
||||
private String reason;
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
|
||||
private String section;
|
||||
private int sectionNumber;
|
||||
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class ManualRedactions {
|
||||
|
||||
private Set<String> idsToRemove = new HashSet<>();
|
||||
private Set<ManualRedactionEntry> entriesToAdd = new HashSet<>();
|
||||
}
|
||||
+10
@@ -3,9 +3,15 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionLogEntry {
|
||||
|
||||
private String id;
|
||||
@@ -16,6 +22,10 @@ public class RedactionLogEntry {
|
||||
private boolean isHint;
|
||||
private String section;
|
||||
private float[] color;
|
||||
|
||||
@Builder.Default
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
private int sectionNumber;
|
||||
private boolean manual;
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -13,4 +13,5 @@ public class RedactionRequest {
|
||||
|
||||
private byte[] document;
|
||||
private boolean flatRedaction;
|
||||
private ManualRedactions manualRedactions;
|
||||
}
|
||||
|
||||
@@ -56,17 +56,22 @@
|
||||
<artifactId>jts-core</artifactId>
|
||||
<version>1.16.1</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.google.guava</groupId>
|
||||
<artifactId>guava</artifactId>
|
||||
<version>29.0-jre</version>
|
||||
</dependency>
|
||||
<!-- commons -->
|
||||
<dependency>
|
||||
<groupId>com.iqser.gin4.commons</groupId>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
<artifactId>spring-commons</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.iqser.gin4.commons</groupId>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
<artifactId>logging-commons</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.iqser.gin4.commons</groupId>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
<artifactId>metric-commons</artifactId>
|
||||
</dependency>
|
||||
<!-- other external -->
|
||||
@@ -99,7 +104,7 @@
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.iqser.gin4.commons</groupId>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
<artifactId>test-commons</artifactId>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
+1
-1
@@ -20,7 +20,7 @@ import org.springframework.cloud.openfeign.EnableFeignClients;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Import;
|
||||
|
||||
import com.iqser.gin4.commons.spring.DefaultWebMvcConfiguration;
|
||||
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
|
||||
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
|
||||
+1
-2
@@ -4,7 +4,6 @@ import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
@@ -18,7 +17,7 @@ public class Document {
|
||||
|
||||
private List<Page> pages = new ArrayList<>();
|
||||
private List<Paragraph> paragraphs = new ArrayList<>();
|
||||
private Map<Integer, Set<Entity>> entities = new HashMap<>();
|
||||
private Map<Integer, List<Entity>> entities = new HashMap<>();
|
||||
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
|
||||
private FloatFrequencyCounter fontSizeCounter= new FloatFrequencyCounter();
|
||||
private StringFrequencyCounter fontCounter= new StringFrequencyCounter();
|
||||
|
||||
+20
-6
@@ -10,7 +10,6 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
public class Paragraph {
|
||||
@@ -18,10 +17,12 @@ public class Paragraph {
|
||||
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
|
||||
private String headline;
|
||||
|
||||
public SearchableText getSearchableText(){
|
||||
|
||||
public SearchableText getSearchableText() {
|
||||
|
||||
SearchableText searchableText = new SearchableText();
|
||||
pageBlocks.forEach(block -> {
|
||||
if(block instanceof TextBlock){
|
||||
if (block instanceof TextBlock) {
|
||||
searchableText.addAll(((TextBlock) block).getSequences());
|
||||
}
|
||||
});
|
||||
@@ -29,14 +30,27 @@ public class Paragraph {
|
||||
}
|
||||
|
||||
|
||||
public List<Table> getTables(){
|
||||
public List<Table> getTables() {
|
||||
|
||||
List<Table> tables = new ArrayList<>();
|
||||
pageBlocks.forEach(block -> {
|
||||
if(block instanceof Table){
|
||||
if (block instanceof Table) {
|
||||
tables.add((Table) block);
|
||||
}
|
||||
});
|
||||
return tables;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
public List<TextBlock> getTextBlocks() {
|
||||
|
||||
List<TextBlock> textBlocks = new ArrayList<>();
|
||||
pageBlocks.forEach(block -> {
|
||||
if (block instanceof TextBlock) {
|
||||
textBlocks.add((TextBlock) block);
|
||||
}
|
||||
});
|
||||
return textBlocks;
|
||||
}
|
||||
|
||||
}
|
||||
+18
-16
@@ -5,43 +5,45 @@ import java.util.Map;
|
||||
|
||||
import lombok.Getter;
|
||||
|
||||
/**
|
||||
*
|
||||
*/
|
||||
public class StringFrequencyCounter {
|
||||
|
||||
@Getter
|
||||
Map<String, Integer> countPerValue = new HashMap<>();
|
||||
private final Map<String, Integer> countPerValue = new HashMap<>();
|
||||
|
||||
public void add(String value){
|
||||
if(!countPerValue.containsKey(value)){
|
||||
|
||||
public void add(String value) {
|
||||
|
||||
if (!countPerValue.containsKey(value)) {
|
||||
countPerValue.put(value, 1);
|
||||
} else {
|
||||
countPerValue.put(value, countPerValue.get(value) + 1);
|
||||
}
|
||||
}
|
||||
|
||||
public void addAll(Map<String, Integer> otherCounter){
|
||||
for(Map.Entry<String, Integer> entry: otherCounter.entrySet()){
|
||||
if(countPerValue.containsKey(entry.getKey())){
|
||||
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey())+ entry.getValue());
|
||||
|
||||
public void addAll(Map<String, Integer> otherCounter) {
|
||||
|
||||
for (Map.Entry<String, Integer> entry : otherCounter.entrySet()) {
|
||||
if (countPerValue.containsKey(entry.getKey())) {
|
||||
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
|
||||
} else {
|
||||
countPerValue.put(entry.getKey(), entry.getValue());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public String getMostPopular(){
|
||||
|
||||
public String getMostPopular() {
|
||||
|
||||
Map.Entry<String, Integer> mostPopular = null;
|
||||
for(Map.Entry<String, Integer> entry: countPerValue.entrySet()){
|
||||
if(mostPopular == null){
|
||||
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
|
||||
if (mostPopular == null) {
|
||||
mostPopular = entry;
|
||||
} else if(entry.getValue() > mostPopular.getValue()){
|
||||
} else if (entry.getValue() > mostPopular.getValue()) {
|
||||
mostPopular = entry;
|
||||
}
|
||||
}
|
||||
return mostPopular != null ? mostPopular.getKey() : null;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
}
|
||||
+13
-19
@@ -29,20 +29,16 @@ public class BlockificationService {
|
||||
float minX = 1000, maxX = 0, minY = 1000, maxY = 0;
|
||||
TextPositionSequence prev = null;
|
||||
|
||||
|
||||
for (TextPositionSequence word : textPositions) {
|
||||
|
||||
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
|
||||
boolean startFromTop = word.getY1() > maxY + word.getHeight();
|
||||
|
||||
if (prev != null &&
|
||||
(lineSeparation
|
||||
|| startFromTop
|
||||
|| word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines)
|
||||
|| word.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines)
|
||||
|| word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines)
|
||||
|| word.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines)
|
||||
)) {
|
||||
if (prev != null && (lineSeparation || startFromTop || word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word
|
||||
.getX1(), word.getY1(), verticalRulingLines) || word.getRotation() == 0 && isSplittedByRuling(minX, minY, word
|
||||
.getX1(), word.getY2(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word
|
||||
.getX1(), word.getY1(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(minX, minY, word
|
||||
.getX1(), word.getY2(), verticalRulingLines))) {
|
||||
|
||||
TextBlock cb1 = buildTextBlock(chunkWords);
|
||||
chunkBlockList1.add(cb1);
|
||||
@@ -100,11 +96,12 @@ public class BlockificationService {
|
||||
styleFrequencyCounter.add(wordBlock.getFontStyle());
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock.getRotation());
|
||||
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
|
||||
.getRotation());
|
||||
} else {
|
||||
TextBlock spatialEntity = textBlock.union(wordBlock);
|
||||
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(),
|
||||
spatialEntity.getWidth(), spatialEntity.getHeight());
|
||||
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
|
||||
.getHeight());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -122,6 +119,7 @@ public class BlockificationService {
|
||||
|
||||
|
||||
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1, List<Ruling> rulingLines) {
|
||||
|
||||
for (Ruling ruling : rulingLines) {
|
||||
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
|
||||
return true;
|
||||
@@ -133,7 +131,6 @@ public class BlockificationService {
|
||||
|
||||
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape) {
|
||||
|
||||
|
||||
float minX = 10000;
|
||||
float maxX = -100;
|
||||
float minY = 10000;
|
||||
@@ -147,7 +144,6 @@ public class BlockificationService {
|
||||
|
||||
for (AbstractTextContainer container : page.getTextBlocks()) {
|
||||
|
||||
|
||||
if (container instanceof TextBlock) {
|
||||
TextBlock textBlock = (TextBlock) container;
|
||||
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
|
||||
@@ -179,16 +175,15 @@ public class BlockificationService {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
if (container instanceof Table) {
|
||||
Table table = (Table) container;
|
||||
for (List<Cell> row : table.getRows()) {
|
||||
for (Cell column : row) {
|
||||
for (Cell cell : row) {
|
||||
|
||||
if (column == null || column.getTextBlocks() == null) {
|
||||
if (cell == null || cell.getTextBlocks() == null) {
|
||||
continue;
|
||||
}
|
||||
for (TextBlock textBlock : column.getTextBlocks()) {
|
||||
for (TextBlock textBlock : cell.getTextBlocks()) {
|
||||
if (textBlock.getMinX() < minX) {
|
||||
minX = textBlock.getMinX();
|
||||
}
|
||||
@@ -211,5 +206,4 @@ public class BlockificationService {
|
||||
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
+1
-1
@@ -2,13 +2,13 @@ package com.iqser.red.service.redaction.v1.server.controller;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
import com.iqser.red.commons.spring.ErrorMessage;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.web.bind.annotation.ExceptionHandler;
|
||||
import org.springframework.web.bind.annotation.ResponseBody;
|
||||
import org.springframework.web.bind.annotation.ResponseStatus;
|
||||
import org.springframework.web.bind.annotation.RestControllerAdvice;
|
||||
|
||||
import com.iqser.gin4.commons.api.errorhandling.ErrorMessage;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
+2
-2
@@ -44,8 +44,8 @@ public class RedactionController implements RedactionResource {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc);
|
||||
annotationHighlightService.highlight(pdDocument, classifiedDoc, redactionRequest.isFlatRedaction());
|
||||
entityRedactionService.processDocument(classifiedDoc, redactionRequest.getManualRedactions());
|
||||
annotationHighlightService.highlight(pdDocument, classifiedDoc, redactionRequest.isFlatRedaction(), redactionRequest.getManualRedactions());
|
||||
|
||||
if (redactionRequest.isFlatRedaction()) {
|
||||
PDDocument flatDocument = pdfFlattenService.flattenPDF(pdDocument);
|
||||
|
||||
+81
-4
@@ -5,6 +5,9 @@ import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.text.TextPosition;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
@@ -22,36 +25,48 @@ public class TextPositionSequence implements CharSequence {
|
||||
|
||||
private final int page;
|
||||
|
||||
public TextPositionSequence(List<TextPosition> textPositions, int page){
|
||||
|
||||
public TextPositionSequence(List<TextPosition> textPositions, int page) {
|
||||
|
||||
this.textPositions = textPositions;
|
||||
this.page = page;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int length() {
|
||||
|
||||
return textPositions.size();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public char charAt(int index) {
|
||||
|
||||
TextPosition textPosition = textPositionAt(index);
|
||||
String text = textPosition.getUnicode();
|
||||
return text.charAt(0);
|
||||
}
|
||||
|
||||
|
||||
public char charAt(int index, boolean caseInSensitive) {
|
||||
|
||||
TextPosition textPosition = textPositionAt(index);
|
||||
String text = textPosition.getUnicode();
|
||||
return caseInSensitive ? text.toLowerCase().charAt(0) : text.charAt(0);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public TextPositionSequence subSequence(int start, int end) {
|
||||
|
||||
return new TextPositionSequence(textPositions.subList(start, end), page);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
StringBuilder builder = new StringBuilder(length());
|
||||
for (int i = 0; i < length(); i++) {
|
||||
builder.append(charAt(i));
|
||||
@@ -59,15 +74,21 @@ public class TextPositionSequence implements CharSequence {
|
||||
return builder.toString();
|
||||
}
|
||||
|
||||
|
||||
public TextPosition textPositionAt(int index) {
|
||||
|
||||
return textPositions.get(index);
|
||||
}
|
||||
|
||||
|
||||
public void add(TextPosition textPosition) {
|
||||
|
||||
this.textPositions.add(textPosition);
|
||||
}
|
||||
|
||||
|
||||
public float getX1() {
|
||||
|
||||
if (textPositions.get(0).getRotation() == 90) {
|
||||
return textPositions.get(0).getYDirAdj() - getTextHeight();
|
||||
} else {
|
||||
@@ -75,15 +96,20 @@ public class TextPositionSequence implements CharSequence {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public float getX2() {
|
||||
|
||||
if (textPositions.get(0).getRotation() == 90) {
|
||||
return textPositions.get(0).getYDirAdj();
|
||||
} else {
|
||||
return textPositions.get(textPositions.size() - 1).getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
|
||||
return textPositions.get(textPositions.size() - 1)
|
||||
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public float getY1() {
|
||||
|
||||
if (textPositions.get(0).getRotation() == 90) {
|
||||
return textPositions.get(0).getXDirAdj();
|
||||
} else {
|
||||
@@ -91,30 +117,46 @@ public class TextPositionSequence implements CharSequence {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public float getY2() {
|
||||
|
||||
if (textPositions.get(0).getRotation() == 90) {
|
||||
return textPositions.get(textPositions.size() - 1).getXDirAdj() + getTextHeight() -2 ;
|
||||
return textPositions.get(textPositions.size() - 1).getXDirAdj() + getTextHeight() - 2;
|
||||
} else {
|
||||
return textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() + getTextHeight();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public float getTextHeight() {
|
||||
|
||||
return textPositions.get(0).getHeightDir() + 2;
|
||||
}
|
||||
|
||||
|
||||
public float getHeight() {
|
||||
|
||||
return getY2() - getY1();
|
||||
}
|
||||
|
||||
|
||||
public float getWidth() {
|
||||
|
||||
return getX2() - getX1();
|
||||
}
|
||||
|
||||
|
||||
public String getFont() {
|
||||
return textPositions.get(0).getFont().toString().toLowerCase().replaceAll(",bold", "").replaceAll(",italic", "");
|
||||
|
||||
return textPositions.get(0)
|
||||
.getFont()
|
||||
.toString()
|
||||
.toLowerCase()
|
||||
.replaceAll(",bold", "")
|
||||
.replaceAll(",italic", "");
|
||||
}
|
||||
|
||||
|
||||
public String getFontStyle() {
|
||||
|
||||
String lowercaseFontName = textPositions.get(0).getFont().toString().toLowerCase();
|
||||
@@ -131,16 +173,51 @@ public class TextPositionSequence implements CharSequence {
|
||||
|
||||
}
|
||||
|
||||
|
||||
public float getFontSize() {
|
||||
|
||||
return textPositions.get(0).getFontSizeInPt();
|
||||
}
|
||||
|
||||
|
||||
public float getSpaceWidth() {
|
||||
|
||||
return textPositions.get(0).getWidthOfSpace();
|
||||
}
|
||||
|
||||
|
||||
public int getRotation() {
|
||||
|
||||
return textPositions.get(0).getRotation();
|
||||
}
|
||||
|
||||
|
||||
public Rectangle getRectangle() {
|
||||
|
||||
float height = textPositions.get(0).getHeightDir() + 2;
|
||||
|
||||
float posXInit;
|
||||
float posXEnd;
|
||||
float posYInit;
|
||||
float posYEnd;
|
||||
|
||||
if (textPositions.get(0).getRotation() == 90) {
|
||||
|
||||
posXEnd = textPositions.get(0).getYDirAdj() + 2;
|
||||
posXInit = textPositions.get(0).getYDirAdj() - height;
|
||||
posYInit = textPositions.get(0).getXDirAdj();
|
||||
posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
|
||||
} else {
|
||||
|
||||
posXInit = textPositions.get(0).getXDirAdj();
|
||||
posXEnd = textPositions.get(textPositions.size() - 1)
|
||||
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
|
||||
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
|
||||
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
|
||||
.getYDirAdj() + 2;
|
||||
}
|
||||
|
||||
return new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
|
||||
}
|
||||
|
||||
}
|
||||
+20
-3
@@ -1,25 +1,37 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
|
||||
public class Entity {
|
||||
|
||||
@EqualsAndHashCode.Include
|
||||
private final String word;
|
||||
private final String type;
|
||||
private boolean redaction;
|
||||
private String redactionReason;
|
||||
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
|
||||
private List<TextPositionSequence> targetSequences;
|
||||
private Integer start;
|
||||
private Integer end;
|
||||
|
||||
@EqualsAndHashCode.Include
|
||||
private String headline;
|
||||
private int matchedRule;
|
||||
|
||||
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule) {
|
||||
@EqualsAndHashCode.Include
|
||||
private int sectionNumber;
|
||||
|
||||
|
||||
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber) {
|
||||
|
||||
this.word = word;
|
||||
this.type = type;
|
||||
this.redaction = redaction;
|
||||
@@ -27,13 +39,18 @@ public class Entity {
|
||||
this.positionSequences = positionSequences;
|
||||
this.headline = headline;
|
||||
this.matchedRule = matchedRule;
|
||||
this.sectionNumber = sectionNumber;
|
||||
}
|
||||
|
||||
public Entity(String word, String type, Integer start, Integer end, String headline) {
|
||||
|
||||
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber) {
|
||||
|
||||
this.word = word;
|
||||
this.type = type;
|
||||
this.start = start;
|
||||
this.end = end;
|
||||
this.headline = headline;
|
||||
this.sectionNumber = sectionNumber;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+7
-3
@@ -2,19 +2,23 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
|
||||
@Data
|
||||
@RequiredArgsConstructor
|
||||
@AllArgsConstructor
|
||||
@EqualsAndHashCode
|
||||
public class EntityPositionSequence {
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
private List<TextPositionSequence> sequences = new ArrayList<>();
|
||||
private int pageNumber;
|
||||
private final UUID id;
|
||||
private final String id;
|
||||
|
||||
}
|
||||
|
||||
+55
-33
@@ -1,17 +1,17 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
|
||||
|
||||
@SuppressWarnings("all")
|
||||
public class SearchableText {
|
||||
|
||||
private List<TextPositionSequence> sequences = new ArrayList<>();
|
||||
private final List<TextPositionSequence> sequences = new ArrayList<>();
|
||||
|
||||
|
||||
public void add(TextPositionSequence textPositionSequence) {
|
||||
@@ -28,6 +28,14 @@ public class SearchableText {
|
||||
|
||||
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive) {
|
||||
|
||||
return getSequences(searchString, caseInsensitive, null);
|
||||
|
||||
}
|
||||
|
||||
@SuppressWarnings("checkstyle:ModifiedControlVariable")
|
||||
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive,
|
||||
List<TextPositionSequence> sequencesSubList) {
|
||||
|
||||
String normalizedSearchString;
|
||||
if (caseInsensitive) {
|
||||
normalizedSearchString = searchString.toLowerCase();
|
||||
@@ -40,37 +48,50 @@ public class SearchableText {
|
||||
|
||||
List<TextPositionSequence> crossSequenceParts = new ArrayList<>();
|
||||
List<EntityPositionSequence> finalMatches = new ArrayList<>();
|
||||
for (int i = 0; i < sequences.size(); i++) {
|
||||
TextPositionSequence partMatch = new TextPositionSequence(sequences.get(i).getPage());
|
||||
for (int j = 0; j < sequences.get(i).length(); j++) {
|
||||
|
||||
if (i > 0 && j == 0 && sequences.get(i).charAt(0, caseInsensitive) == ' ' && sequences.get(i - 1)
|
||||
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && sequences.get(i)
|
||||
.charAt(j, caseInsensitive) == ' ' && sequences.get(i).charAt(j - 1, caseInsensitive) == ' ') {
|
||||
if (j == sequences.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
|
||||
List<TextPositionSequence> searchSpace;
|
||||
if (sequencesSubList != null) {
|
||||
int subListIndex = Collections.indexOfSubList(sequences, sequencesSubList);
|
||||
if (subListIndex != -1) {
|
||||
searchSpace = sequences.subList(subListIndex, subListIndex + sequencesSubList.size());
|
||||
} else {
|
||||
searchSpace = sequences;
|
||||
}
|
||||
} else {
|
||||
searchSpace = sequences;
|
||||
}
|
||||
|
||||
for (int i = 0; i < searchSpace.size(); i++) {
|
||||
TextPositionSequence partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
|
||||
for (int j = 0; j < searchSpace.get(i).length(); j++) {
|
||||
|
||||
if (i > 0 && j == 0 && searchSpace.get(i).charAt(0, caseInsensitive) == ' ' && searchSpace.get(i - 1)
|
||||
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i).charAt(j - 1, caseInsensitive) == ' ') {
|
||||
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
|
||||
crossSequenceParts.add(partMatch);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (j == 0 && sequences.get(i).charAt(j, caseInsensitive) != ' ' && i != 0 && sequences.get(i - 1)
|
||||
.charAt(sequences.get(i - 1)
|
||||
if (j == 0 && searchSpace.get(i).charAt(j, caseInsensitive) != ' ' && i != 0 && searchSpace.get(i - 1)
|
||||
.charAt(searchSpace.get(i - 1)
|
||||
.length() - 1, caseInsensitive) != ' ' && searchChars[counter] == ' ') {
|
||||
counter++;
|
||||
}
|
||||
|
||||
if (sequences.get(i)
|
||||
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && sequences.get(i)
|
||||
if (searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) == '-') {
|
||||
|
||||
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(sequences.get(i)
|
||||
.charAt(j - 1, caseInsensitive)) || j == 0 && i != 0 && isSeparator(sequences.get(i - 1)
|
||||
.charAt(sequences.get(i - 1)
|
||||
.length() - 1, caseInsensitive)) || j == 0 && i != 0 && sequences.get(i - 1)
|
||||
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) != ' ' && sequences.get(i)
|
||||
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(searchSpace.get(i)
|
||||
.charAt(j - 1, caseInsensitive)) || j == 0 && i != 0 && isSeparator(searchSpace.get(i - 1)
|
||||
.charAt(searchSpace.get(i - 1)
|
||||
.length() - 1, caseInsensitive)) || j == 0 && i != 0 && searchSpace.get(i - 1)
|
||||
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) != ' ' && searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) != ' ') {
|
||||
partMatch.add(sequences.get(i).textPositionAt(j));
|
||||
if (!(j == sequences.get(i).length() - 1 && sequences.get(i)
|
||||
partMatch.add(searchSpace.get(i).textPositionAt(j));
|
||||
if (!(j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) == '-' && searchChars[counter] != '-')) {
|
||||
counter++;
|
||||
}
|
||||
@@ -79,19 +100,19 @@ public class SearchableText {
|
||||
if (counter == searchString.length()) {
|
||||
crossSequenceParts.add(partMatch);
|
||||
|
||||
if (i == sequences.size() - 1 && j == sequences.get(i).length() - 1 || j != sequences.get(i)
|
||||
.length() - 1 && isSeparator(sequences.get(i)
|
||||
.charAt(j + 1, caseInsensitive)) || j == sequences.get(i)
|
||||
.length() - 1 && isSeparator(sequences.get(i + 1)
|
||||
.charAt(0, caseInsensitive)) || j == sequences.get(i).length() - 1 && sequences.get(i)
|
||||
.charAt(j, caseInsensitive) != ' ' && sequences.get(i + 1)
|
||||
if (i == searchSpace.size() - 1 && j == searchSpace.get(i).length() - 1 || j != searchSpace.get(i)
|
||||
.length() - 1 && isSeparator(searchSpace.get(i)
|
||||
.charAt(j + 1, caseInsensitive)) || j == searchSpace.get(i)
|
||||
.length() - 1 && isSeparator(searchSpace.get(i + 1)
|
||||
.charAt(0, caseInsensitive)) || j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
|
||||
.charAt(j, caseInsensitive) != ' ' && searchSpace.get(i + 1)
|
||||
.charAt(0, caseInsensitive) != ' ') {
|
||||
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts));
|
||||
}
|
||||
|
||||
counter = 0;
|
||||
crossSequenceParts = new ArrayList<>();
|
||||
partMatch = new TextPositionSequence(sequences.get(i).getPage());
|
||||
partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
|
||||
}
|
||||
} else {
|
||||
counter = 0;
|
||||
@@ -99,22 +120,23 @@ public class SearchableText {
|
||||
j--;
|
||||
}
|
||||
crossSequenceParts = new ArrayList<>();
|
||||
partMatch = new TextPositionSequence(sequences.get(i).getPage());
|
||||
partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
|
||||
}
|
||||
|
||||
if (j == sequences.get(i).length() - 1 && counter != 0) {
|
||||
if (j == searchSpace.get(i).length() - 1 && counter != 0) {
|
||||
crossSequenceParts.add(partMatch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return finalMatches;
|
||||
|
||||
}
|
||||
|
||||
|
||||
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
|
||||
|
||||
UUID id = UUID.randomUUID();
|
||||
String id = IdBuilder.buildId(crossSequenceParts);
|
||||
List<EntityPositionSequence> result = new ArrayList<>();
|
||||
int currentPage = -1;
|
||||
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
|
||||
@@ -163,7 +185,7 @@ public class SearchableText {
|
||||
|
||||
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
|
||||
.replaceAll("\n", " ")
|
||||
.replaceAll(" ", " ");
|
||||
.replaceAll(" {2}", " ");
|
||||
}
|
||||
|
||||
|
||||
@@ -187,4 +209,4 @@ public class SearchableText {
|
||||
return sb.append("\n").toString();
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
+72
-15
@@ -3,15 +3,20 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.commons.lang3.StringUtils;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Data
|
||||
@Slf4j
|
||||
@Builder
|
||||
public class Section {
|
||||
|
||||
@@ -25,6 +30,28 @@ public class Section {
|
||||
|
||||
private String headline;
|
||||
|
||||
private int sectionNumber;
|
||||
|
||||
private Map<String, TextBlock> tabularData;
|
||||
|
||||
|
||||
public boolean isVertebrateStudy() {
|
||||
return tabularData != null
|
||||
&& (tabularData.containsKey("Vertebrate study Y/N")
|
||||
&& tabularData.get("Vertebrate study Y/N").getText().equals("Y")
|
||||
|| tabularData.containsKey("Verte brate study Y/N")
|
||||
&& tabularData.get("Verte brate study Y/N").getText().equals("Y"));
|
||||
}
|
||||
|
||||
|
||||
public boolean isNotVertebrateStudy() {
|
||||
return tabularData != null
|
||||
&& (tabularData.containsKey("Vertebrate study Y/N")
|
||||
&& tabularData.get("Vertebrate study Y/N").getText().equals("N")
|
||||
|| tabularData.containsKey("Verte brate study Y/N")
|
||||
&& tabularData.get("Verte brate study Y/N").getText().equals("N"));
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(String type) {
|
||||
|
||||
@@ -32,6 +59,12 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
public boolean headlineContainsWord(String word) {
|
||||
|
||||
return StringUtils.containsIgnoreCase(headline, word);
|
||||
}
|
||||
|
||||
|
||||
public void redact(String type, int ruleNumber, String reason) {
|
||||
|
||||
entities.forEach(entity -> {
|
||||
@@ -58,11 +91,15 @@ public class Section {
|
||||
|
||||
public void redactLineAfter(String start, String asType, int ruleNumber, String reason) {
|
||||
|
||||
String value = StringUtils.substringBetween(text, start, "\n");
|
||||
String[] values = StringUtils.substringsBetween(text, start, "\n");
|
||||
|
||||
if (value != null) {
|
||||
Set<Entity> found = findEntity(value.trim(), asType);
|
||||
entities.addAll(found);
|
||||
if (values != null) {
|
||||
for (String value : values) {
|
||||
if (StringUtils.isNotBlank(value)) {
|
||||
Set<Entity> found = findEntities(value.trim(), asType);
|
||||
entities.addAll(found);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TODO No need to iterate
|
||||
@@ -79,11 +116,15 @@ public class Section {
|
||||
|
||||
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason) {
|
||||
|
||||
String value = StringUtils.substringBetween(searchText, start, stop);
|
||||
String[] values = StringUtils.substringsBetween(searchText, start, stop);
|
||||
|
||||
if (value != null) {
|
||||
Set<Entity> found = findEntity(value.trim(), asType);
|
||||
entities.addAll(found);
|
||||
if (values != null) {
|
||||
for (String value : values) {
|
||||
if (StringUtils.isNotBlank(value)) {
|
||||
Set<Entity> found = findEntities(value.trim(), asType);
|
||||
entities.addAll(found);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TODO No need to iterate
|
||||
@@ -97,7 +138,7 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> findEntity(String value, String asType) {
|
||||
private Set<Entity> findEntities(String value, String asType) {
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
|
||||
@@ -109,13 +150,11 @@ public class Section {
|
||||
|
||||
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText
|
||||
.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
|
||||
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline));
|
||||
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline, sectionNumber));
|
||||
}
|
||||
} while (startIndex > -1);
|
||||
|
||||
removeEntitiesContainedInLarger(found);
|
||||
|
||||
return found;
|
||||
return removeEntitiesContainedInLarger(found);
|
||||
}
|
||||
|
||||
|
||||
@@ -125,7 +164,7 @@ public class Section {
|
||||
}
|
||||
|
||||
|
||||
public void removeEntitiesContainedInLarger(Set<Entity> entities) {
|
||||
public Set<Entity> removeEntitiesContainedInLarger(Set<Entity> entities) {
|
||||
|
||||
List<Entity> wordsToRemove = new ArrayList<>();
|
||||
for (Entity word : entities) {
|
||||
@@ -137,6 +176,24 @@ public class Section {
|
||||
}
|
||||
}
|
||||
entities.removeAll(wordsToRemove);
|
||||
return entities;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
public void highlightCell(String cellHeader, int ruleNumber, String type) {
|
||||
|
||||
TextBlock value = tabularData.get(cellHeader);
|
||||
if (value == null) {
|
||||
log.warn("Could not find any data for {}.", cellHeader);
|
||||
} else {
|
||||
Entity entity = new Entity(value.getText(), type, 0, value.getText().length(), headline, sectionNumber);
|
||||
entity.setRedaction(false);
|
||||
entity.setMatchedRule(ruleNumber);
|
||||
entity.setRedactionReason(cellHeader);
|
||||
entity.setTargetSequences(value.getSequences()); // Make sure no other cells with same content are highlighted
|
||||
entities.add(entity);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+1
-1
@@ -73,7 +73,7 @@ public class DictionaryService {
|
||||
.filter(TypeResult::isCaseInsensitive)
|
||||
.map(TypeResult::getType)
|
||||
.collect(Collectors.toList());
|
||||
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, s -> convertEntries(s)));
|
||||
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, this::convertEntries));
|
||||
defaultColor = dictionaryClient.getDefaultColor().getColor();
|
||||
}
|
||||
} catch (FeignException e) {
|
||||
|
||||
+94
-38
@@ -1,25 +1,34 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.commons.collections4.CollectionUtils;
|
||||
import org.apache.commons.lang3.StringUtils;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class EntityRedactionService {
|
||||
@@ -28,12 +37,13 @@ public class EntityRedactionService {
|
||||
private final DroolsExecutionService droolsExecutionService;
|
||||
|
||||
|
||||
public void processDocument(Document classifiedDoc) {
|
||||
public void processDocument(Document classifiedDoc, ManualRedactions manualRedactions) {
|
||||
|
||||
dictionaryService.updateDictionary();
|
||||
droolsExecutionService.updateRules();
|
||||
|
||||
Set<Entity> documentEntities = new HashSet<>();
|
||||
int sectionNumber = 1;
|
||||
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
|
||||
|
||||
SearchableText searchableText = paragraph.getSearchableText();
|
||||
@@ -43,86 +53,113 @@ public class EntityRedactionService {
|
||||
for (Table table : tables) {
|
||||
for (List<Cell> row : table.getRows()) {
|
||||
SearchableText searchableRow = new SearchableText();
|
||||
for (Cell column : row) {
|
||||
if (column == null || column.getTextBlocks() == null) {
|
||||
Map<String, TextBlock> tabularData = new HashMap<>();
|
||||
for (Cell cell : row) {
|
||||
if (cell.isHeaderCell() || CollectionUtils.isEmpty(cell.getTextBlocks())) {
|
||||
continue;
|
||||
}
|
||||
for (TextBlock textBlock : column.getTextBlocks()) {
|
||||
addSectionToManualRedactions(cell.getTextBlocks(), manualRedactions, table.getHeadline(), sectionNumber);
|
||||
cell.getHeaderCells().forEach(headerCell -> {
|
||||
String headerName = headerCell.getTextBlocks().get(0).getText()
|
||||
.replaceAll("\n", " ")
|
||||
.replaceAll(" ", " ");
|
||||
tabularData.put(headerName, cell.getTextBlocks().get(0));
|
||||
});
|
||||
for (TextBlock textBlock : cell.getTextBlocks()) {
|
||||
searchableRow.addAll(textBlock.getSequences());
|
||||
}
|
||||
|
||||
}
|
||||
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline());
|
||||
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber);
|
||||
|
||||
Section analysedRowSection = droolsExecutionService.executeRules(Section.builder()
|
||||
.entities(rowEntities)
|
||||
.text(searchableRow.getAsStringWithLinebreaks())
|
||||
.searchText(searchableRow.toString())
|
||||
.headline(table.getHeadline())
|
||||
.sectionNumber(sectionNumber)
|
||||
.tabularData(tabularData)
|
||||
.build());
|
||||
|
||||
for (Entity entity : analysedRowSection.getEntities()) {
|
||||
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
|
||||
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), true));
|
||||
} else {
|
||||
entity.setPositionSequences(searchableRow.getSequences(entity.getWord(), false));
|
||||
}
|
||||
}
|
||||
documentEntities.addAll(analysedRowSection.getEntities());
|
||||
documentEntities.addAll(clearAndFindPositions(analysedRowSection.getEntities(), searchableRow));
|
||||
sectionNumber++;
|
||||
}
|
||||
sectionNumber++;
|
||||
}
|
||||
|
||||
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline());
|
||||
addSectionToManualRedactions(paragraph.getTextBlocks(), manualRedactions, paragraph.getHeadline(), sectionNumber);
|
||||
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline(), sectionNumber);
|
||||
Section analysedSection = droolsExecutionService.executeRules(Section.builder()
|
||||
.entities(entities)
|
||||
.text(searchableText.getAsStringWithLinebreaks())
|
||||
.searchText(searchableText.toString())
|
||||
.headline(paragraph.getHeadline())
|
||||
.sectionNumber(sectionNumber)
|
||||
.build());
|
||||
|
||||
for (Entity entity : analysedSection.getEntities()) {
|
||||
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
|
||||
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), true));
|
||||
} else {
|
||||
entity.setPositionSequences(searchableText.getSequences(entity.getWord(), false));
|
||||
}
|
||||
}
|
||||
|
||||
documentEntities.addAll(analysedSection.getEntities());
|
||||
documentEntities.addAll(clearAndFindPositions(analysedSection.getEntities(), searchableText));
|
||||
sectionNumber++;
|
||||
}
|
||||
|
||||
documentEntities.forEach(entity -> {
|
||||
entity.getPositionSequences().forEach(sequence -> {
|
||||
for (Entity entity : documentEntities) {
|
||||
Map<Integer, List<EntityPositionSequence>> sequenceOnPage = new HashMap<>();
|
||||
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
|
||||
sequenceOnPage.computeIfAbsent(entityPositionSequence.getPageNumber(), (x) -> new ArrayList<>())
|
||||
.add(entityPositionSequence);
|
||||
}
|
||||
|
||||
for (Map.Entry<Integer, List<EntityPositionSequence>> entry : sequenceOnPage.entrySet()) {
|
||||
classifiedDoc.getEntities()
|
||||
.computeIfAbsent(sequence.getPageNumber(), (x) -> new HashSet<>())
|
||||
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), List
|
||||
.of(sequence), entity.getHeadline(), entity.getMatchedRule()));
|
||||
});
|
||||
});
|
||||
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
|
||||
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
|
||||
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber()));
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> findEntities(SearchableText searchableText, String headline) {
|
||||
private Set<Entity> clearAndFindPositions(Set<Entity> entities, SearchableText text) {
|
||||
|
||||
removeEntitiesContainedInLarger(entities);
|
||||
|
||||
for (Entity entity : entities) {
|
||||
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
|
||||
entity.setPositionSequences(text.getSequences(entity.getWord(), true, entity.getTargetSequences()));
|
||||
} else {
|
||||
entity.setPositionSequences(text.getSequences(entity.getWord(), false, entity.getTargetSequences()));
|
||||
}
|
||||
}
|
||||
|
||||
return entities;
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber) {
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
if (StringUtils.isEmpty(searchableText.toString()) && StringUtils.isEmpty(headline)) {
|
||||
return found;
|
||||
}
|
||||
|
||||
String inputString = searchableText.toString();
|
||||
String lowercaseInputString = inputString.toLowerCase();
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
for (Map.Entry<String, Set<String>> entry : dictionaryService.getDictionary().entrySet()) {
|
||||
|
||||
if (dictionaryService.getCaseInsensitiveTypes().contains(entry.getKey())) {
|
||||
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline));
|
||||
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
|
||||
} else {
|
||||
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline));
|
||||
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
|
||||
}
|
||||
}
|
||||
|
||||
removeEntitiesContainedInLarger(found);
|
||||
|
||||
return found;
|
||||
|
||||
}
|
||||
|
||||
|
||||
private Set<Entity> find(String inputString, Set<String> values, String type, String headline) {
|
||||
private Set<Entity> find(String inputString, Set<String> values, String type, String headline, int sectionNumber) {
|
||||
|
||||
Set<Entity> found = new HashSet<>();
|
||||
for (String value : values) {
|
||||
@@ -134,7 +171,7 @@ public class EntityRedactionService {
|
||||
|
||||
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(inputString.charAt(startIndex - 1)) || isSeparator(inputString
|
||||
.charAt(startIndex - 1))) && (stopIndex == inputString.length() || isSeparator(inputString.charAt(stopIndex)))) {
|
||||
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline));
|
||||
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline, sectionNumber));
|
||||
}
|
||||
} while (startIndex > -1);
|
||||
}
|
||||
@@ -162,4 +199,23 @@ public class EntityRedactionService {
|
||||
entities.removeAll(wordsToRemove);
|
||||
}
|
||||
|
||||
|
||||
private void addSectionToManualRedactions(List<TextBlock> textBlocks, ManualRedactions manualRedactions, String section, int sectionNumber) {
|
||||
|
||||
if (manualRedactions == null || manualRedactions.getEntriesToAdd().isEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (TextBlock textBlock : textBlocks) {
|
||||
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
|
||||
for (Rectangle rectangle : manualRedactionEntry.getPositions()) {
|
||||
if (textBlock.contains(rectangle)) {
|
||||
manualRedactionEntry.setSection(section);
|
||||
manualRedactionEntry.setSectionNumber(sectionNumber);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+32
@@ -0,0 +1,32 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.utils;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.List;
|
||||
|
||||
import com.google.common.hash.HashFunction;
|
||||
import com.google.common.hash.Hashing;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.experimental.UtilityClass;
|
||||
|
||||
@UtilityClass
|
||||
public class IdBuilder {
|
||||
|
||||
private final HashFunction hashFunction = Hashing.murmur3_128();
|
||||
|
||||
public String buildId(List<TextPositionSequence> crossSequenceParts) {
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
crossSequenceParts.forEach(sequencePart -> sequencePart.getTextPositions().forEach(textPosition -> {
|
||||
sb.append(textPosition.getTextMatrix());
|
||||
}));
|
||||
|
||||
return hashFunction.hashString(sb.toString(), StandardCharsets.UTF_8).toString();
|
||||
}
|
||||
|
||||
|
||||
public String buildId(ManualRedactionEntry manualRedactionEntry) {
|
||||
return hashFunction.hashString(manualRedactionEntry.toString(), StandardCharsets.UTF_8).toString();
|
||||
}
|
||||
}
|
||||
-1
@@ -29,7 +29,6 @@ import lombok.extern.slf4j.Slf4j;
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
@SuppressWarnings("PMD")
|
||||
public class PdfSegmentationService {
|
||||
|
||||
private final RulingCleaningService rulingCleaningService;
|
||||
|
||||
+77
-16
@@ -1,9 +1,11 @@
|
||||
package com.iqser.red.service.redaction.v1.server.segmentation;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.Iterator;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.commons.collections4.CollectionUtils;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
@@ -11,10 +13,10 @@ import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
@Service
|
||||
@SuppressWarnings("all")
|
||||
public class SectionsBuilderService {
|
||||
|
||||
public void buildSections(Document document) {
|
||||
@@ -25,6 +27,7 @@ public class SectionsBuilderService {
|
||||
AbstractTextContainer prev = null;
|
||||
|
||||
String lastHeadline = "";
|
||||
Table previousTable = null;
|
||||
for (Page page : document.getPages()) {
|
||||
for (AbstractTextContainer current : page.getTextBlocks()) {
|
||||
|
||||
@@ -36,32 +39,30 @@ public class SectionsBuilderService {
|
||||
current.setPage(page.getPageNumber());
|
||||
|
||||
if (prev != null && current.getClassification().startsWith("H ") || !document.isHeadlines()) {
|
||||
|
||||
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
|
||||
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
|
||||
chunkBlock.setHeadline(lastHeadline);
|
||||
lastHeadline = current.getText();
|
||||
if (CollectionUtils.isNotEmpty(chunkBlock.getTables())) {
|
||||
previousTable = chunkBlock.getTables().get(0);
|
||||
}
|
||||
chunkBlockList.add(chunkBlock);
|
||||
chunkWords = new ArrayList<>();
|
||||
|
||||
}
|
||||
|
||||
chunkWords.add(current);
|
||||
|
||||
prev = current;
|
||||
}
|
||||
}
|
||||
|
||||
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
|
||||
if (chunkBlock != null) {
|
||||
chunkBlockList.add(chunkBlock);
|
||||
chunkBlock.setHeadline(lastHeadline);
|
||||
}
|
||||
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
|
||||
chunkBlock.setHeadline(lastHeadline);
|
||||
chunkBlockList.add(chunkBlock);
|
||||
|
||||
document.setParagraphs(chunkBlockList);
|
||||
}
|
||||
|
||||
|
||||
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline) {
|
||||
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline, Table previousTable) {
|
||||
|
||||
Paragraph paragraph = new Paragraph();
|
||||
TextBlock textBlock = null;
|
||||
@@ -76,19 +77,41 @@ public class SectionsBuilderService {
|
||||
AbstractTextContainer container = itty.next();
|
||||
|
||||
if (container instanceof Table) {
|
||||
Table table = (Table) container;
|
||||
splitByTable = true;
|
||||
|
||||
if (previous != null && previous instanceof TextBlock && previous.getText().startsWith("Table ")) {
|
||||
((Table) container).setHeadline(previous.getText());
|
||||
if (previous != null && previous.getText().startsWith("Table ")) {
|
||||
table.setHeadline(previous.getText());
|
||||
} else {
|
||||
((Table) container).setHeadline("Table in: " + lastHeadline);
|
||||
table.setHeadline("Table in: " + lastHeadline);
|
||||
}
|
||||
// Distribute header information for subsequent tables
|
||||
if (previousTable != null && hasInvalidHeaderInformation(table) && hasValidHeaderInformation(previousTable)) {
|
||||
List<Cell> previousTableNonHeaderRow = getRowWithNonHeaderCells(previousTable);
|
||||
List<Cell> tableNonHeaderRow = getRowWithNonHeaderCells(table);
|
||||
// Allow merging of tables if header row is separated from first logical non-header row
|
||||
if (previousTableNonHeaderRow.isEmpty() && previousTable.getRowCount() == 1
|
||||
&& previousTable.getRows().get(0).size() == tableNonHeaderRow.size()) {
|
||||
previousTableNonHeaderRow = previousTable.getRows().get(0);
|
||||
}
|
||||
if (previousTableNonHeaderRow.size() == tableNonHeaderRow.size()) {
|
||||
for (int i = table.getRows().size() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
|
||||
List<Cell> row = table.getRows().get(i);
|
||||
if (row.size() == tableNonHeaderRow.size()
|
||||
&& row.stream().allMatch(cell -> cell.getHeaderCells().isEmpty())) {
|
||||
for (int j = 0; j < row.size(); j++) {
|
||||
row.get(j).setHeaderCells(previousTableNonHeaderRow.get(j).getHeaderCells());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (textBlock != null && !alreadyAdded) {
|
||||
paragraph.getPageBlocks().add(textBlock);
|
||||
alreadyAdded = true;
|
||||
}
|
||||
paragraph.getPageBlocks().add(container);
|
||||
paragraph.getPageBlocks().add(table);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -125,4 +148,42 @@ public class SectionsBuilderService {
|
||||
return paragraph;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
private boolean hasValidHeaderInformation(Table table) {
|
||||
|
||||
return !hasInvalidHeaderInformation(table);
|
||||
}
|
||||
|
||||
|
||||
private boolean hasInvalidHeaderInformation(Table table) {
|
||||
|
||||
return table.getRows().stream()
|
||||
.flatMap(row -> row.stream()
|
||||
.filter(cell -> CollectionUtils.isNotEmpty(cell.getHeaderCells())))
|
||||
.findAny()
|
||||
.isEmpty();
|
||||
|
||||
}
|
||||
|
||||
|
||||
private List<Cell> getRowWithNonHeaderCells(Table table) {
|
||||
|
||||
for (int i = table.getRows().size() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
|
||||
List<Cell> row = table.getRows().get(i);
|
||||
boolean allNonHeader = true;
|
||||
for (Cell cell : row) {
|
||||
if (cell.isHeaderCell()) {
|
||||
allNonHeader = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (allNonHeader) {
|
||||
return row;
|
||||
}
|
||||
}
|
||||
|
||||
return Collections.emptyList();
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+6
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.server.tableextraction.model;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
@@ -22,4 +24,8 @@ public abstract class AbstractTextContainer {
|
||||
return this.minX <= other.minX && this.maxX >= other.maxX && this.minY >= other.minY && this.maxY <= other.maxY;
|
||||
}
|
||||
|
||||
public boolean contains(Rectangle other) {
|
||||
return page == other.getPage() && this.minX <= other.getTopLeft().getX() && this.maxX >= other.getTopLeft().getX() + other.getWidth() && this.minY <= other.getTopLeft().getY() && this.maxY >= other.getTopLeft().getY() + other.getHeight();
|
||||
}
|
||||
|
||||
}
|
||||
+12
-2
@@ -16,11 +16,21 @@ public class Cell extends Rectangle {
|
||||
|
||||
private List<TextBlock> textBlocks = new ArrayList<>();
|
||||
|
||||
private List<Cell> headerCells = new ArrayList<>();
|
||||
|
||||
private boolean isHeaderCell;
|
||||
|
||||
public Cell(Point2D topLeft, Point2D bottomRight) {
|
||||
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight.getY() - topLeft.getY()));
|
||||
|
||||
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()),
|
||||
(float) (bottomRight
|
||||
.getY() - topLeft.getY()));
|
||||
}
|
||||
|
||||
|
||||
public void addTextBlock(TextBlock textBlock) {
|
||||
|
||||
textBlocks.add(textBlock);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package com.iqser.red.service.redaction.v1.server.tableextraction.model;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.Value;
|
||||
|
||||
@Value
|
||||
@RequiredArgsConstructor
|
||||
public class CellPosition implements Comparable<CellPosition> {
|
||||
|
||||
int row;
|
||||
|
||||
int col;
|
||||
|
||||
|
||||
@Override
|
||||
public int compareTo(CellPosition other) {
|
||||
|
||||
int rowDiff = row - other.row;
|
||||
return rowDiff != 0 ? rowDiff : col - other.col;
|
||||
}
|
||||
|
||||
}
|
||||
+17
-10
@@ -8,25 +8,28 @@ import org.locationtech.jts.index.strtree.STRtree;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
|
||||
|
||||
|
||||
@SuppressWarnings("all")
|
||||
public class RectangleSpatialIndex<T extends Rectangle> {
|
||||
|
||||
|
||||
private final STRtree si = new STRtree();
|
||||
private final List<T> rectangles = new ArrayList<>();
|
||||
|
||||
|
||||
public void add(T te) {
|
||||
|
||||
rectangles.add(te);
|
||||
si.insert(new Envelope(te.getLeft(), te.getRight(), te.getBottom(), te.getTop()), te);
|
||||
}
|
||||
|
||||
public List<T> contains(Rectangle r) {
|
||||
List<T> intersection = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
|
||||
|
||||
|
||||
public List<T> contains(Rectangle rectangle) {
|
||||
|
||||
List<T> intersection = si.query(new Envelope(rectangle.getLeft(), rectangle.getRight(), rectangle.getTop(), rectangle
|
||||
.getBottom()));
|
||||
List<T> rv = new ArrayList<T>();
|
||||
|
||||
for (T ir: intersection) {
|
||||
if (r.contains(ir)) {
|
||||
for (T ir : intersection) {
|
||||
if (rectangle.contains(ir)) {
|
||||
rv.add(ir);
|
||||
}
|
||||
}
|
||||
@@ -34,18 +37,22 @@ public class RectangleSpatialIndex<T extends Rectangle> {
|
||||
Utils.sort(rv, Rectangle.ILL_DEFINED_ORDER);
|
||||
return rv;
|
||||
}
|
||||
|
||||
|
||||
|
||||
public List<T> intersects(Rectangle r) {
|
||||
|
||||
List rv = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
|
||||
return rv;
|
||||
}
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* Minimum bounding box of all the Rectangles contained on this RectangleSpatialIndex
|
||||
*
|
||||
*
|
||||
* @return a Rectangle
|
||||
*/
|
||||
public Rectangle getBounds() {
|
||||
|
||||
return Rectangle.boundingBoxOf(rectangles);
|
||||
}
|
||||
|
||||
|
||||
+154
-83
@@ -9,31 +9,36 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.TreeMap;
|
||||
|
||||
import org.apache.commons.collections4.CollectionUtils;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@SuppressWarnings("all")
|
||||
@Slf4j
|
||||
public class Table extends AbstractTextContainer {
|
||||
|
||||
private final TreeMap<CellPosition, Cell> cells = new TreeMap<>();
|
||||
|
||||
private RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
|
||||
private final RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
|
||||
|
||||
@Getter
|
||||
@Setter
|
||||
private String headline;
|
||||
|
||||
@Getter
|
||||
private int rowCount = 0;
|
||||
private int rowCount;
|
||||
|
||||
@Getter
|
||||
private int colCount = 0;
|
||||
private int colCount;
|
||||
|
||||
private int rotation = 0;
|
||||
private final int rotation;
|
||||
|
||||
private List<List<Cell>> rows;
|
||||
|
||||
private List<List<Cell>> memoizedRows = null;
|
||||
|
||||
public Table(List<Cell> cells, Rectangle area, int rotation) {
|
||||
|
||||
@@ -47,16 +52,123 @@ public class Table extends AbstractTextContainer {
|
||||
|
||||
}
|
||||
|
||||
|
||||
public List<List<Cell>> getRows() {
|
||||
|
||||
if (memoizedRows == null) {
|
||||
memoizedRows = computeRows();
|
||||
if (rows == null) {
|
||||
rows = computeRows();
|
||||
computeHeaders();
|
||||
}
|
||||
|
||||
return memoizedRows;
|
||||
return rows;
|
||||
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Detect header cells (either first row or first column):
|
||||
* Column is marked as header if cell text is bold and row cell text is not bold.
|
||||
* Defaults to row.
|
||||
*/
|
||||
private void computeHeaders() {
|
||||
|
||||
// A bold cell is a header cell as long as every cell to the left/top is bold, too
|
||||
cells.forEach((position, cell) -> {
|
||||
List<Cell> cellsToTheLeft = getCellsToTheLeft(position);
|
||||
Cell lastHeaderCell = null;
|
||||
for (Cell leftCell : cellsToTheLeft) {
|
||||
if (CollectionUtils.isNotEmpty(leftCell.getTextBlocks()) && leftCell.getTextBlocks()
|
||||
.get(0)
|
||||
.getMostPopularWordStyle()
|
||||
.equals("bold")) {
|
||||
lastHeaderCell = leftCell;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (lastHeaderCell != null) {
|
||||
cell.getHeaderCells().add(lastHeaderCell);
|
||||
}
|
||||
lastHeaderCell = null;
|
||||
List<Cell> cellsToTheTop = getCellToTheTop(position);
|
||||
for (Cell topCell : cellsToTheTop) {
|
||||
if (CollectionUtils.isNotEmpty(topCell.getTextBlocks()) && topCell.getTextBlocks()
|
||||
.get(0)
|
||||
.getMostPopularWordStyle()
|
||||
.equals("bold")) {
|
||||
lastHeaderCell = topCell;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (lastHeaderCell != null) {
|
||||
cell.getHeaderCells().add(lastHeaderCell);
|
||||
}
|
||||
if (CollectionUtils.isNotEmpty(cell.getTextBlocks()) && cell.getTextBlocks()
|
||||
.get(0)
|
||||
.getMostPopularWordStyle()
|
||||
.equals("bold")) {
|
||||
cell.setHeaderCell(true);
|
||||
}
|
||||
});
|
||||
|
||||
}
|
||||
|
||||
|
||||
private List<Cell> getCellsToTheLeft(CellPosition cellPosition) {
|
||||
|
||||
List<Cell> result = new ArrayList<>();
|
||||
if (cellPosition.getCol() == 0) {
|
||||
return result;
|
||||
}
|
||||
int row = cellPosition.getRow();
|
||||
for (int i = cellPosition.getCol() - 1; i >= 0; i--) {
|
||||
if (cells.get(new CellPosition(row, i)) != null) {
|
||||
result.add(cells.get(new CellPosition(row, i)));
|
||||
} else {
|
||||
Cell spanningCell = null;
|
||||
while (spanningCell == null && row >= 0) {
|
||||
row--;
|
||||
spanningCell = cells.get(new CellPosition(row, i));
|
||||
}
|
||||
if (spanningCell != null) {
|
||||
result.add(spanningCell);
|
||||
}
|
||||
row = cellPosition.getRow();
|
||||
}
|
||||
}
|
||||
Collections.reverse(result);
|
||||
return result;
|
||||
}
|
||||
|
||||
|
||||
private List<Cell> getCellToTheTop(CellPosition cellPosition) {
|
||||
|
||||
List<Cell> result = new ArrayList<>();
|
||||
if (cellPosition.getRow() == 0) {
|
||||
return result;
|
||||
}
|
||||
int col = cellPosition.getCol();
|
||||
for (int i = cellPosition.getRow() - 1; i >= 0; i--) {
|
||||
if (cells.get(new CellPosition(i, col)) != null) {
|
||||
result.add(cells.get(new CellPosition(i, col)));
|
||||
} else {
|
||||
Cell spanningCell = null;
|
||||
while (spanningCell == null && col >= 0) {
|
||||
col--;
|
||||
spanningCell = cells.get(new CellPosition(i, col));
|
||||
}
|
||||
if (spanningCell != null) {
|
||||
result.add(spanningCell);
|
||||
}
|
||||
col = cellPosition.getCol();
|
||||
}
|
||||
}
|
||||
Collections.reverse(result);
|
||||
return result;
|
||||
}
|
||||
|
||||
|
||||
private List<List<Cell>> computeRows() {
|
||||
|
||||
List<List<Cell>> rows = new ArrayList<>();
|
||||
@@ -65,7 +177,9 @@ public class Table extends AbstractTextContainer {
|
||||
List<Cell> lastRow = new ArrayList<>();
|
||||
for (int j = rowCount - 1; j >= 0; j--) { // cols
|
||||
Cell cell = cells.get(new CellPosition(j, i));
|
||||
lastRow.add(cell);
|
||||
if (cell != null) {
|
||||
lastRow.add(cell);
|
||||
}
|
||||
}
|
||||
rows.add(lastRow);
|
||||
}
|
||||
@@ -74,7 +188,9 @@ public class Table extends AbstractTextContainer {
|
||||
List<Cell> lastRow = new ArrayList<>();
|
||||
for (int j = 0; j < rowCount; j++) { // cols
|
||||
Cell cell = cells.get(new CellPosition(i, j));
|
||||
lastRow.add(cell);
|
||||
if (cell != null) {
|
||||
lastRow.add(cell);
|
||||
}
|
||||
}
|
||||
rows.add(lastRow);
|
||||
}
|
||||
@@ -83,7 +199,9 @@ public class Table extends AbstractTextContainer {
|
||||
List<Cell> lastRow = new ArrayList<>();
|
||||
for (int j = 0; j < colCount; j++) {
|
||||
Cell cell = cells.get(new CellPosition(i, j)); // JAVA_8 use getOrDefault()
|
||||
lastRow.add(cell);
|
||||
if (cell != null) {
|
||||
lastRow.add(cell);
|
||||
}
|
||||
}
|
||||
rows.add(lastRow);
|
||||
}
|
||||
@@ -93,7 +211,8 @@ public class Table extends AbstractTextContainer {
|
||||
|
||||
}
|
||||
|
||||
public void add(Cell chunk, int row, int col) {
|
||||
|
||||
private void add(Cell chunk, int row, int col) {
|
||||
|
||||
rowCount = Math.max(rowCount, row + 1);
|
||||
colCount = Math.max(colCount, col + 1);
|
||||
@@ -103,6 +222,7 @@ public class Table extends AbstractTextContainer {
|
||||
|
||||
}
|
||||
|
||||
|
||||
private void addCells(List<Cell> cells) {
|
||||
|
||||
if (cells.isEmpty()) {
|
||||
@@ -131,25 +251,21 @@ public class Table extends AbstractTextContainer {
|
||||
while (rowCells.hasNext()) {
|
||||
Cell cell = rowCells.next();
|
||||
if (i > 0) {
|
||||
List<List<Cell>> others = rowsOfCells(
|
||||
si.contains(
|
||||
new Rectangle(cell.getBottom(),
|
||||
si.getBounds().getLeft(),
|
||||
cell.getLeft() - si.getBounds().getLeft() + 1,
|
||||
si.getBounds().getBottom() - cell.getBottom()
|
||||
)
|
||||
));
|
||||
Rectangle rectangle = new Rectangle(cell.getBottom(),
|
||||
si.getBounds().getLeft(),
|
||||
cell.getLeft() - si.getBounds().getLeft() + 1,
|
||||
si.getBounds().getBottom() - cell.getBottom());
|
||||
List<List<Cell>> others = rowsOfCells(si.contains(rectangle));
|
||||
|
||||
for (List<Cell> r : others) {
|
||||
jumpToColumn = Math.max(jumpToColumn, r.size());
|
||||
}
|
||||
}
|
||||
|
||||
while (startColumn != jumpToColumn) {
|
||||
add(previousNonNullCellForColumnIndex.get(startColumn), i, startColumn);
|
||||
startColumn++;
|
||||
while (startColumn != jumpToColumn) {
|
||||
add(previousNonNullCellForColumnIndex.get(startColumn), i, startColumn);
|
||||
startColumn++;
|
||||
}
|
||||
}
|
||||
|
||||
add(cell, i, startColumn);
|
||||
previousNonNullCellForColumnIndex.put(startColumn, cell);
|
||||
startColumn++;
|
||||
@@ -158,34 +274,24 @@ public class Table extends AbstractTextContainer {
|
||||
}
|
||||
}
|
||||
|
||||
private static List<List<Cell>> rowsOfCells(List<Cell> cells) {
|
||||
Cell c;
|
||||
float lastTop;
|
||||
|
||||
private List<List<Cell>> rowsOfCells(List<Cell> cells) {
|
||||
|
||||
List<List<Cell>> rv = new ArrayList<>();
|
||||
List<Cell> lastRow;
|
||||
|
||||
if (cells.isEmpty()) {
|
||||
return rv;
|
||||
}
|
||||
cells.sort(Comparator.comparingDouble(Rectangle::getLeft));
|
||||
|
||||
Collections.sort(cells, new Comparator<Cell>() {
|
||||
@Override
|
||||
public int compare(Cell arg0, Cell arg1) {
|
||||
return Double.compare(arg0.getLeft(), arg1.getLeft());
|
||||
}
|
||||
});
|
||||
|
||||
Collections.sort(cells, Collections.reverseOrder(new Comparator<Cell>() {
|
||||
@Override
|
||||
public int compare(Cell arg0, Cell arg1) {
|
||||
return Float.compare(Utils.round(arg0.getBottom(), 2), Utils.round(arg1.getBottom(),2));
|
||||
}
|
||||
}));
|
||||
cells.sort(Collections.reverseOrder((arg0, arg1) -> Float.compare(Utils.round(arg0.getBottom(), 2),
|
||||
Utils.round(arg1
|
||||
.getBottom(), 2))));
|
||||
|
||||
Iterator<Cell> iter = cells.iterator();
|
||||
c = iter.next();
|
||||
lastTop = c.getBottom();
|
||||
lastRow = new ArrayList<>();
|
||||
Cell c = iter.next();
|
||||
float lastTop = c.getBottom();
|
||||
List<Cell> lastRow = new ArrayList<>();
|
||||
lastRow.add(c);
|
||||
rv.add(lastRow);
|
||||
|
||||
@@ -201,6 +307,7 @@ public class Table extends AbstractTextContainer {
|
||||
return rv;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String getText() {
|
||||
|
||||
@@ -237,6 +344,7 @@ public class Table extends AbstractTextContainer {
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
|
||||
public String getTextAsHtml() {
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
@@ -270,41 +378,4 @@ public class Table extends AbstractTextContainer {
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
class CellPosition implements Comparable<CellPosition> {
|
||||
|
||||
CellPosition(int row, int col) {
|
||||
this.row = row;
|
||||
this.col = col;
|
||||
}
|
||||
|
||||
final int row, col;
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
return row + 101 * col;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean equals(Object obj) {
|
||||
if (this == obj) {
|
||||
return true;
|
||||
}
|
||||
if (obj == null) {
|
||||
return false;
|
||||
}
|
||||
if (getClass() != obj.getClass()) {
|
||||
return false;
|
||||
}
|
||||
CellPosition other = (CellPosition) obj;
|
||||
return row == other.row && col == other.col;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int compareTo(CellPosition other) {
|
||||
int rowdiff = row - other.row;
|
||||
return rowdiff != 0 ? rowdiff : col - other.col;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+72
-78
@@ -2,7 +2,6 @@ package com.iqser.red.service.redaction.v1.server.tableextraction.service;
|
||||
|
||||
import java.awt.geom.Point2D;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
@@ -25,26 +24,28 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
|
||||
|
||||
@Service
|
||||
@SuppressWarnings("all")
|
||||
public class TableExtractionService {
|
||||
|
||||
public void extractTables(CleanRulings cleanRulings, Page page){
|
||||
public void extractTables(CleanRulings cleanRulings, Page page) {
|
||||
|
||||
List<Cell> cells = findCells(cleanRulings.getHorizontal(), cleanRulings.getVertical());
|
||||
|
||||
Iterator<AbstractTextContainer> itty = page.getTextBlocks().iterator();
|
||||
while (itty.hasNext()) {
|
||||
TextBlock textBlock = (TextBlock) itty.next();
|
||||
for (AbstractTextContainer abstractTextContainer : page.getTextBlocks()) {
|
||||
TextBlock textBlock = (TextBlock) abstractTextContainer;
|
||||
for (Cell cell : cells) {
|
||||
if (cell.intersects(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight())) {
|
||||
if (cell.intersects(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(),
|
||||
textBlock.getHeight())) {
|
||||
cell.addTextBlock(textBlock);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells)
|
||||
.stream()
|
||||
cells = new ArrayList<>(new HashSet<>(cells));
|
||||
Utils.sort(cells, Rectangle.ILL_DEFINED_ORDER);
|
||||
|
||||
|
||||
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells).stream()
|
||||
.filter(r -> r.getWidth() > 0f && r.getHeight() > 0f)
|
||||
.collect(Collectors.toList());
|
||||
|
||||
@@ -63,9 +64,9 @@ public class TableExtractionService {
|
||||
for (Table table : tables) {
|
||||
int position = -1;
|
||||
|
||||
itty = page.getTextBlocks().iterator();
|
||||
Iterator<AbstractTextContainer> itty = page.getTextBlocks().iterator();
|
||||
while (itty.hasNext()) {
|
||||
AbstractTextContainer textBlock = (AbstractTextContainer) itty.next();
|
||||
AbstractTextContainer textBlock = itty.next();
|
||||
if (table.contains(textBlock)) {
|
||||
if (position == -1) {
|
||||
position = page.getTextBlocks().indexOf(textBlock);
|
||||
@@ -79,17 +80,18 @@ public class TableExtractionService {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public List<Cell> findCells(List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
|
||||
|
||||
List<Cell> cellsFound = new ArrayList<>();
|
||||
Map<Point2D, Ruling[]> intersectionPoints = Ruling.findIntersections(horizontalRulingLines, verticalRulingLines);
|
||||
Map<Point2D, Ruling[]> intersectionPoints = Ruling.findIntersections(horizontalRulingLines,
|
||||
verticalRulingLines);
|
||||
List<Point2D> intersectionPointsList = new ArrayList<>(intersectionPoints.keySet());
|
||||
Collections.sort(intersectionPointsList, POINT_COMPARATOR);
|
||||
boolean doBreak;
|
||||
intersectionPointsList.sort(POINT_COMPARATOR);
|
||||
|
||||
for (int i = 0; i < intersectionPointsList.size(); i++) {
|
||||
Point2D topLeft = intersectionPointsList.get(i);
|
||||
Ruling[] hv = intersectionPoints.get(topLeft);
|
||||
doBreak = false;
|
||||
|
||||
// CrossingPointsDirectlyBelow( topLeft );
|
||||
List<Point2D> xPoints = new ArrayList<>();
|
||||
@@ -106,10 +108,6 @@ public class TableExtractionService {
|
||||
}
|
||||
outer:
|
||||
for (Point2D xPoint : xPoints) {
|
||||
if (doBreak) {
|
||||
break;
|
||||
}
|
||||
|
||||
// is there a vertical edge b/w topLeft and xPoint?
|
||||
if (!hv[1].equals(intersectionPoints.get(xPoint)[1])) {
|
||||
continue;
|
||||
@@ -120,11 +118,9 @@ public class TableExtractionService {
|
||||
continue;
|
||||
}
|
||||
Point2D btmRight = new Point2D.Float((float) yPoint.getX(), (float) xPoint.getY());
|
||||
if (intersectionPoints.containsKey(btmRight)
|
||||
&& intersectionPoints.get(btmRight)[0].equals(intersectionPoints.get(xPoint)[0])
|
||||
&& intersectionPoints.get(btmRight)[1].equals(intersectionPoints.get(yPoint)[1])) {
|
||||
if (intersectionPoints.containsKey(btmRight) && intersectionPoints.get(btmRight)[0].equals(intersectionPoints
|
||||
.get(xPoint)[0]) && intersectionPoints.get(btmRight)[1].equals(intersectionPoints.get(yPoint)[1])) {
|
||||
cellsFound.add(new Cell(topLeft, btmRight));
|
||||
doBreak = true;
|
||||
break outer;
|
||||
}
|
||||
}
|
||||
@@ -139,7 +135,7 @@ public class TableExtractionService {
|
||||
}
|
||||
|
||||
|
||||
public List<Rectangle> findSpreadsheetsFromCells(List<? extends Rectangle> cells) {
|
||||
private List<Rectangle> findSpreadsheetsFromCells(List<? extends Rectangle> cells) {
|
||||
// via: http://stackoverflow.com/questions/13746284/merging-multiple-adjacent-rectangles-into-one-polygon
|
||||
List<Rectangle> rectangles = new ArrayList<>();
|
||||
Set<Point2D> pointSet = new HashSet<>();
|
||||
@@ -147,10 +143,6 @@ public class TableExtractionService {
|
||||
Map<Point2D, Point2D> edgesV = new HashMap<>();
|
||||
int i = 0;
|
||||
|
||||
cells = new ArrayList<>(new HashSet<>(cells));
|
||||
|
||||
Utils.sort(cells, Rectangle.ILL_DEFINED_ORDER);
|
||||
|
||||
for (Rectangle cell : cells) {
|
||||
for (Point2D pt : cell.getPoints()) {
|
||||
if (pointSet.contains(pt)) { // shared vertex, remove it
|
||||
@@ -163,10 +155,10 @@ public class TableExtractionService {
|
||||
|
||||
// X first sort
|
||||
List<Point2D> pointsSortX = new ArrayList<>(pointSet);
|
||||
Collections.sort(pointsSortX, X_FIRST_POINT_COMPARATOR);
|
||||
pointsSortX.sort(X_FIRST_POINT_COMPARATOR);
|
||||
// Y first sort
|
||||
List<Point2D> pointsSortY = new ArrayList<>(pointSet);
|
||||
Collections.sort(pointsSortY, POINT_COMPARATOR);
|
||||
pointsSortY.sort(POINT_COMPARATOR);
|
||||
|
||||
while (i < pointSet.size()) {
|
||||
float currY = (float) pointsSortY.get(i).getY();
|
||||
@@ -203,13 +195,12 @@ public class TableExtractionService {
|
||||
nextVertex = edgesV.get(curr.point);
|
||||
edgesV.remove(curr.point);
|
||||
lastAddedVertex = new PolygonVertex(nextVertex, Direction.VERTICAL);
|
||||
polygon.add(lastAddedVertex);
|
||||
} else {
|
||||
nextVertex = edgesH.get(curr.point);
|
||||
edgesH.remove(curr.point);
|
||||
lastAddedVertex = new PolygonVertex(nextVertex, Direction.HORIZONTAL);
|
||||
polygon.add(lastAddedVertex);
|
||||
}
|
||||
polygon.add(lastAddedVertex);
|
||||
|
||||
if (lastAddedVertex.equals(polygon.get(0))) {
|
||||
// closed polygon
|
||||
@@ -227,10 +218,10 @@ public class TableExtractionService {
|
||||
|
||||
// calculate grid-aligned minimum area rectangles for each found polygon
|
||||
for (List<PolygonVertex> poly : polygons) {
|
||||
float top = java.lang.Float.MAX_VALUE;
|
||||
float left = java.lang.Float.MAX_VALUE;
|
||||
float bottom = java.lang.Float.MIN_VALUE;
|
||||
float right = java.lang.Float.MIN_VALUE;
|
||||
float top = Float.MAX_VALUE;
|
||||
float left = Float.MAX_VALUE;
|
||||
float bottom = Float.MIN_VALUE;
|
||||
float right = Float.MIN_VALUE;
|
||||
for (PolygonVertex pt : poly) {
|
||||
top = (float) Math.min(top, pt.point.getY());
|
||||
left = (float) Math.min(left, pt.point.getX());
|
||||
@@ -244,69 +235,66 @@ public class TableExtractionService {
|
||||
}
|
||||
|
||||
|
||||
private static final Comparator<Point2D> X_FIRST_POINT_COMPARATOR = new Comparator<Point2D>() {
|
||||
@Override
|
||||
public int compare(Point2D arg0, Point2D arg1) {
|
||||
int rv = 0;
|
||||
float arg0X = Utils.round(arg0.getX(), 2);
|
||||
float arg0Y = Utils.round(arg0.getY(), 2);
|
||||
float arg1X = Utils.round(arg1.getX(), 2);
|
||||
float arg1Y = Utils.round(arg1.getY(), 2);
|
||||
private static final Comparator<Point2D> X_FIRST_POINT_COMPARATOR = (arg0, arg1) -> {
|
||||
|
||||
if (arg0X > arg1X) {
|
||||
rv = 1;
|
||||
} else if (arg0X < arg1X) {
|
||||
rv = -1;
|
||||
} else if (arg0Y > arg1Y) {
|
||||
rv = 1;
|
||||
} else if (arg0Y < arg1Y) {
|
||||
rv = -1;
|
||||
}
|
||||
return rv;
|
||||
int rv = 0;
|
||||
float arg0X = Utils.round(arg0.getX(), 2);
|
||||
float arg0Y = Utils.round(arg0.getY(), 2);
|
||||
float arg1X = Utils.round(arg1.getX(), 2);
|
||||
float arg1Y = Utils.round(arg1.getY(), 2);
|
||||
|
||||
if (arg0X > arg1X) {
|
||||
rv = 1;
|
||||
} else if (arg0X < arg1X) {
|
||||
rv = -1;
|
||||
} else if (arg0Y > arg1Y) {
|
||||
rv = 1;
|
||||
} else if (arg0Y < arg1Y) {
|
||||
rv = -1;
|
||||
}
|
||||
return rv;
|
||||
};
|
||||
|
||||
private static final Comparator<Point2D> POINT_COMPARATOR = (arg0, arg1) -> {
|
||||
|
||||
private static final Comparator<Point2D> POINT_COMPARATOR = new Comparator<Point2D>() {
|
||||
@Override
|
||||
public int compare(Point2D arg0, Point2D arg1) {
|
||||
int rv = 0;
|
||||
float arg0X = Utils.round(arg0.getX(), 2);
|
||||
float arg0Y = Utils.round(arg0.getY(), 2);
|
||||
float arg1X = Utils.round(arg1.getX(), 2);
|
||||
float arg1Y = Utils.round(arg1.getY(), 2);
|
||||
int rv = 0;
|
||||
float arg0X = Utils.round(arg0.getX(), 2);
|
||||
float arg0Y = Utils.round(arg0.getY(), 2);
|
||||
float arg1X = Utils.round(arg1.getX(), 2);
|
||||
float arg1Y = Utils.round(arg1.getY(), 2);
|
||||
|
||||
|
||||
if (arg0Y > arg1Y) {
|
||||
rv = 1;
|
||||
} else if (arg0Y < arg1Y) {
|
||||
rv = -1;
|
||||
} else if (arg0X > arg1X) {
|
||||
rv = 1;
|
||||
} else if (arg0X < arg1X) {
|
||||
rv = -1;
|
||||
}
|
||||
return rv;
|
||||
if (arg0Y > arg1Y) {
|
||||
rv = 1;
|
||||
} else if (arg0Y < arg1Y) {
|
||||
rv = -1;
|
||||
} else if (arg0X > arg1X) {
|
||||
rv = 1;
|
||||
} else if (arg0X < arg1X) {
|
||||
rv = -1;
|
||||
}
|
||||
return rv;
|
||||
};
|
||||
|
||||
|
||||
private enum Direction {
|
||||
HORIZONTAL,
|
||||
VERTICAL
|
||||
HORIZONTAL, VERTICAL
|
||||
}
|
||||
|
||||
static class PolygonVertex {
|
||||
|
||||
Point2D point;
|
||||
Direction direction;
|
||||
|
||||
public PolygonVertex(Point2D point, Direction direction) {
|
||||
|
||||
PolygonVertex(Point2D point, Direction direction) {
|
||||
|
||||
this.direction = direction;
|
||||
this.point = point;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean equals(Object other) {
|
||||
|
||||
if (this == other) {
|
||||
return true;
|
||||
}
|
||||
@@ -316,15 +304,21 @@ public class TableExtractionService {
|
||||
return this.point.equals(((PolygonVertex) other).point);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
|
||||
return this.point.hashCode();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return String.format("%s[point=%s,direction=%s]", this.getClass().getName(), this.point.toString(), this.direction.toString());
|
||||
|
||||
return String.format("%s[point=%s,direction=%s]", this.getClass()
|
||||
.getName(), this.point.toString(), this.direction.toString());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+188
-129
@@ -16,7 +16,8 @@ import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
@@ -26,6 +27,7 @@ import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSeque
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
@@ -41,138 +43,171 @@ public class AnnotationHighlightService {
|
||||
private final DictionaryService dictionaryService;
|
||||
|
||||
|
||||
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction) throws IOException {
|
||||
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction, ManualRedactions manualRedactions) throws IOException {
|
||||
|
||||
for (int page = 1; page <= document.getNumberOfPages(); page++) {
|
||||
|
||||
PDPage pdPage = document.getPage(page - 1);
|
||||
|
||||
if (!flatRedaction) {
|
||||
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
|
||||
|
||||
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
|
||||
|
||||
for (int i = 0; i <= paragraph.getPageBlocks().size() - 1; i++) {
|
||||
|
||||
AbstractTextContainer textBlock = paragraph.getPageBlocks().get(i);
|
||||
|
||||
if (textBlock.getPage() != page) {
|
||||
continue;
|
||||
}
|
||||
if (textBlock instanceof TextBlock) {
|
||||
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
|
||||
visualizeTextBlock((TextBlock) textBlock, contentStream);
|
||||
} else if (textBlock instanceof Table) {
|
||||
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
|
||||
visualizeTable((Table) textBlock, contentStream);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
contentStream.close();
|
||||
}
|
||||
drawSectionFrames(document, classifiedDoc, flatRedaction, pdPage, page);
|
||||
|
||||
if (classifiedDoc.getEntities().get(page) == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (Entity entity : classifiedDoc.getEntities().get(page)) {
|
||||
addAnnotations(pdPage, classifiedDoc, flatRedaction, manualRedactions, page);
|
||||
addManualAnnotations(pdPage, classifiedDoc, manualRedactions, page);
|
||||
}
|
||||
}
|
||||
|
||||
RedactionLogEntry redactionLogEntry = new RedactionLogEntry();
|
||||
|
||||
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
|
||||
private void addAnnotations(PDPage pdPage, Document classifiedDoc, boolean flatRedaction, ManualRedactions manualRedactions, int page) throws IOException {
|
||||
|
||||
if (flatRedaction && !isRedactionType(entity)) {
|
||||
continue;
|
||||
}
|
||||
List<PDAnnotation> annotations = pdPage.getAnnotations();
|
||||
|
||||
for (TextPositionSequence textPositions : entityPositionSequence.getSequences()) {
|
||||
for (Entity entity : classifiedDoc.getEntities().get(page)) {
|
||||
|
||||
float height = textPositions.getTextPositions().get(0).getHeightDir() + 2;
|
||||
|
||||
float posXInit;
|
||||
float posXEnd;
|
||||
float posYInit;
|
||||
float posYEnd;
|
||||
|
||||
if (textPositions.getTextPositions().get(0).getRotation() == 90) {
|
||||
|
||||
posXEnd = textPositions.getTextPositions().get(0).getYDirAdj() + 2;
|
||||
posXInit = textPositions.getTextPositions().get(0).getYDirAdj() - height;
|
||||
posYInit = textPositions.getTextPositions().get(0).getXDirAdj();
|
||||
posYEnd = textPositions.getTextPositions()
|
||||
.get(textPositions.getTextPositions().size() - 1)
|
||||
.getXDirAdj() - height + 4;
|
||||
} else {
|
||||
|
||||
posXInit = textPositions.getTextPositions().get(0).getXDirAdj();
|
||||
posXEnd = textPositions.getTextPositions()
|
||||
.get(textPositions.getTextPositions().size() - 1)
|
||||
.getXDirAdj() + textPositions.getTextPositions()
|
||||
.get(textPositions.getTextPositions().size() - 1)
|
||||
.getWidth() + 1;
|
||||
posYInit = textPositions.getTextPositions()
|
||||
.get(0)
|
||||
.getPageHeight() - textPositions.getTextPositions().get(0).getYDirAdj() - 2;
|
||||
posYEnd = textPositions.getTextPositions()
|
||||
.get(0)
|
||||
.getPageHeight() - textPositions.getTextPositions()
|
||||
.get(textPositions.getTextPositions().size() - 1)
|
||||
.getYDirAdj() + 2;
|
||||
}
|
||||
|
||||
Rectangle textHighlightRectangle = new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
|
||||
|
||||
List<PDAnnotation> annotations = pdPage.getAnnotations();
|
||||
PDAnnotationTextMarkup highlight = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
|
||||
highlight.constructAppearances();
|
||||
|
||||
PDRectangle annotationPosition = new PDRectangle();
|
||||
annotationPosition.setLowerLeftX(posXInit);
|
||||
annotationPosition.setLowerLeftY(posYEnd);
|
||||
annotationPosition.setUpperRightX(posXEnd);
|
||||
annotationPosition.setUpperRightY(posYEnd + height);
|
||||
|
||||
highlight.setRectangle(annotationPosition);
|
||||
if (!flatRedaction && !isHint(entity)) {
|
||||
highlight.setAnnotationName(entityPositionSequence.getId().toString());
|
||||
highlight.setTitlePopup(entityPositionSequence.getId().toString());
|
||||
highlight.setContents("\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\n" + "In Section : \"" + entity
|
||||
.getHeadline() + "\"");
|
||||
}
|
||||
|
||||
highlight.setQuadPoints(toQuadPoints(textHighlightRectangle));
|
||||
|
||||
PDColor color;
|
||||
if (flatRedaction) {
|
||||
color = new PDColor(new float[]{0, 0, 0}, PDDeviceRGB.INSTANCE);
|
||||
} else {
|
||||
color = new PDColor(getColor(entity), PDDeviceRGB.INSTANCE);
|
||||
}
|
||||
|
||||
highlight.setColor(color);
|
||||
annotations.add(highlight);
|
||||
|
||||
redactionLogEntry.getPositions().add(textHighlightRectangle);
|
||||
|
||||
}
|
||||
redactionLogEntry.setId(entityPositionSequence.getId().toString());
|
||||
}
|
||||
redactionLogEntry.setColor(getColor(entity));
|
||||
redactionLogEntry.setReason(entity.getRedactionReason());
|
||||
redactionLogEntry.setValue(entity.getWord());
|
||||
redactionLogEntry.setType(entity.getType());
|
||||
redactionLogEntry.setRedacted(entity.isRedaction());
|
||||
redactionLogEntry.setSection(entity.getHeadline());
|
||||
redactionLogEntry.setHint(isHint(entity));
|
||||
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
|
||||
if (flatRedaction && !isRedactionType(entity)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(entity);
|
||||
|
||||
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
|
||||
|
||||
if (manualRedactions != null && manualRedactions.getIdsToRemove()
|
||||
.contains(entityPositionSequence.getId())) {
|
||||
entity.setRedaction(false);
|
||||
entity.setRedactionReason(entity.getRedactionReason() + ", removed by manual override");
|
||||
redactionLogEntry.setManual(true);
|
||||
}
|
||||
|
||||
for (TextPositionSequence textPositions : entityPositionSequence.getSequences()) {
|
||||
|
||||
Rectangle rectangle = textPositions.getRectangle();
|
||||
redactionLogEntry.getPositions().add(rectangle);
|
||||
annotations.add(createAnnotation(rectangle, entityPositionSequence.getId(), createAnnotationContent(entity), getColor(entity), !flatRedaction && !isHint(entity)));
|
||||
}
|
||||
redactionLogEntry.setId(entityPositionSequence.getId());
|
||||
}
|
||||
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addManualAnnotations(PDPage pdPage, Document classifiedDoc, ManualRedactions manualRedactions, int page) throws IOException {
|
||||
|
||||
if (manualRedactions == null) {
|
||||
return;
|
||||
}
|
||||
|
||||
List<PDAnnotation> annotations = pdPage.getAnnotations();
|
||||
|
||||
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
|
||||
|
||||
String id = IdBuilder.buildId(manualRedactionEntry);
|
||||
|
||||
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(manualRedactionEntry);
|
||||
|
||||
for (Rectangle rectangle : manualRedactionEntry.getPositions()) {
|
||||
|
||||
if (page != rectangle.getPage()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
PDAnnotationTextMarkup highlight = createAnnotation(rectangle, id, createAnnotationContent(manualRedactionEntry), getColor(manualRedactionEntry
|
||||
.getType()), true);
|
||||
annotations.add(highlight);
|
||||
|
||||
redactionLogEntry.getPositions().add(rectangle);
|
||||
}
|
||||
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry) {
|
||||
|
||||
return RedactionLogEntry.builder()
|
||||
.color(getColor(manualRedactionEntry.getType()))
|
||||
.reason(manualRedactionEntry.getReason())
|
||||
.value(manualRedactionEntry.getValue())
|
||||
.type(manualRedactionEntry.getType())
|
||||
.redacted(true)
|
||||
.isHint(false)
|
||||
.section(manualRedactionEntry.getSection())
|
||||
.sectionNumber(manualRedactionEntry.getSectionNumber())
|
||||
.manual(true)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private RedactionLogEntry createRedactionLogEntry(Entity entity) {
|
||||
|
||||
return RedactionLogEntry.builder()
|
||||
.color(getColor(entity))
|
||||
.reason(entity.getRedactionReason())
|
||||
.value(entity.getWord())
|
||||
.type(entity.getType())
|
||||
.redacted(entity.isRedaction())
|
||||
.isHint(isHint(entity))
|
||||
.section(entity.getHeadline())
|
||||
.sectionNumber(entity.getSectionNumber())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private PDAnnotationTextMarkup createAnnotation(Rectangle rectangle, String id, String content, float[] color, boolean popup) {
|
||||
|
||||
PDAnnotationTextMarkup annotation = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
|
||||
annotation.constructAppearances();
|
||||
annotation.setRectangle(toPDRectangle(rectangle));
|
||||
annotation.setQuadPoints(toQuadPoints(rectangle));
|
||||
if (popup) {
|
||||
annotation.setAnnotationName(id);
|
||||
annotation.setTitlePopup(id);
|
||||
annotation.setContents(content);
|
||||
}
|
||||
annotation.setColor(new PDColor(color, PDDeviceRGB.INSTANCE));
|
||||
return annotation;
|
||||
}
|
||||
|
||||
|
||||
private String createAnnotationContent(Entity entity) {
|
||||
|
||||
return new StringBuilder().append("\nRule ")
|
||||
.append(entity.getMatchedRule())
|
||||
.append(" matched")
|
||||
.append("\n\n")
|
||||
.append(entity.getRedactionReason())
|
||||
.append("\n\nIn Section : \"")
|
||||
.append(entity.getHeadline())
|
||||
.append("\"")
|
||||
.toString();
|
||||
}
|
||||
|
||||
|
||||
private String createAnnotationContent(ManualRedactionEntry entry) {
|
||||
|
||||
return new StringBuilder().append("\nManual Redaction")
|
||||
.append("\n\nIn Section : \"")
|
||||
.append(entry.getSection())
|
||||
.append("\"")
|
||||
.toString();
|
||||
}
|
||||
|
||||
|
||||
private PDRectangle toPDRectangle(Rectangle rectangle) {
|
||||
|
||||
PDRectangle annotationPosition = new PDRectangle();
|
||||
annotationPosition.setLowerLeftX(rectangle.getTopLeft().getX());
|
||||
annotationPosition.setLowerLeftY(rectangle.getTopLeft().getY() + rectangle.getHeight());
|
||||
annotationPosition.setUpperRightX(rectangle.getTopLeft().getX() + rectangle.getWidth());
|
||||
annotationPosition.setUpperRightY(rectangle.getTopLeft().getY());
|
||||
return annotationPosition;
|
||||
}
|
||||
|
||||
|
||||
private float[] toQuadPoints(Rectangle rectangle) {
|
||||
|
||||
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
|
||||
@@ -201,15 +236,24 @@ public class AnnotationHighlightService {
|
||||
if (!entity.isRedaction() && !isHint(entity)) {
|
||||
return new float[]{0.627f, 0.627f, 0.627f};
|
||||
}
|
||||
|
||||
if (!dictionaryService.getEntryColors().containsKey(entity.getType())) {
|
||||
return dictionaryService.getDefaultColor();
|
||||
}
|
||||
|
||||
return dictionaryService.getEntryColors().get(entity.getType());
|
||||
}
|
||||
|
||||
|
||||
private float[] getColor(String type) {
|
||||
|
||||
if (!dictionaryService.getEntryColors().containsKey(type)) {
|
||||
return dictionaryService.getDefaultColor();
|
||||
}
|
||||
return dictionaryService.getEntryColors().get(type);
|
||||
}
|
||||
|
||||
|
||||
private boolean isHint(Entity entity) {
|
||||
|
||||
List<String> hintTypes = dictionaryService.getHintTypes();
|
||||
if (CollectionUtils.isNotEmpty(hintTypes) && hintTypes.contains(entity.getType())) {
|
||||
return true;
|
||||
@@ -217,24 +261,49 @@ public class AnnotationHighlightService {
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
private void drawSectionFrames(PDDocument document, Document classifiedDoc, boolean flatRedaction, PDPage pdPage, int page) throws IOException {
|
||||
|
||||
if (flatRedaction) {
|
||||
return;
|
||||
}
|
||||
|
||||
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
|
||||
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
|
||||
|
||||
for (int i = 0; i <= paragraph.getPageBlocks().size() - 1; i++) {
|
||||
|
||||
AbstractTextContainer textBlock = paragraph.getPageBlocks().get(i);
|
||||
|
||||
if (textBlock.getPage() != page) {
|
||||
continue;
|
||||
}
|
||||
if (textBlock instanceof TextBlock) {
|
||||
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
|
||||
visualizeTextBlock((TextBlock) textBlock, contentStream);
|
||||
} else if (textBlock instanceof Table) {
|
||||
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
|
||||
visualizeTable((Table) textBlock, contentStream);
|
||||
}
|
||||
}
|
||||
}
|
||||
contentStream.close();
|
||||
}
|
||||
|
||||
|
||||
private void visualizeTextBlock(TextBlock textBlock, PDPageContentStream contentStream) throws IOException {
|
||||
|
||||
contentStream.setStrokingColor(Color.LIGHT_GRAY);
|
||||
contentStream.setLineWidth(0.5f);
|
||||
|
||||
contentStream.addRect(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight());
|
||||
contentStream.stroke();
|
||||
|
||||
if (textBlock.getClassification() != null) {
|
||||
contentStream.beginText();
|
||||
|
||||
contentStream.setNonStrokingColor(Color.DARK_GRAY);
|
||||
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
|
||||
|
||||
contentStream.newLineAtOffset(textBlock.getMinX(), textBlock.getMaxY());
|
||||
|
||||
contentStream.showText(textBlock.getClassification());
|
||||
|
||||
contentStream.endText();
|
||||
}
|
||||
}
|
||||
@@ -251,26 +320,16 @@ public class AnnotationHighlightService {
|
||||
contentStream.addRect((float) cell.getX(), (float) cell.getY(), (float) cell.getWidth(), (float) cell
|
||||
.getHeight());
|
||||
contentStream.stroke();
|
||||
|
||||
// contentStream.setStrokingColor(Color.GREEN);
|
||||
// for (TextBlock textBlock : cell.getTextBlocks()) {
|
||||
// contentStream.addRect(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight());
|
||||
// contentStream.stroke();
|
||||
// }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (table.getClassification() != null) {
|
||||
contentStream.beginText();
|
||||
|
||||
contentStream.setNonStrokingColor(Color.DARK_GRAY);
|
||||
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
|
||||
|
||||
contentStream.newLineAtOffset(table.getMinX(), table.getMinY());
|
||||
|
||||
contentStream.showText(table.getClassification());
|
||||
|
||||
contentStream.endText();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,12 +8,6 @@ spring:
|
||||
profiles:
|
||||
active: kubernetes
|
||||
|
||||
platform.multi-tenancy:
|
||||
enabled: ${multitenancy.enabled:false}
|
||||
tenantFilter:
|
||||
urlPatterns: /redact
|
||||
urlPatternsToIgnore:
|
||||
|
||||
management:
|
||||
endpoint:
|
||||
metrics.enabled: ${monitoring.enabled:false}
|
||||
|
||||
+154
-11
@@ -5,6 +5,8 @@ import static org.springframework.boot.test.context.SpringBootTest.WebEnvironmen
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.File;
|
||||
import java.io.FileInputStream;
|
||||
import java.io.FileOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
@@ -15,11 +17,11 @@ import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.commons.io.IOUtils;
|
||||
import org.junit.Before;
|
||||
import org.junit.Ignore;
|
||||
import org.junit.Test;
|
||||
import org.junit.runner.RunWith;
|
||||
import org.kie.api.KieServices;
|
||||
@@ -40,6 +42,10 @@ import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
|
||||
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionResult;
|
||||
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
|
||||
@@ -48,7 +54,6 @@ import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
|
||||
|
||||
@Ignore
|
||||
@RunWith(SpringRunner.class)
|
||||
@SpringBootTest(webEnvironment = DEFINED_PORT)
|
||||
public class RedactionIntegrationTest {
|
||||
@@ -58,6 +63,9 @@ public class RedactionIntegrationTest {
|
||||
private static final String ADDRESS_CODE = "address";
|
||||
private static final String NAME_CODE = "name";
|
||||
private static final String NO_REDACTION_INDICATOR = "no_redaction_indicator";
|
||||
private static final String REDACTION_INDICATOR = "redaction_indicator";
|
||||
private static final String HINT_ONLY = "hint_only";
|
||||
private static final String MUST_REDACT = "must_redact";
|
||||
|
||||
@Autowired
|
||||
private RedactionController redactionController;
|
||||
@@ -96,7 +104,7 @@ public class RedactionIntegrationTest {
|
||||
|
||||
|
||||
@Before
|
||||
public void stubRulesClient() {
|
||||
public void stubClients() {
|
||||
|
||||
when(rulesClient.getVersion()).thenReturn(0L);
|
||||
when(rulesClient.getRules()).thenReturn(new RulesResponse(RULES));
|
||||
@@ -109,6 +117,9 @@ public class RedactionIntegrationTest {
|
||||
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(getDictionaryResponse(ADDRESS_CODE));
|
||||
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(getDictionaryResponse(NAME_CODE));
|
||||
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
|
||||
when(dictionaryClient.getDictionaryForType(REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(REDACTION_INDICATOR));
|
||||
when(dictionaryClient.getDictionaryForType(HINT_ONLY)).thenReturn(getDictionaryResponse(HINT_ONLY));
|
||||
when(dictionaryClient.getDictionaryForType(MUST_REDACT)).thenReturn(getDictionaryResponse(MUST_REDACT));
|
||||
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor(new float[]{1f, 0.502f, 0f}));
|
||||
}
|
||||
|
||||
@@ -131,7 +142,22 @@ public class RedactionIntegrationTest {
|
||||
.map(this::cleanDictionaryEntry)
|
||||
.collect(Collectors.toSet()));
|
||||
dictionary.computeIfAbsent(NO_REDACTION_INDICATOR, v -> new ArrayList<>())
|
||||
.addAll(ResourceLoader.load("dictionaries/NoRedactionIndicator.txt")
|
||||
.addAll(ResourceLoader.load("dictionaries/no_redaction_indicator.txt")
|
||||
.stream()
|
||||
.map(this::cleanDictionaryEntry)
|
||||
.collect(Collectors.toSet()));
|
||||
dictionary.computeIfAbsent(REDACTION_INDICATOR, v -> new ArrayList<>())
|
||||
.addAll(ResourceLoader.load("dictionaries/redaction_indicator.txt")
|
||||
.stream()
|
||||
.map(this::cleanDictionaryEntry)
|
||||
.collect(Collectors.toSet()));
|
||||
dictionary.computeIfAbsent(HINT_ONLY, v -> new ArrayList<>())
|
||||
.addAll(ResourceLoader.load("dictionaries/hint_only.txt")
|
||||
.stream()
|
||||
.map(this::cleanDictionaryEntry)
|
||||
.collect(Collectors.toSet()));
|
||||
dictionary.computeIfAbsent(MUST_REDACT, v -> new ArrayList<>())
|
||||
.addAll(ResourceLoader.load("dictionaries/must_redact.txt")
|
||||
.stream()
|
||||
.map(this::cleanDictionaryEntry)
|
||||
.collect(Collectors.toSet()));
|
||||
@@ -149,17 +175,26 @@ public class RedactionIntegrationTest {
|
||||
typeColorMap.put(VERTEBRATES_CODE, new float[]{0, 1, 0});
|
||||
typeColorMap.put(ADDRESS_CODE, new float[]{0, 1, 1});
|
||||
typeColorMap.put(NAME_CODE, new float[]{1, 1, 0});
|
||||
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{1, 0.502f, 0});
|
||||
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{0.8f, 0, 0.8f});
|
||||
typeColorMap.put(REDACTION_INDICATOR, new float[]{1, 0.502f, 0.1f});
|
||||
typeColorMap.put(HINT_ONLY, new float[]{0.8f, 1, 0.8f});
|
||||
typeColorMap.put(MUST_REDACT, new float[]{1, 0, 0});
|
||||
|
||||
hintTypeMap.put(VERTEBRATES_CODE, true);
|
||||
hintTypeMap.put(ADDRESS_CODE, false);
|
||||
hintTypeMap.put(NAME_CODE, false);
|
||||
hintTypeMap.put(NO_REDACTION_INDICATOR, true);
|
||||
hintTypeMap.put(REDACTION_INDICATOR, true);
|
||||
hintTypeMap.put(HINT_ONLY, true);
|
||||
hintTypeMap.put(MUST_REDACT, true);
|
||||
|
||||
caseInSensitiveMap.put(VERTEBRATES_CODE, true);
|
||||
caseInSensitiveMap.put(ADDRESS_CODE, false);
|
||||
caseInSensitiveMap.put(NAME_CODE, false);
|
||||
caseInSensitiveMap.put(NO_REDACTION_INDICATOR, true);
|
||||
caseInSensitiveMap.put(REDACTION_INDICATOR, true);
|
||||
caseInSensitiveMap.put(HINT_ONLY, true);
|
||||
caseInSensitiveMap.put(MUST_REDACT, true);
|
||||
}
|
||||
|
||||
|
||||
@@ -189,11 +224,52 @@ public class RedactionIntegrationTest {
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void noExceptionShouldBeThrownForAnyFiles() throws IOException {
|
||||
|
||||
System.out.println("noExceptionShouldBeThrownForAnyFiles");
|
||||
ClassLoader loader = getClass().getClassLoader();
|
||||
URL url = loader.getResource("files");
|
||||
File[] files = new File(url.getPath()).listFiles();
|
||||
List<File> input = new ArrayList<>();
|
||||
for (File file : files) {
|
||||
input.addAll(getPathsRecursively(file));
|
||||
}
|
||||
for (File path : input) {
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(new FileInputStream(path)))
|
||||
.build();
|
||||
System.out.println("Redacting file : " + path.getName());
|
||||
redactionController.redact(request);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
private List<File> getPathsRecursively(File path) {
|
||||
|
||||
List<File> result = new ArrayList<>();
|
||||
if (path == null || path.listFiles() == null) {
|
||||
return result;
|
||||
}
|
||||
for (File f : path.listFiles()) {
|
||||
if (f.isFile()) {
|
||||
result.add(f);
|
||||
} else {
|
||||
result.addAll(getPathsRecursively(f));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void redactionTest() throws IOException {
|
||||
|
||||
System.out.println("redactionTest");
|
||||
long start = System.currentTimeMillis();
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Trinexapac/96 Trinexapac-ethyl_RAR_09_Volume_3CA_B-7_2018-02-23.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
@@ -212,10 +288,70 @@ public class RedactionIntegrationTest {
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testTableRedaction() throws IOException {
|
||||
|
||||
System.out.println("testTableRedaction");
|
||||
long start = System.currentTimeMillis();
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
.build();
|
||||
|
||||
RedactionResult result = redactionController.redact(request);
|
||||
|
||||
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
|
||||
fileOutputStream.write(result.getDocument());
|
||||
}
|
||||
long end = System.currentTimeMillis();
|
||||
|
||||
System.out.println("duration: " + (end - start));
|
||||
System.out.println("numberOfPages: " + result.getNumberOfPages());
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testManualRedaction() throws IOException {
|
||||
|
||||
System.out.println("testManualRedaction");
|
||||
long start = System.currentTimeMillis();
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
|
||||
|
||||
ManualRedactions manualRedactions = new ManualRedactions();
|
||||
manualRedactions.setIdsToRemove(Set.of("0836727c3508a0b2ea271da69c04cc2f"));
|
||||
|
||||
ManualRedactionEntry manualRedactionEntry = new ManualRedactionEntry();
|
||||
manualRedactionEntry.setType("name");
|
||||
manualRedactionEntry.setValue("O'Loughlin C.K.");
|
||||
manualRedactionEntry.setReason("Manual Redaction");
|
||||
manualRedactionEntry.setPositions(List.of(new Rectangle(new Point(375.61096f, 241.282f), 7.648041f, 43.72262f, 1), new Rectangle(new Point(384.83517f, 241.282f), 7.648041f, 17.043358f, 1)));
|
||||
|
||||
manualRedactions.getEntriesToAdd().add(manualRedactionEntry);
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
.manualRedactions(manualRedactions)
|
||||
.build();
|
||||
|
||||
RedactionResult result = redactionController.redact(request);
|
||||
|
||||
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
|
||||
fileOutputStream.write(result.getDocument());
|
||||
}
|
||||
long end = System.currentTimeMillis();
|
||||
|
||||
System.out.println("duration: " + (end - start));
|
||||
System.out.println("numberOfPages: " + result.getNumberOfPages());
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void classificationTest() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
System.out.println("classificationTest");
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
|
||||
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
@@ -232,7 +368,9 @@ public class RedactionIntegrationTest {
|
||||
@Test
|
||||
public void sectionsTest() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
System.out.println("sectionsTest");
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
|
||||
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
@@ -249,7 +387,9 @@ public class RedactionIntegrationTest {
|
||||
@Test
|
||||
public void htmlTablesTest() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
System.out.println("htmlTablesTest");
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
|
||||
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
@@ -266,7 +406,9 @@ public class RedactionIntegrationTest {
|
||||
@Test
|
||||
public void htmlTableRotationTest() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
|
||||
System.out.println("htmlTableRotationTest");
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S" +
|
||||
"-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
|
||||
|
||||
RedactionRequest request = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
@@ -286,7 +428,8 @@ public class RedactionIntegrationTest {
|
||||
if (resource == null) {
|
||||
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
|
||||
}
|
||||
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
|
||||
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(),
|
||||
StandardCharsets.UTF_8))) {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
String str;
|
||||
while ((str = br.readLine()) != null) {
|
||||
|
||||
+268
@@ -0,0 +1,268 @@
|
||||
package com.iqser.red.service.redaction.v1.server.redaction.service;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
import static org.mockito.Mockito.when;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
import java.net.URL;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import org.apache.commons.io.IOUtils;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.junit.Before;
|
||||
import org.junit.Test;
|
||||
import org.junit.runner.RunWith;
|
||||
import org.kie.api.KieServices;
|
||||
import org.kie.api.builder.KieBuilder;
|
||||
import org.kie.api.builder.KieFileSystem;
|
||||
import org.kie.api.builder.KieModule;
|
||||
import org.kie.api.runtime.KieContainer;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.boot.test.context.SpringBootTest;
|
||||
import org.springframework.boot.test.context.TestConfiguration;
|
||||
import org.springframework.boot.test.mock.mockito.MockBean;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.core.io.ClassPathResource;
|
||||
import org.springframework.test.context.junit4.SpringRunner;
|
||||
|
||||
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
|
||||
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
|
||||
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
|
||||
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
|
||||
|
||||
@SpringBootTest
|
||||
@RunWith(SpringRunner.class)
|
||||
public class EntityRedactionServiceTest {
|
||||
|
||||
private static final String DEFAULT_RULES = loadFromClassPath("drools/rules.drl");
|
||||
private static final String NAME_CODE = "name";
|
||||
private static final String ADDRESS_CODE = "address";
|
||||
|
||||
private static final AtomicLong DICTIONARY_VERSION = new AtomicLong();
|
||||
@MockBean
|
||||
private DictionaryClient dictionaryClient;
|
||||
|
||||
@MockBean
|
||||
private RulesClient rulesClient;
|
||||
|
||||
@Autowired
|
||||
private EntityRedactionService entityRedactionService;
|
||||
|
||||
@Autowired
|
||||
private PdfSegmentationService pdfSegmentationService;
|
||||
|
||||
@TestConfiguration
|
||||
public static class RedactionIntegrationTestConfiguration {
|
||||
|
||||
@Bean
|
||||
public KieContainer kieContainer() {
|
||||
|
||||
KieServices kieServices = KieServices.Factory.get();
|
||||
|
||||
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
|
||||
InputStream input = new ByteArrayInputStream(DEFAULT_RULES.getBytes(StandardCharsets.UTF_8));
|
||||
kieFileSystem.write("src/test/resources/drools/rules.drl", kieServices.getResources()
|
||||
.newInputStreamResource(input));
|
||||
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
|
||||
kieBuilder.buildAll();
|
||||
KieModule kieModule = kieBuilder.getKieModule();
|
||||
|
||||
return kieServices.newKieContainer(kieModule.getReleaseId());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testNestedEntitiesRemoval() {
|
||||
|
||||
Set<Entity> entities = new HashSet<>();
|
||||
Entity nested = new Entity("nested", "fake type", 10, 16, "fake headline", 0);
|
||||
Entity nesting = new Entity("nesting nested", "fake type", 2, 16, "fake headline", 0);
|
||||
entities.add(nested);
|
||||
entities.add(nesting);
|
||||
entityRedactionService.removeEntitiesContainedInLarger(entities);
|
||||
|
||||
assertThat(entities.size()).isEqualTo(1);
|
||||
assertThat(entities).contains(nesting);
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testTableRedaction() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
|
||||
|
||||
RedactionRequest redactionRequest = RedactionRequest.builder()
|
||||
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
|
||||
.build();
|
||||
|
||||
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
|
||||
.entries(Arrays.asList("Casey, H.W.", "O’Loughlin, C.K.", "Salamon, C.M.", "Smith, S.H."))
|
||||
.build();
|
||||
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
|
||||
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
|
||||
DictionaryResponse addressResponse = DictionaryResponse.builder()
|
||||
.entries(Collections.singletonList("Toxigenics, Inc., Decatur, IL 62526, USA"))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
|
||||
|
||||
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc, null);
|
||||
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
|
||||
assertThat(classifiedDoc.getEntities().get(1)).hasSize(5); // 4 out of 5 entities recognized on page 1
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testTrueNegativesInTable() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Cyprodinil/40 Cyprodinil - EU AIR3 - LCA Section 1" +
|
||||
" Supplement - Identity of the active substance - Reference list.pdf");
|
||||
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
|
||||
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
|
||||
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/names.txt")))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
|
||||
DictionaryResponse addressResponse = DictionaryResponse.builder()
|
||||
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/addresses.txt")))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc, null);
|
||||
assertThat(classifiedDoc.getEntities()
|
||||
.entrySet()
|
||||
.stream()
|
||||
.noneMatch(entry -> entry.getValue().stream().anyMatch(e -> e.getMatchedRule() == 9))).isTrue();
|
||||
}
|
||||
pdfFileResource = new ClassPathResource("files/Compounds/27 A8637C - EU AIR3 - MCP Section 1 - Identity of " +
|
||||
"the plant protection product.pdf");
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc, null);
|
||||
assertThat(classifiedDoc.getEntities()
|
||||
.entrySet()
|
||||
.stream()
|
||||
.noneMatch(entry -> entry.getValue().stream().anyMatch(e -> e.getMatchedRule() == 9))).isTrue();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testFalsePositiveInWrongCell() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Row With Ambiguous Redaction.pdf");
|
||||
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
|
||||
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
|
||||
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/names.txt")))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
|
||||
DictionaryResponse addressResponse = DictionaryResponse.builder()
|
||||
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/addresses.txt")))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc, null);
|
||||
assertThat(classifiedDoc.getEntities()).hasSize(1); // two pages
|
||||
assertThat(classifiedDoc.getEntities().get(1).stream()
|
||||
.filter(entity -> entity.getMatchedRule() == 9)
|
||||
.count()).isEqualTo(10);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@Test
|
||||
public void headerPropagation() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Header Propagation.pdf");
|
||||
|
||||
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
|
||||
.entries(Arrays.asList("Bissig R.", "Thanei P."))
|
||||
.build();
|
||||
|
||||
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
|
||||
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
|
||||
DictionaryResponse addressResponse = DictionaryResponse.builder()
|
||||
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
|
||||
.build();
|
||||
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
|
||||
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
|
||||
entityRedactionService.processDocument(classifiedDoc, null);
|
||||
assertThat(classifiedDoc.getEntities()).hasSize(2); // two pages
|
||||
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(8);
|
||||
assertThat(classifiedDoc.getEntities().get(2).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(4);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Before
|
||||
public void stubRedaction() {
|
||||
String tableRules = "package drools\n" +
|
||||
"\n" +
|
||||
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
|
||||
"\n" +
|
||||
"global Section section\n" +
|
||||
"rule \"9: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study\"\n" +
|
||||
" when\n" +
|
||||
" Section(isVertebrateStudy())\n" +
|
||||
" then\n" +
|
||||
" section.redact(\"name\", 9, \"Redacted because row is a vertebrate study\");\n" +
|
||||
" section.redact(\"address\", 9, \"Redacted because rows is a vertebrate study\");\n" +
|
||||
" section.highlightCell(\"Vertebrate study Y/N\", 9, \"must_redact\");\n" +
|
||||
" end";
|
||||
when(rulesClient.getVersion()).thenReturn(1L);
|
||||
when(rulesClient.getRules()).thenReturn(new RulesResponse(tableRules));
|
||||
TypeResponse typeResponse = TypeResponse.builder()
|
||||
.types(Arrays.asList(
|
||||
TypeResult.builder().type(NAME_CODE).color(new float[]{1, 1, 0}).build(),
|
||||
TypeResult.builder().type(ADDRESS_CODE).color(new float[]{0, 1, 1}).build()))
|
||||
.build();
|
||||
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
|
||||
when(dictionaryClient.getAllTypes()).thenReturn(typeResponse);
|
||||
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor());
|
||||
}
|
||||
|
||||
|
||||
private static String loadFromClassPath(String path) {
|
||||
|
||||
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
|
||||
if (resource == null) {
|
||||
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
|
||||
}
|
||||
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
String str;
|
||||
while ((str = br.readLine()) != null) {
|
||||
sb.append(str).append("\n");
|
||||
}
|
||||
return sb.toString();
|
||||
} catch (IOException e) {
|
||||
throw new IllegalArgumentException("could not load classpath resource: " + path, e);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
package com.iqser.red.service.redaction.v1.server.segmentation;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.junit.Test;
|
||||
import org.junit.runner.RunWith;
|
||||
import org.kie.api.runtime.KieContainer;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.boot.test.context.SpringBootTest;
|
||||
import org.springframework.boot.test.mock.mockito.MockBean;
|
||||
import org.springframework.core.io.ClassPathResource;
|
||||
import org.springframework.test.context.junit4.SpringRunner;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.service.BlockificationService;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.service.RulingCleaningService;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.service.TableExtractionService;
|
||||
|
||||
@SpringBootTest
|
||||
@RunWith(SpringRunner.class)
|
||||
public class PdfSegmentationServiceTest {
|
||||
|
||||
@Autowired
|
||||
private PdfSegmentationService pdfSegmentationService;
|
||||
|
||||
@Autowired
|
||||
private RulingCleaningService rulingCleaningService;
|
||||
|
||||
@Autowired
|
||||
private TableExtractionService tableExtractionService;
|
||||
|
||||
@Autowired
|
||||
private BlockificationService blockificationService;
|
||||
|
||||
@MockBean
|
||||
private KieContainer kieContainer;
|
||||
|
||||
|
||||
@Test
|
||||
public void testPDFSegmentationWithComplexTable() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Spanning Cells.pdf");
|
||||
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document document = pdfSegmentationService.parseDocument(pdDocument);
|
||||
assertThat(document.getParagraphs()
|
||||
.stream()
|
||||
.flatMap(paragraph -> paragraph.getTables().stream())
|
||||
.collect(Collectors.toList())).isNotEmpty();
|
||||
Table table = document.getParagraphs()
|
||||
.stream()
|
||||
.flatMap(paragraph -> paragraph.getTables().stream())
|
||||
.collect(Collectors.toList())
|
||||
.get(0);
|
||||
assertThat(table.getColCount()).isEqualTo(6);
|
||||
assertThat(table.getRowCount()).isEqualTo(13);
|
||||
assertThat(table.getRows().stream().mapToInt(List::size).sum()).isEqualTo(6 * 13);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Test
|
||||
public void testTableExtraction() throws IOException {
|
||||
|
||||
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Merge Table.pdf");
|
||||
|
||||
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
|
||||
Document document = pdfSegmentationService.parseDocument(pdDocument);
|
||||
assertThat(document.getParagraphs()
|
||||
.stream()
|
||||
.flatMap(paragraph -> paragraph.getTables().stream())
|
||||
.collect(Collectors.toList())).isNotEmpty();
|
||||
Table firstTable = document.getParagraphs()
|
||||
.stream()
|
||||
.flatMap(paragraph -> paragraph.getTables().stream())
|
||||
.collect(Collectors.toList())
|
||||
.get(0);
|
||||
assertThat(firstTable.getColCount()).isEqualTo(8);
|
||||
assertThat(firstTable.getRowCount()).isEqualTo(1);
|
||||
Table secondTable = document.getParagraphs()
|
||||
.stream()
|
||||
.flatMap(paragraph -> paragraph.getTables().stream())
|
||||
.collect(Collectors.toList())
|
||||
.get(1);
|
||||
assertThat(secondTable.getColCount()).isEqualTo(8);
|
||||
assertThat(secondTable.getRowCount()).isEqualTo(2);
|
||||
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
|
||||
.get(0)
|
||||
.stream()
|
||||
.map(Cell::getHeaderCells)
|
||||
.collect(Collectors.toList());
|
||||
assertThat(secondTable.getRows().stream()
|
||||
.allMatch(row -> row.stream()
|
||||
.map(Cell::getHeaderCells)
|
||||
.collect(Collectors.toList())
|
||||
.equals(firstTableHeaderCells)))
|
||||
.isTrue();
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
-1
@@ -1 +0,0 @@
|
||||
In Vitro
|
||||
+1546
-776
File diff suppressed because it is too large
Load Diff
+2
@@ -0,0 +1,2 @@
|
||||
guideline
|
||||
unpublished
|
||||
+3
@@ -0,0 +1,3 @@
|
||||
Batches Produced at
|
||||
CTL
|
||||
for determination of residues
|
||||
+7628
-2649
File diff suppressed because it is too large
Load Diff
+3
@@ -0,0 +1,3 @@
|
||||
published paper
|
||||
in vitro
|
||||
in-vitro
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
in vivo
|
||||
in-vivo
|
||||
dermal penetration
|
||||
oral toxicity
|
||||
oral-toxicity
|
||||
acute toxicity
|
||||
acute-toxicity
|
||||
eco toxicity
|
||||
eco-toxicity
|
||||
+91
-129
@@ -1,48 +1,63 @@
|
||||
Vulpes vulpes
|
||||
a. sylvaticus
|
||||
african clawed frog
|
||||
agalychnis callidryas
|
||||
albino rat
|
||||
american bullfrog tadpole
|
||||
american toad
|
||||
amphibian
|
||||
amphibians
|
||||
American bullfrog tadpole
|
||||
american toad
|
||||
anad platyrhynchos
|
||||
Anas platyrhynchos
|
||||
anas platyrhynchos
|
||||
anuran
|
||||
anurans
|
||||
apodemus
|
||||
apodemus flavicollis
|
||||
apodemus syl vaticus
|
||||
apodemus sylvaticus
|
||||
arvicola terrestris
|
||||
avian
|
||||
bank vole
|
||||
bird
|
||||
birds
|
||||
bluegill
|
||||
bluegill sunfish
|
||||
bobwhite
|
||||
bobwhite quail
|
||||
bullfrog
|
||||
Bufo americanus
|
||||
brachydanio rerio
|
||||
brown hare
|
||||
bufo americanus
|
||||
bullfrog
|
||||
canary
|
||||
carassius carassius
|
||||
carp
|
||||
catesbeiana
|
||||
catfish
|
||||
cattle
|
||||
cattles
|
||||
channel catfish
|
||||
Chinook
|
||||
chicken
|
||||
Colinus virginianus
|
||||
chinese hamster
|
||||
chinese hamsters
|
||||
chinook
|
||||
coho salmon
|
||||
colinus virginianus
|
||||
Common carp
|
||||
columba palumbus
|
||||
columbidae
|
||||
common carp
|
||||
common vole
|
||||
coturnix japonica
|
||||
Coturnix japonica
|
||||
cow
|
||||
cows
|
||||
Crucian carp
|
||||
crocidura russula
|
||||
crucian carp
|
||||
cyprinodon variegatus
|
||||
cyprinus carpio
|
||||
dog
|
||||
dogs
|
||||
duck
|
||||
ducks
|
||||
european brown hare
|
||||
european rabbit
|
||||
fathead minnow
|
||||
fish
|
||||
fishes
|
||||
@@ -56,174 +71,121 @@ galaxias truttaceus
|
||||
gasterosteus aculeatus
|
||||
goat
|
||||
goats
|
||||
greater white-toothed shrew
|
||||
guinea
|
||||
guinea pig
|
||||
guinea pigs
|
||||
Guppy
|
||||
guinea-pigs
|
||||
guppy
|
||||
hamster
|
||||
hamsters
|
||||
hen
|
||||
hens
|
||||
Hyla versicolor
|
||||
house mouse
|
||||
hyla versicolor
|
||||
ictalurus melas
|
||||
ictalurus punctatus
|
||||
japanese quail
|
||||
japonica
|
||||
kisutch
|
||||
lagomorph
|
||||
lebistes reticulatus
|
||||
leiostomus xanthurus
|
||||
leisostomus xanthurus
|
||||
lepomis macrochirus
|
||||
lepus europaeus
|
||||
limnocharis
|
||||
limnodynastes
|
||||
limnodynastes tasmaniensis
|
||||
livestock
|
||||
livestocks
|
||||
mallard
|
||||
mallard duck
|
||||
mammal
|
||||
mammalian
|
||||
mammals
|
||||
Mammalian
|
||||
marten
|
||||
martes
|
||||
mice
|
||||
microtus
|
||||
microtus agrestis
|
||||
microtus arvalis
|
||||
microtus subterraneus
|
||||
midwestern anurans
|
||||
minnow
|
||||
minnows
|
||||
monkey
|
||||
mouse
|
||||
mus musculus
|
||||
myodes glareolus
|
||||
northern bobwhite
|
||||
o. cuniculus
|
||||
o. mykiss
|
||||
Oncorhynchus mykiss
|
||||
Oncorhynchus
|
||||
O. mykiss
|
||||
o. tshawytscha
|
||||
oncorhynchus
|
||||
oncorhynchus mykiss
|
||||
oncorhynchus tshawytscha
|
||||
oryctolagus cuniculus
|
||||
oryzias melastigma
|
||||
oryzias melastigma larvae
|
||||
p. promelas
|
||||
pagrus major
|
||||
palumbus
|
||||
pig
|
||||
pigeon
|
||||
pigeons
|
||||
pigs
|
||||
pimephales promela
|
||||
pimephales promelas
|
||||
Pseudacris triseriata
|
||||
poecilia reticulata
|
||||
poultry
|
||||
pseudacris
|
||||
pseudacris triseriata
|
||||
quail
|
||||
r. catesbeiana
|
||||
rabbit
|
||||
rabbits
|
||||
rainbow trout
|
||||
Rana limnocharis
|
||||
rana
|
||||
limnocharis
|
||||
rana catesbeiana
|
||||
rana limnocharis
|
||||
rana pipiens
|
||||
rat
|
||||
rats
|
||||
reptile
|
||||
reptiles
|
||||
ricefish
|
||||
ruminant
|
||||
ruminants
|
||||
salmo gairdneri
|
||||
salmon
|
||||
serinus canaria
|
||||
sheepshead minnow
|
||||
sheepshead minnows
|
||||
spea multiplicata
|
||||
Salmo gairdneri
|
||||
salmon
|
||||
spotted march frog
|
||||
tadpoles
|
||||
treefrog
|
||||
toad
|
||||
terrestrial vertrebrates
|
||||
Limnodynastes tasmaniensis
|
||||
trout
|
||||
Vulpes vulpes
|
||||
wistar
|
||||
xenopus laevis
|
||||
xenpous leavis
|
||||
zebra fish
|
||||
zebrafish
|
||||
Salmo gairdneri
|
||||
minnow
|
||||
minnows
|
||||
Pimephales promela
|
||||
Cyprinodon variegatus
|
||||
limnodynastes
|
||||
Rana catesbeiana
|
||||
R. catesbeiana
|
||||
coho salmon
|
||||
Oncorhynchus tshawytscha
|
||||
O. tshawytscha
|
||||
tshawytscha
|
||||
catesbeiana
|
||||
kisutch
|
||||
Pseudacris triseriata
|
||||
Pseudacris
|
||||
triseriata
|
||||
Wood pigeon
|
||||
Columba palumbus
|
||||
palumbus
|
||||
Columbidae
|
||||
shrew
|
||||
shrews
|
||||
bank vole
|
||||
common vole
|
||||
sorex araneus
|
||||
spea multiplicata
|
||||
spotted march frog
|
||||
tadpoles
|
||||
terrestrial vertrebrates
|
||||
toad
|
||||
treefrog
|
||||
triseriata
|
||||
trout
|
||||
tshawytscha
|
||||
vole
|
||||
voles
|
||||
lagomorph
|
||||
Wood mouse
|
||||
Apodemus sylvaticus
|
||||
A. sylvaticus
|
||||
Apodemus flavicollis
|
||||
Apodemus
|
||||
mus musculus
|
||||
Microtus arvalis
|
||||
Microtus agrestis
|
||||
Microtus
|
||||
Arvicola terrestris
|
||||
Sorex araneus
|
||||
Myodes glareolus
|
||||
yellow-necked mouse
|
||||
house mouse
|
||||
Oryctolagus cuniculus
|
||||
marten
|
||||
martes
|
||||
vulpes vulpes
|
||||
white rabbits
|
||||
white-toothed shrew
|
||||
greater white-toothed shrew
|
||||
Lepus europaeus
|
||||
brown hare
|
||||
European brown hare
|
||||
European rabbit
|
||||
O. cuniculus
|
||||
Crocidura russula
|
||||
Chinese Hamster
|
||||
Rat
|
||||
Rats
|
||||
Dog
|
||||
Chinese hamsters
|
||||
Chinese hamster
|
||||
Mouse
|
||||
Guinea pig
|
||||
Wistar rats
|
||||
Rabbit
|
||||
mammalian
|
||||
Japanese quail
|
||||
Microtus subterraneus
|
||||
Lepomis macrochirus
|
||||
P. promelas
|
||||
Cyprinus carpio
|
||||
Fish
|
||||
Ictalurus punctatus
|
||||
Carassius carassius
|
||||
Lepomis macrochirus
|
||||
Poecilia reticulata
|
||||
Lebistes reticulatus
|
||||
Lepomis macrochirus
|
||||
Leiostomus xanthurus
|
||||
Pimephales promelas
|
||||
Lepomis macrochirus
|
||||
Albino rat
|
||||
Hen
|
||||
Goat
|
||||
Livestock
|
||||
Guinea Pigs
|
||||
Hamster
|
||||
wistar
|
||||
wistar rats
|
||||
wood mice
|
||||
wood mouse
|
||||
Rabbits
|
||||
Mice
|
||||
Rainbow trout
|
||||
Canary
|
||||
Serinus canaria
|
||||
Guinea Pig
|
||||
Cow
|
||||
Pigs
|
||||
Poultry
|
||||
Guinea-pigs
|
||||
White rabbits
|
||||
Birds
|
||||
Wood mice
|
||||
wood pigeon
|
||||
xenopus laevis
|
||||
xenpous leavis
|
||||
yellow-necked mouse
|
||||
zebra fish
|
||||
zebrafish
|
||||
+80
-17
@@ -32,27 +32,90 @@ rule "3: Do not redact Names and Addresses if no redaction Indicator is containe
|
||||
end
|
||||
|
||||
|
||||
rule "4: Redact contact information, if applicant is found"
|
||||
rule "4: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
|
||||
when
|
||||
eval(section.getText().toLowerCase().contains("applicant"));
|
||||
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true && section.contains("redaction_indicator")==true);
|
||||
then
|
||||
section.redactLineAfter("Name:", "address", 4, "Redacted because of Rule 4");
|
||||
section.redactBetween("Address:", "Contact", "address", 4, "Redacted because of Rule 4");
|
||||
section.redactLineAfter("Contact point:", "address", 4, "Redacted because of Rule 4");
|
||||
section.redactLineAfter("Phone:", "address", 4, "Redacted because of Rule 4");
|
||||
section.redactLineAfter("Fax:", "address", 4, "Redacted because of Rule 4");
|
||||
section.redactLineAfter("E-mail:", "address", 4, "Redacted because of Rule 4");
|
||||
section.redact("name", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
|
||||
section.redact("address", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
|
||||
end
|
||||
|
||||
|
||||
rule "5: Redact contact information, if 'Producer of the plant protection product' is found"
|
||||
rule "5: Do not redact in guideline sections"
|
||||
when
|
||||
eval(section.getText().contains("Producer of the plant protection product"));
|
||||
eval(section.headlineContainsWord("guideline") || section.headlineContainsWord("Guidance"));
|
||||
then
|
||||
section.redactLineAfter("Name:", "address", 5, "xxxx");
|
||||
section.redactBetween("Address:", "Contact", "address", 5, "xxxx");
|
||||
section.redactBetween("Contact:", "Phone", "address", 5, "xxxx");
|
||||
section.redactLineAfter("Phone:", "address", 5, "xxxx");
|
||||
section.redactLineAfter("Fax:", "address", 5, "xxxx");
|
||||
section.redactLineAfter("E-mail:", "address", 5, "xxxx");
|
||||
end
|
||||
section.redactNot("name", 5, "Section is a guideline section.");
|
||||
section.redactNot("address", 5, "Section is a guideline section.");
|
||||
end
|
||||
|
||||
rule "6: Redact contact information, if applicant is found"
|
||||
when
|
||||
eval(section.headlineContainsWord("applicant") || section.getText().contains("Applicant"));
|
||||
then
|
||||
section.redactLineAfter("Name:", "address", 6, "Applicant information was found");
|
||||
section.redactBetween("Address:", "Contact", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Contact point:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Phone:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Fax:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Tel.:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Tel:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("E-mail:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Email:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Contact:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Telephone number:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Fax number:", "address", 6, "Applicant information was found");
|
||||
section.redactLineAfter("Telephone:", "address", 6, "Applicant information was found");
|
||||
section.redactBetween("No:", "Fax", "address", 6, "Applicant information was found");
|
||||
section.redactBetween("Contact:", "Tel.:", "address", 6, "Applicant information was found");
|
||||
end
|
||||
|
||||
rule "7: Redact contact information, if Producer is found"
|
||||
when
|
||||
eval(section.getText().toLowerCase().contains("producer of the plant protection") || section.getText().toLowerCase().contains("producer of the active substance") || section.getText().contains("Manufacturer of the active substance") || section.getText().contains("Manufacturer:") || section.getText().contains("Producer or producers of the active substance"));
|
||||
then
|
||||
section.redactLineAfter("Name:", "address", 7, "Producer was found");
|
||||
section.redactBetween("Address:", "Contact", "address", 7, "Producer was found");
|
||||
section.redactBetween("Contact:", "Phone", "address", 7, "Producer was found");
|
||||
section.redactBetween("Contact:", "Telephone number:", "address", 7, "Producer was found");
|
||||
section.redactBetween("Address:", "Manufacturing", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Telephone:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Phone:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Fax:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("E-mail:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Contact:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Fax number:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Telephone number:", "address", 7, "Producer was found");
|
||||
section.redactLineAfter("Tel:", "address", 7, "Producer was found");
|
||||
section.redactBetween("No:", "Fax", "address", 7, "Producer was found");
|
||||
end
|
||||
|
||||
|
||||
rule "8: Not redacted because Vertebrate Study = N"
|
||||
when
|
||||
Section(isNotVertebrateStudy())
|
||||
then
|
||||
section.redactNot("name", 8, "Not redacted because row is not a vertebrate study");
|
||||
section.redactNot("address", 8, "Not redacted because row is not a vertebrate study");
|
||||
section.highlightCell("Vertebrate study Y/N", 8, "hint_only");
|
||||
section.highlightCell("Verte brate study Y/N", 8, "hint_only");
|
||||
end
|
||||
|
||||
|
||||
rule "9: Redact if must redact entry is found"
|
||||
when
|
||||
eval(section.contains("must_redact")==true);
|
||||
then
|
||||
section.redact("name", 9, "must_redact entry was found.");
|
||||
section.redact("address", 9, "must_redact entry was found.");
|
||||
end
|
||||
|
||||
rule "10: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study"
|
||||
when
|
||||
Section(isVertebrateStudy())
|
||||
then
|
||||
section.redact("name", 10, "Redacted because row is a vertebrate study");
|
||||
section.redact("address", 10, "Redacted because row is a vertebrate study");
|
||||
section.highlightCell("Vertebrate study Y/N", 10, "must_redact");
|
||||
section.highlightCell("Verte brate study Y/N", 10, "must_redact");
|
||||
end
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Reference in New Issue
Block a user