Compare commits

..
Author SHA1 Message Date
deiflaender 5241099141 Fixed duplicated redaction/RedactionLog entries 2020-07-31 16:23:37 +02:00
120 changed files with 5319 additions and 30356 deletions
+2 -2
View File
@@ -5,7 +5,7 @@
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>7.1.2</version>
<version>7.0.4</version>
<relativePath/>
</parent>
@@ -34,4 +34,4 @@
<!-- run 'mvn test' to perform offline validation of the plan -->
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
</project>
</project>
@@ -51,7 +51,7 @@ public class PlanSpec {
private PlanPermissions createPlanPermission(PlanIdentifier planIdentifier) {
Permissions permission = new Permissions()
.userPermissions("atlbamboo", PermissionType.EDIT, PermissionType.VIEW, PermissionType.ADMIN, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("red-backend", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.groupPermissions("gin4", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
.loggedInUserPermissions(PermissionType.VIEW)
.anonymousUserPermissionView();
return new PlanPermissions(planIdentifier.getProjectKey(), planIdentifier.getPlanKey()).permissions(permission);
@@ -134,4 +134,4 @@ public class PlanSpec {
.whenInactiveInRepositoryAfterDays(14))
.notificationForCommitters());
}
}
}
@@ -1,4 +1,4 @@
FROM red/base-image:1.0.0
FROM gin5/platform-base:5.2.0
ARG PLATFORM_JAR
@@ -6,13 +6,4 @@ ENV PLATFORM_JAR ${PLATFORM_JAR}
ENV USES_ELASTICSEARCH false
COPY ["${PLATFORM_JAR}", "/"]
RUN apt-get update \
&& DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
wget cabextract xfonts-utils fonts-liberation \
&& rm -rf /var/lib/apt/lists/*
RUN curl http://ftp.br.debian.org/debian/pool/contrib/m/msttcorefonts/ttf-mscorefonts-installer_3.7_all.deb -o /tmp/ttf-mscorefonts-installer_3.7_all.deb \
&& dpkg -i /tmp/ttf-mscorefonts-installer_3.7_all.deb \
&& rm /tmp/ttf-mscorefonts-installer_3.7_all.deb \
COPY ["${PLATFORM_JAR}", "/"]
+13 -13
View File
@@ -5,7 +5,7 @@
<parent>
<artifactId>platform-dependency</artifactId>
<groupId>com.iqser.red</groupId>
<version>1.0.2</version>
<version>1.0.1</version>
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -22,7 +22,7 @@
</modules>
<properties>
<pdfbox.version>2.0.21</pdfbox.version>
<pdfbox.version>2.0.16</pdfbox.version>
</properties>
@@ -32,21 +32,21 @@
<dependency>
<groupId>com.iqser.red</groupId>
<artifactId>platform-commons-dependency</artifactId>
<version>1.1.7</version>
<version>1.0.0</version>
<scope>import</scope>
<type>pom</type>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
<version>${pdfbox.version}</version>
</dependency>
</dependencies>
@@ -1,16 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class CellRectangle {
private Point topLeft;
private float width;
private float height;
}
@@ -1,21 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.time.OffsetDateTime;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class Comment {
private String id;
private OffsetDateTime date;
private String text;
private String user;
}
@@ -1,19 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class IdRemoval {
private String id;
private String user;
private Status status;
private boolean removeFromDictionary;
}
@@ -1,30 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactionEntry {
private String id;
private String user;
private String type;
private String value;
private String reason;
private String legalBasis;
private List<Rectangle> positions = new ArrayList<>();
private Status status;
private boolean addToDictionary;
private String section;
private int sectionNumber;
}
@@ -1,5 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum ManualRedactionType {
ADD, REMOVE
}
@@ -1,29 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactions {
@Builder.Default
private Set<IdRemoval> idsToRemove = new HashSet<>();
@Builder.Default
private Set<ManualRedactionEntry> entriesToAdd = new HashSet<>();
@Builder.Default
private Map<String, List<Comment>> comments = new HashMap<>();
}
@@ -1,11 +1,11 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import java.util.List;
@Data
@AllArgsConstructor
@NoArgsConstructor
@@ -13,20 +13,4 @@ public class RedactionLog {
private List<RedactionLogEntry> redactionLogEntry;
private long dictionaryVersion = -1;
private long rulesVersion = -1;
private String ruleSetId;
private String filename;
public RedactionLog(List<RedactionLogEntry> redactionLogEntry, long dictionaryVersion, long rulesVersion, String ruleSetId) {
this.redactionLogEntry = redactionLogEntry;
this.dictionaryVersion = dictionaryVersion;
this.rulesVersion = rulesVersion;
this.ruleSetId = ruleSetId;
}
}
@@ -3,38 +3,19 @@ package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class RedactionLogEntry {
private String id;
private String type;
private String value;
private String reason;
private int matchedRule;
private String legalBasis;
private boolean redacted;
private boolean isHint;
private boolean isRecommendation;
private String section;
private float[] color;
@Builder.Default
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private boolean manual;
private Status status;
private ManualRedactionType manualRedactionType;
private boolean isDictionaryEntry;
private String textBefore;
private String textAfter;
}
@@ -12,7 +12,5 @@ import lombok.NoArgsConstructor;
public class RedactionRequest {
private byte[] document;
private String ruleSetId;
private boolean flatRedaction;
private ManualRedactions manualRedactions;
}
@@ -14,6 +14,5 @@ public class RedactionResult {
private byte[] document;
private int numberOfPages;
private RedactionLog redactionLog;
private SectionGrid sectionGrid;
}
@@ -1,18 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
public class SectionGrid {
private Map<Integer, List<SectionRectangle>> rectanglesPerPage = new HashMap<>();
}
@@ -1,33 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
@Data
@AllArgsConstructor
@NoArgsConstructor
@RequiredArgsConstructor
public class SectionRectangle {
@NonNull
private Point topLeft;
@NonNull
private float width;
@NonNull
private float height;
@NonNull
private int part;
@NonNull
private int numberOfParts;
private List<CellRectangle> tableCells;
}
@@ -1,5 +0,0 @@
package com.iqser.red.service.redaction.v1.model;
public enum Status {
REQUESTED, APPROVED, DECLINED
}
@@ -1,19 +1,16 @@
package com.iqser.red.service.redaction.v1.resources;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
public interface RedactionResource {
String SERVICE_NAME = "redaction-service-v1";
String RULE_SET_PARAMETER_NAME = "ruleSetId";
String RULE_SET_PATH_VARIABLE = "/{" + RULE_SET_PARAMETER_NAME + "}";
@PostMapping(value = "/redact", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult redact(@RequestBody RedactionRequest redactionRequest);
@@ -26,10 +23,7 @@ public interface RedactionResource {
@PostMapping(value = "/debug/htmlTables", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/rules/update"+RULE_SET_PATH_VARIABLE, consumes = MediaType.APPLICATION_JSON_VALUE)
void updateRules(@PathVariable(RULE_SET_PARAMETER_NAME) String ruleSetId);
@PostMapping(value = "/rules/test", consumes = MediaType.APPLICATION_JSON_VALUE)
void testRules(@RequestBody String rules);
@PostMapping(value = "/rules/update", consumes = MediaType.APPLICATION_JSON_VALUE)
void updateRules(@RequestBody String rules);
}
@@ -11,6 +11,25 @@
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<pdfbox.version>2.0.20</pdfbox.version>
</properties>
<dependencyManagement>
<dependencies>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
<version>${pdfbox.version}</version>
</dependency>
</dependencies>
</dependencyManagement>
<dependencies>
<dependency>
<groupId>com.iqser.red.service</groupId>
@@ -20,7 +39,7 @@
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>configuration-service-api-v1</artifactId>
<version>2.0.0</version>
<version>1.0.12</version>
</dependency>
<dependency>
<groupId>org.drools</groupId>
@@ -37,22 +56,17 @@
<artifactId>jts-core</artifactId>
<version>1.16.1</version>
</dependency>
<dependency>
<groupId>com.google.guava</groupId>
<artifactId>guava</artifactId>
<version>29.0-jre</version>
</dependency>
<!-- commons -->
<dependency>
<groupId>com.iqser.red.commons</groupId>
<groupId>com.iqser.gin4.commons</groupId>
<artifactId>spring-commons</artifactId>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<groupId>com.iqser.gin4.commons</groupId>
<artifactId>logging-commons</artifactId>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<groupId>com.iqser.gin4.commons</groupId>
<artifactId>metric-commons</artifactId>
</dependency>
<!-- other external -->
@@ -85,7 +99,7 @@
<scope>test</scope>
</dependency>
<dependency>
<groupId>com.iqser.red.commons</groupId>
<groupId>com.iqser.gin4.commons</groupId>
<artifactId>test-commons</artifactId>
<scope>test</scope>
</dependency>
@@ -1,26 +1,70 @@
package com.iqser.red.service.redaction.v1.server;
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.metrics.web.servlet.WebMvcMetricsAutoConfiguration;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
import org.springframework.boot.context.properties.EnableConfigurationProperties;
import org.springframework.cloud.openfeign.EnableFeignClients;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Import;
import com.iqser.gin4.commons.spring.DefaultWebMvcConfiguration;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
@Import({DefaultWebMvcConfiguration.class})
@EnableFeignClients(basePackageClasses = RulesClient.class)
@EnableConfigurationProperties(RedactionServiceSettings.class)
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class, WebMvcMetricsAutoConfiguration.class})
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
public class Application {
@Autowired
private RulesClient rulesClient;
public static void main(String[] args) {
SpringApplication.run(Application.class, args);
}
@Bean
public KieContainer kieContainer() {
try {
KieServices kieServices = KieServices.Factory.get();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
RulesResponse rules = rulesClient.getRules();
if (StringUtils.isEmpty(rules.getRules())) {
throw new RuntimeException("Rules cannot be empty.");
}
InputStream input = new ByteArrayInputStream(rules.getRules().getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/main/resources/drools/rules.drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
return kieServices.newKieContainer(kieModule.getReleaseId());
} catch (Exception e) {
throw new RulesValidationException("Could not update rules: " + e.getMessage(), e);
}
}
}
@@ -6,7 +6,6 @@ import java.util.List;
import java.util.Map;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import lombok.Data;
@@ -26,7 +25,4 @@ public class Document {
private boolean headlines;
private List<RedactionLogEntry> redactionLogEntities = new ArrayList<>();
private SectionGrid sectionGrid = new SectionGrid();
private long dictionaryVersion;
private long rulesVersion;
}
@@ -10,6 +10,7 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
public class Paragraph {
@@ -17,12 +18,10 @@ public class Paragraph {
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
private String headline;
public SearchableText getSearchableText() {
public SearchableText getSearchableText(){
SearchableText searchableText = new SearchableText();
pageBlocks.forEach(block -> {
if (block instanceof TextBlock) {
if(block instanceof TextBlock){
searchableText.addAll(((TextBlock) block).getSequences());
}
});
@@ -30,27 +29,14 @@ public class Paragraph {
}
public List<Table> getTables() {
public List<Table> getTables(){
List<Table> tables = new ArrayList<>();
pageBlocks.forEach(block -> {
if (block instanceof Table) {
if(block instanceof Table){
tables.add((Table) block);
}
});
return tables;
}
public List<TextBlock> getTextBlocks() {
List<TextBlock> textBlocks = new ArrayList<>();
pageBlocks.forEach(block -> {
if (block instanceof TextBlock) {
textBlocks.add((TextBlock) block);
}
});
return textBlocks;
}
}
}
@@ -5,45 +5,43 @@ import java.util.Map;
import lombok.Getter;
/**
*
*/
public class StringFrequencyCounter {
@Getter
private final Map<String, Integer> countPerValue = new HashMap<>();
Map<String, Integer> countPerValue = new HashMap<>();
public void add(String value) {
if (!countPerValue.containsKey(value)) {
public void add(String value){
if(!countPerValue.containsKey(value)){
countPerValue.put(value, 1);
} else {
countPerValue.put(value, countPerValue.get(value) + 1);
}
}
public void addAll(Map<String, Integer> otherCounter) {
for (Map.Entry<String, Integer> entry : otherCounter.entrySet()) {
if (countPerValue.containsKey(entry.getKey())) {
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
public void addAll(Map<String, Integer> otherCounter){
for(Map.Entry<String, Integer> entry: otherCounter.entrySet()){
if(countPerValue.containsKey(entry.getKey())){
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey())+ entry.getValue());
} else {
countPerValue.put(entry.getKey(), entry.getValue());
}
}
}
public String getMostPopular() {
public String getMostPopular(){
Map.Entry<String, Integer> mostPopular = null;
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
if (mostPopular == null) {
for(Map.Entry<String, Integer> entry: countPerValue.entrySet()){
if(mostPopular == null){
mostPopular = entry;
} else if (entry.getValue() > mostPopular.getValue()) {
} else if(entry.getValue() > mostPopular.getValue()){
mostPopular = entry;
}
}
return mostPopular != null ? mostPopular.getKey() : null;
}
}
}
@@ -97,7 +97,13 @@ public class TextBlock extends AbstractTextContainer {
this.maxY = Math.max(y1, y2);
}
public float getHeight() {
return maxY - minY;
}
public float getWidth() {
return maxX - minX;
}
@Override
public String toString() {
@@ -125,7 +131,7 @@ public class TextBlock extends AbstractTextContainer {
TextPositionSequence previous = null;
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -29,16 +29,20 @@ public class BlockificationService {
float minX = 1000, maxX = 0, minY = 1000, maxY = 0;
TextPositionSequence prev = null;
for (TextPositionSequence word : textPositions) {
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
boolean startFromTop = word.getY1() > maxY + word.getHeight();
if (prev != null && (lineSeparation || startFromTop || word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word
.getX1(), word.getY1(), verticalRulingLines) || word.getRotation() == 0 && isSplittedByRuling(minX, minY, word
.getX1(), word.getY2(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word
.getX1(), word.getY1(), horizontalRulingLines) || word.getRotation() == 90 && isSplittedByRuling(minX, minY, word
.getX1(), word.getY2(), verticalRulingLines))) {
if (prev != null &&
(lineSeparation
|| startFromTop
|| word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines)
|| word.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines)
|| word.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines)
|| word.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines)
)) {
TextBlock cb1 = buildTextBlock(chunkWords);
chunkBlockList1.add(cb1);
@@ -96,12 +100,11 @@ public class BlockificationService {
styleFrequencyCounter.add(wordBlock.getFontStyle());
if (textBlock == null) {
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
.getRotation());
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock.getRotation());
} else {
TextBlock spatialEntity = textBlock.union(wordBlock);
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
.getHeight());
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(),
spatialEntity.getWidth(), spatialEntity.getHeight());
}
}
@@ -119,7 +122,6 @@ public class BlockificationService {
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1, List<Ruling> rulingLines) {
for (Ruling ruling : rulingLines) {
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
return true;
@@ -131,6 +133,7 @@ public class BlockificationService {
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape) {
float minX = 10000;
float maxX = -100;
float minY = 10000;
@@ -144,6 +147,7 @@ public class BlockificationService {
for (AbstractTextContainer container : page.getTextBlocks()) {
if (container instanceof TextBlock) {
TextBlock textBlock = (TextBlock) container;
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
@@ -175,15 +179,16 @@ public class BlockificationService {
}
}
if (container instanceof Table) {
Table table = (Table) container;
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
for (Cell column : row) {
if (cell == null || cell.getTextBlocks() == null) {
if (column == null || column.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : cell.getTextBlocks()) {
for (TextBlock textBlock : column.getTextBlocks()) {
if (textBlock.getMinX() < minX) {
minX = textBlock.getMinX();
}
@@ -206,4 +211,5 @@ public class BlockificationService {
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
}
}
@@ -5,17 +5,15 @@ import java.util.regex.Pattern;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class ClassificationService {
@@ -30,7 +28,7 @@ public class ClassificationService {
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
System.out.println(document.getFontSizeCounter().getCountPerValue());
for (Page page : document.getPages()) {
Rectangle btf = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
@@ -41,7 +39,6 @@ public class ClassificationService {
public void classifyPage(Rectangle bodyTextFrame, Page page, Document document, List<Float> headlineFontSizes) {
for (AbstractTextContainer textBlock : page.getTextBlocks()) {
if (textBlock instanceof TextBlock) {
classifyBlock((TextBlock) textBlock, bodyTextFrame, page, document, headlineFontSizes);
@@ -50,33 +47,24 @@ public class ClassificationService {
}
public void classifyBlock(TextBlock textBlock, Rectangle bodyTextFrame, Page page, Document document,
List<Float> headlineFontSizes) {
public void classifyBlock(TextBlock textBlock, Rectangle bodyTextFrame, Page page, Document document, List<Float> headlineFontSizes) {
if (document.getFontSizeCounter().getMostPopular() == null) {
// TODO Figure out why this happens.
return;
}
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter().getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification("Header");
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock) && (document.getFontSizeCounter()
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
.getMostPopular())) {
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock) && (document.getFontSizeCounter().getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
textBlock.setClassification("Footer");
} else if (page.getPageNumber() == 1 && (!PositionUtils.isTouchingUnderBodyTextFrame(bodyTextFrame, textBlock) && PositionUtils
.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock, document.getTextHeightCounter()
.getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter()
.getMostPopular() || page.getTextBlocks().size() == 1)) {
} else if (page.getPageNumber() == 1
&& (!PositionUtils.isTouchingUnderBodyTextFrame(bodyTextFrame, textBlock)
&& PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock, document.getTextHeightCounter().getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter().getMostPopular() || page.getTextBlocks().size() == 1)) {
if (!Pattern.matches("[0-9]+", textBlock.toString())) {
textBlock.setClassification("Title");
}
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document
.getFontSizeCounter()
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && textBlock.getMostPopularWordStyle()
.equals("bold")) {
}
else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document.getFontSizeCounter().getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && textBlock.getMostPopularWordStyle().equals("bold")) {
for (int i = 1; i <= headlineFontSizes.size(); i++) {
if (textBlock.getMostPopularWordFontSize() == headlineFontSizes.get(i - 1)) {
@@ -84,34 +72,20 @@ public class ClassificationService {
document.setHeadlines(true);
}
}
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText()
.startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter()
.getMostPopular()
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
}else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText().startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular() && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
textBlock.setClassification("H " + (headlineFontSizes.size() + 1));
document.setHeadlines(true);
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
}
else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular() && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
textBlock.setClassification("TextBlock Bold");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFont()
.equals(document.getFontCounter().getMostPopular()) && textBlock.getMostPopularWordStyle()
.equals(document.getFontStyleCounter()
.getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
.getMostPopular()) {
}
else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFont().equals(document.getFontCounter().getMostPopular()) && textBlock.getMostPopularWordStyle().equals(document.getFontStyleCounter().getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular()) {
textBlock.setClassification("TextBlock");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
.getFontSizeCounter()
.getMostPopular() && textBlock.getMostPopularWordStyle()
.equals("italic") && !document.getFontStyleCounter()
.getMostPopular()
.equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
}
else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular() && textBlock.getMostPopularWordStyle().equals("italic") && !document.getFontStyleCounter().getMostPopular().equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
textBlock.setClassification("TextBlock Italic");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock)) {
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getSequences().size() > 3){
textBlock.setClassification("TextBlock Unknown");
}
}
}
@@ -1,8 +1,10 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.configuration.v1.api.resource.DictionaryResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = "DictionaryResource", url = "${configuration-service.url}")
import com.iqser.red.service.configuration.v1.api.resource.DictionaryResource;
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
@FeignClient(name = "DictionaryResource", url = "http://" + RulesResource.SERVICE_NAME + ":8080")
public interface DictionaryClient extends DictionaryResource {
}
@@ -1,8 +1,9 @@
package com.iqser.red.service.redaction.v1.server.client;
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
import org.springframework.cloud.openfeign.FeignClient;
@FeignClient(name = RulesResource.SERVICE_NAME, url = "${configuration-service.url}")
import com.iqser.red.service.configuration.v1.api.resource.RulesResource;
@FeignClient(name = RulesResource.SERVICE_NAME, url = "http://" + RulesResource.SERVICE_NAME + ":8080")
public interface RulesClient extends RulesResource {
}
@@ -2,13 +2,13 @@ package com.iqser.red.service.redaction.v1.server.controller;
import java.time.OffsetDateTime;
import com.iqser.red.commons.spring.ErrorMessage;
import org.springframework.http.HttpStatus;
import org.springframework.web.bind.annotation.ExceptionHandler;
import org.springframework.web.bind.annotation.ResponseBody;
import org.springframework.web.bind.annotation.ResponseStatus;
import org.springframework.web.bind.annotation.RestControllerAdvice;
import com.iqser.gin4.commons.api.errorhandling.ErrorMessage;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import lombok.extern.slf4j.Slf4j;
@@ -1,10 +1,16 @@
package com.iqser.red.service.redaction.v1.server.controller;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
@@ -15,20 +21,11 @@ import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationSer
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.AnnotationHighlightService;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfFlattenService;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.List;
@Slf4j
@RestController
@RequiredArgsConstructor
public class RedactionController implements RedactionResource {
@@ -37,9 +34,9 @@ public class RedactionController implements RedactionResource {
private final PdfSegmentationService pdfSegmentationService;
private final AnnotationHighlightService annotationHighlightService;
private final EntityRedactionService entityRedactionService;
private final PdfFlattenService pdfFlattenService;
private final DroolsExecutionService droolsExecutionService;
@Override
public RedactionResult redact(@RequestBody RedactionRequest redactionRequest) {
@@ -47,30 +44,22 @@ public class RedactionController implements RedactionResource {
pdDocument.setAllSecurityToBeRemoved(true);
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc);
annotationHighlightService.highlight(pdDocument, classifiedDoc, redactionRequest.isFlatRedaction());
log.info("Document structure analysis successful, starting redaction analysis...");
if (redactionRequest.isFlatRedaction()) {
PDDocument flatDocument = pdfFlattenService.flattenPDF(pdDocument);
return convert(flatDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
}
entityRedactionService.processDocument(classifiedDoc, redactionRequest.getRuleSetId(), redactionRequest.getManualRedactions());
annotationHighlightService.highlight(pdDocument, classifiedDoc, redactionRequest.isFlatRedaction(), redactionRequest
.getManualRedactions(), redactionRequest.getRuleSetId());
return convert(pdDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
log.info("Redaction analysis successful...");
return convert(pdDocument,
classifiedDoc.getPages().size(),
classifiedDoc.getRedactionLogEntities(),
classifiedDoc.getSectionGrid(),
classifiedDoc.getDictionaryVersion(),
classifiedDoc.getRulesVersion(),
redactionRequest.getRuleSetId());
} catch (Exception e) {
} catch (IOException e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult classify(@RequestBody RedactionRequest pdfSegmentationRequest) {
@@ -80,7 +69,7 @@ public class RedactionController implements RedactionResource {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
pdfVisualisationService.visualizeClassifications(classifiedDoc, pdDocument);
return convert(pdDocument, classifiedDoc.getPages().size(), pdfSegmentationRequest.getRuleSetId());
return convert(pdDocument, classifiedDoc.getPages().size());
} catch (IOException e) {
throw new RedactionException(e);
@@ -88,7 +77,6 @@ public class RedactionController implements RedactionResource {
}
@Override
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
@@ -98,7 +86,7 @@ public class RedactionController implements RedactionResource {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
pdfVisualisationService.visualizeParagraphs(classifiedDoc, pdDocument);
return convert(pdDocument, classifiedDoc.getPages().size(), redactionRequest.getRuleSetId());
return convert(pdDocument, classifiedDoc.getPages().size());
} catch (IOException e) {
throw new RedactionException(e);
@@ -106,7 +94,6 @@ public class RedactionController implements RedactionResource {
}
@Override
public RedactionResult htmlTables(@RequestBody RedactionRequest redactionRequest) {
@@ -133,35 +120,24 @@ public class RedactionController implements RedactionResource {
}
@Override
public void updateRules(@PathVariable(RULE_SET_PARAMETER_NAME) String ruleSetId) {
droolsExecutionService.updateRules(ruleSetId);
}
@Override
public void testRules(@RequestBody String rules) {
droolsExecutionService.testRules(rules);
public void updateRules(@RequestBody String rules) {
droolsExecutionService.updateRules(rules);
}
private RedactionResult convert(PDDocument document, int numberOfPages, String ruleSetId) throws IOException {
return convert(document, numberOfPages, null, null, 0, 0, ruleSetId);
private RedactionResult convert(PDDocument document, int numberOfPages) throws IOException {
return convert(document, numberOfPages, null);
}
private RedactionResult convert(PDDocument document, int numberOfPages,
List<RedactionLogEntry> redactionLogEntities,
SectionGrid sectionGrid, long dictionaryVersion, long rulesVersion, String ruleSetId) throws IOException {
private RedactionResult convert(PDDocument document, int numberOfPages, RedactionLog redactionLog) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
return RedactionResult.builder()
.document(byteArrayOutputStream.toByteArray())
.numberOfPages(numberOfPages)
.redactionLog(new RedactionLog(redactionLogEntities, dictionaryVersion, rulesVersion, ruleSetId))
.sectionGrid(sectionGrid)
.redactionLog(redactionLog)
.build();
}
@@ -44,10 +44,10 @@ import lombok.extern.slf4j.Slf4j;
public class PDFLinesTextStripper extends PDFTextStripper {
@Getter
private int maxCharWidths;
private float minCharWidth = Float.MAX_VALUE;
@Getter
private int maxCharHeight;
private float minCharHeight = Float.MAX_VALUE;
@Getter
private final List<TextPositionSequence> textPositionSequences = new ArrayList<>();
@@ -201,18 +201,10 @@ public class PDFLinesTextStripper extends PDFTextStripper {
int startIndex = 0;
for (int i = 0; i <= textPositions.size() - 1; i++) {
minCharWidth = Math.min(minCharWidth, textPositions.get(i).getWidthDirAdj());
minCharHeight = Math.min(minCharHeight, textPositions.get(i).getHeightDir());
int charHeight = (int) textPositions.get(i).getHeightDir();
if(charHeight > maxCharHeight){
maxCharHeight = charHeight;
}
int charWidth = (int) textPositions.get(i).getWidthDirAdj();
if(charWidth > maxCharWidths){
maxCharWidths = charWidth;
}
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0"))) {
if (i == 0 && textPositions.get(i).getUnicode().equals(" ")) {
startIndex++;
continue;
}
@@ -220,15 +212,15 @@ public class PDFLinesTextStripper extends PDFTextStripper {
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && sublist.get(0).getUnicode().equals(" "))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0")) && i <= textPositions.size() - 2) {
if (i > 0 && textPositions.get(i).getUnicode().equals(" ") && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && sublist.get(0).getUnicode().equals(" "))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i + 1;
@@ -236,20 +228,21 @@ public class PDFLinesTextStripper extends PDFTextStripper {
}
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
if (!sublist.isEmpty() && sublist.get(sublist.size() - 1).getUnicode().equals(" ")) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && sublist.get(0).getUnicode().equals(" "))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
super.writeString(text);
}
@Override
public String getText(PDDocument doc) throws IOException {
maxCharWidths = 0;
maxCharWidths = 0;
minCharWidth = Float.MAX_VALUE;
minCharHeight = Float.MAX_VALUE;
textPositionSequences.clear();
rulings.clear();
graphicsPath.clear();
@@ -17,6 +17,6 @@ public class ParsedElements {
private boolean landscape;
private boolean rotated;
private float maxCharWidth;
private float maxCharHeight;
private float minCharWidth;
private float minCharHeight;
}
@@ -5,11 +5,10 @@ import java.util.List;
import org.apache.pdfbox.text.TextPosition;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import lombok.Data;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.Setter;
@Data
@RequiredArgsConstructor
@@ -17,50 +16,42 @@ public class TextPositionSequence implements CharSequence {
private List<TextPosition> textPositions = new ArrayList<>();
@Getter
@Setter
private float[] annotationColor;
private final int page;
public TextPositionSequence(List<TextPosition> textPositions, int page) {
public TextPositionSequence(List<TextPosition> textPositions, int page){
this.textPositions = textPositions;
this.page = page;
}
@Override
public int length() {
return textPositions.size();
}
@Override
public char charAt(int index) {
TextPosition textPosition = textPositionAt(index);
String text = textPosition.getUnicode();
return text.charAt(0);
}
public char charAt(int index, boolean caseInSensitive) {
TextPosition textPosition = textPositionAt(index);
String text = textPosition.getUnicode();
return caseInSensitive ? text.toLowerCase().charAt(0) : text.charAt(0);
}
@Override
public TextPositionSequence subSequence(int start, int end) {
return new TextPositionSequence(textPositions.subList(start, end), page);
}
@Override
public String toString() {
StringBuilder builder = new StringBuilder(length());
for (int i = 0; i < length(); i++) {
builder.append(charAt(i));
@@ -68,21 +59,15 @@ public class TextPositionSequence implements CharSequence {
return builder.toString();
}
public TextPosition textPositionAt(int index) {
return textPositions.get(index);
}
public void add(TextPosition textPosition) {
this.textPositions.add(textPosition);
}
public float getX1() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getYDirAdj() - getTextHeight();
} else {
@@ -90,23 +75,15 @@ public class TextPositionSequence implements CharSequence {
}
}
public float getX2() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getYDirAdj();
} else {
return textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
return textPositions.get(textPositions.size() - 1).getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
}
}
public float getRotationAdjustedY() {
return textPositions.get(0).getY();
}
public float getY1() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(0).getXDirAdj();
} else {
@@ -114,46 +91,30 @@ public class TextPositionSequence implements CharSequence {
}
}
public float getY2() {
if (textPositions.get(0).getRotation() == 90) {
return textPositions.get(textPositions.size() - 1).getXDirAdj() + getTextHeight() - 2;
return textPositions.get(textPositions.size() - 1).getXDirAdj() + getTextHeight() -2 ;
} else {
return textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() + getTextHeight();
}
}
public float getTextHeight() {
return textPositions.get(0).getHeightDir() + 2;
}
public float getHeight() {
return getY2() - getY1();
}
public float getWidth() {
return getX2() - getX1();
}
public String getFont() {
return textPositions.get(0)
.getFont()
.toString()
.toLowerCase()
.replaceAll(",bold", "")
.replaceAll(",italic", "");
return textPositions.get(0).getFont().toString().toLowerCase().replaceAll(",bold", "").replaceAll(",italic", "");
}
public String getFontStyle() {
String lowercaseFontName = textPositions.get(0).getFont().toString().toLowerCase();
@@ -170,48 +131,16 @@ public class TextPositionSequence implements CharSequence {
}
public float getFontSize() {
return textPositions.get(0).getFontSizeInPt();
}
public float getSpaceWidth() {
return textPositions.get(0).getWidthOfSpace();
}
public int getRotation() {
return textPositions.get(0).getRotation();
}
public Rectangle getRectangle() {
float height = getTextHeight();
float posXInit = getX1();
float posXEnd;
float posYInit;
float posYEnd;
if (textPositions.get(0).getRotation() == 90) {
posXEnd = textPositions.get(0).getYDirAdj() + 2;
posYInit = getY1();
posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
} else {
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
.getYDirAdj() + 2;
}
return new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
}
}
@@ -1,50 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.Iterator;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import lombok.Value;
@Value
public class CellValue {
private List<TextBlock> textBlocks;
private int rowSpanStart;
@Override
public String toString() {
StringBuilder sb = new StringBuilder();
Iterator<TextBlock> itty = textBlocks.iterator();
while (itty.hasNext()) {
TextBlock textBlock = itty.next();
TextPositionSequence previous = null;
for (TextPositionSequence word : textBlock.getSequences()) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
}
}
sb.append(word.toString());
previous = word;
}
if (itty.hasNext()) {
sb.append(' ');
}
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" {2}", " ");
}
}
@@ -1,88 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import lombok.Data;
import lombok.Getter;
@Data
public class Dictionary {
public static final String RECOMMENDATION_PREFIX = "recommendation_";
@Getter
private List<DictionaryModel> dictionaryModels;
private Map<String, DictionaryModel> localAccessMap = new HashMap<>();
@Getter
private long version;
public Dictionary(List<DictionaryModel> dictionaryModels, long dictionaryVersion){
this.dictionaryModels = dictionaryModels;
this.dictionaryModels.forEach(dm -> localAccessMap.put(dm.getType(), dm));
this.version = dictionaryVersion;
}
public boolean isRecommendation(String type) {
DictionaryModel model = localAccessMap.get(type);
if (model != null) {
return model.isRecommendation();
}
return false;
}
public boolean hasLocalEntries() {
return dictionaryModels.stream().anyMatch(dm -> !dm.getLocalEntries().isEmpty());
}
public Set<String> getTypes() {
return localAccessMap.keySet();
}
public boolean containsValue(String type, String value) {
if (localAccessMap.containsKey(type) && localAccessMap.get(type)
.getEntries()
.contains(value) || localAccessMap.containsKey(type) && localAccessMap.get(type)
.getLocalEntries()
.contains(value) || localAccessMap.containsKey(RECOMMENDATION_PREFIX + type) && localAccessMap.get(RECOMMENDATION_PREFIX + type)
.getEntries()
.contains(value) || localAccessMap.containsKey(RECOMMENDATION_PREFIX + type) && localAccessMap.get(RECOMMENDATION_PREFIX + type)
.getLocalEntries()
.contains(value)) {
return true;
}
return false;
}
public boolean isHint(String type) {
DictionaryModel model = localAccessMap.get(type);
if (model != null) {
return model.isHint();
}
return false;
}
public boolean isCaseInsensitiveDictionary(String type) {
DictionaryModel dictionaryModel = localAccessMap.get(type);
if (dictionaryModel != null) {
return dictionaryModel.isCaseInsensitive();
}
return false;
}
}
@@ -1,27 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.io.Serializable;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class DictionaryModel implements Serializable {
private String type;
private int rank;
private float[] color;
private boolean caseInsensitive;
private boolean hint;
private boolean recommendation;
private Set<String> entries;
private Set<String> localEntries;
public Set<String> getValues(boolean local){
return local ? localEntries : entries;
}
}
@@ -1,24 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import lombok.Data;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
@Data
public class DictionaryRepresentation {
private String ruleSetId;
private long dictionaryVersion = -1;
private List<DictionaryModel> dictionary = new ArrayList<>();
private float[] defaultColor;
private float[] requestAddColor;
private float[] requestRemoveColor;
private float[] notRedactedColor;
private Map<String, DictionaryModel> localAccessMap = new HashMap<>();
}
@@ -1,46 +1,27 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import java.util.Objects;
import lombok.Data;
import lombok.EqualsAndHashCode;
@Data
@EqualsAndHashCode(onlyExplicitlyIncluded = true)
public class Entity {
private final String word;
private final String type;
private boolean redaction;
private String redactionReason;
private String legalBasis;
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
private List<TextPositionSequence> targetSequences;
@EqualsAndHashCode.Include
private Integer start;
@EqualsAndHashCode.Include
private Integer end;
@EqualsAndHashCode.Include
private String headline;
private int matchedRule;
@EqualsAndHashCode.Include
private int sectionNumber;
private boolean isDictionaryEntry;
private String textBefore;
private String textAfter;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber, String legalBasis, boolean isDictionaryEntry, String textBefore, String textAfter) {
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber) {
this.word = word;
this.type = type;
this.redaction = redaction;
@@ -49,22 +30,37 @@ public class Entity {
this.headline = headline;
this.matchedRule = matchedRule;
this.sectionNumber = sectionNumber;
this.legalBasis = legalBasis;
this.isDictionaryEntry = isDictionaryEntry;
this.textBefore = textBefore;
this.textAfter = textAfter;
}
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber, boolean isDictionaryEntry) {
public Entity(String word, String type, Integer start, Integer end, String headline, int sectionNumber) {
this.word = word;
this.type = type;
this.start = start;
this.end = end;
this.headline = headline;
this.sectionNumber = sectionNumber;
this.isDictionaryEntry = isDictionaryEntry;
}
@Override
public boolean equals(Object o) {
if (this == o) {
return true;
}
if (o == null || getClass() != o.getClass()) {
return false;
}
Entity entity = (Entity) o;
return sectionNumber == entity.sectionNumber && Objects.equals(word, entity.word) && Objects.equals(type, entity.type) && Objects
.equals(headline, entity.headline);
}
@Override
public int hashCode() {
return Objects.hash(word, type, headline, sectionNumber);
}
}
@@ -2,23 +2,44 @@ package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.ArrayList;
import java.util.List;
import java.util.Objects;
import java.util.UUID;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.RequiredArgsConstructor;
@Data
@RequiredArgsConstructor
@AllArgsConstructor
@EqualsAndHashCode
public class EntityPositionSequence {
@EqualsAndHashCode.Exclude
private List<TextPositionSequence> sequences = new ArrayList<>();
private int pageNumber;
private final String id;
private final UUID id;
@Override
public boolean equals(Object o) {
if (this == o) {
return true;
}
if (o == null || getClass() != o.getClass()) {
return false;
}
EntityPositionSequence that = (EntityPositionSequence) o;
return pageNumber == that.pageNumber && Objects.equals(id, that.id);
}
@Override
public int hashCode() {
return Objects.hash(pageNumber, id);
}
}
@@ -1,17 +1,17 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.ArrayList;
import java.util.Collections;
import java.util.List;
import java.util.UUID;
import java.util.regex.Pattern;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
@SuppressWarnings("all")
public class SearchableText {
private final List<TextPositionSequence> sequences = new ArrayList<>();
private List<TextPositionSequence> sequences = new ArrayList<>();
public void add(TextPositionSequence textPositionSequence) {
@@ -28,14 +28,6 @@ public class SearchableText {
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive) {
return getSequences(searchString, caseInsensitive, null);
}
@SuppressWarnings("checkstyle:ModifiedControlVariable")
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive,
List<TextPositionSequence> sequencesSubList) {
String normalizedSearchString;
if (caseInsensitive) {
normalizedSearchString = searchString.toLowerCase();
@@ -48,50 +40,37 @@ public class SearchableText {
List<TextPositionSequence> crossSequenceParts = new ArrayList<>();
List<EntityPositionSequence> finalMatches = new ArrayList<>();
for (int i = 0; i < sequences.size(); i++) {
TextPositionSequence partMatch = new TextPositionSequence(sequences.get(i).getPage());
for (int j = 0; j < sequences.get(i).length(); j++) {
List<TextPositionSequence> searchSpace;
if (sequencesSubList != null) {
int subListIndex = Collections.indexOfSubList(sequences, sequencesSubList);
if (subListIndex != -1) {
searchSpace = sequences.subList(subListIndex, subListIndex + sequencesSubList.size());
} else {
searchSpace = sequences;
}
} else {
searchSpace = sequences;
}
for (int i = 0; i < searchSpace.size(); i++) {
TextPositionSequence partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
for (int j = 0; j < searchSpace.get(i).length(); j++) {
if (i > 0 && j == 0 && searchSpace.get(i).charAt(0, caseInsensitive) == ' ' && searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i).charAt(j - 1, caseInsensitive) == ' ') {
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
if (i > 0 && j == 0 && sequences.get(i).charAt(0, caseInsensitive) == ' ' && sequences.get(i - 1)
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && sequences.get(i)
.charAt(j, caseInsensitive) == ' ' && sequences.get(i).charAt(j - 1, caseInsensitive) == ' ') {
if (j == sequences.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
crossSequenceParts.add(partMatch);
}
continue;
}
if (j == 0 && searchSpace.get(i).charAt(j, caseInsensitive) != ' ' && i != 0 && searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1)
if (j == 0 && sequences.get(i).charAt(j, caseInsensitive) != ' ' && i != 0 && sequences.get(i - 1)
.charAt(sequences.get(i - 1)
.length() - 1, caseInsensitive) != ' ' && searchChars[counter] == ' ') {
counter++;
}
if (searchSpace.get(i)
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace.get(i)
if (sequences.get(i)
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && sequences.get(i)
.charAt(j, caseInsensitive) == '-') {
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(searchSpace.get(i)
.charAt(j - 1, caseInsensitive)) || j == 0 && i != 0 && isSeparator(searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1)
.length() - 1, caseInsensitive)) || j == 0 && i != 0 && searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) != ' ' && searchSpace.get(i)
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(sequences.get(i)
.charAt(j - 1, caseInsensitive)) || j == 0 && i != 0 && isSeparator(sequences.get(i - 1)
.charAt(sequences.get(i - 1)
.length() - 1, caseInsensitive)) || j == 0 && i != 0 && sequences.get(i - 1)
.charAt(sequences.get(i - 1).length() - 1, caseInsensitive) != ' ' && sequences.get(i)
.charAt(j, caseInsensitive) != ' ') {
partMatch.add(searchSpace.get(i).textPositionAt(j));
if (!(j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
partMatch.add(sequences.get(i).textPositionAt(j));
if (!(j == sequences.get(i).length() - 1 && sequences.get(i)
.charAt(j, caseInsensitive) == '-' && searchChars[counter] != '-')) {
counter++;
}
@@ -100,19 +79,19 @@ public class SearchableText {
if (counter == searchString.length()) {
crossSequenceParts.add(partMatch);
if (i == searchSpace.size() - 1 && j == searchSpace.get(i).length() - 1 || j != searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i)
.charAt(j + 1, caseInsensitive)) || j == searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i + 1)
.charAt(0, caseInsensitive)) || j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
.charAt(j, caseInsensitive) != ' ' && searchSpace.get(i + 1)
if (i == sequences.size() - 1 && j == sequences.get(i).length() - 1 || j != sequences.get(i)
.length() - 1 && isSeparator(sequences.get(i)
.charAt(j + 1, caseInsensitive)) || j == sequences.get(i)
.length() - 1 && isSeparator(sequences.get(i + 1)
.charAt(0, caseInsensitive)) || j == sequences.get(i).length() - 1 && sequences.get(i)
.charAt(j, caseInsensitive) != ' ' && sequences.get(i + 1)
.charAt(0, caseInsensitive) != ' ') {
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts));
}
counter = 0;
crossSequenceParts = new ArrayList<>();
partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
partMatch = new TextPositionSequence(sequences.get(i).getPage());
}
} else {
counter = 0;
@@ -120,27 +99,24 @@ public class SearchableText {
j--;
}
crossSequenceParts = new ArrayList<>();
partMatch = new TextPositionSequence(searchSpace.get(i).getPage());
partMatch = new TextPositionSequence(sequences.get(i).getPage());
}
if (j == searchSpace.get(i).length() - 1 && counter != 0) {
if (j == sequences.get(i).length() - 1 && counter != 0) {
crossSequenceParts.add(partMatch);
}
}
}
return finalMatches;
}
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
String plainId = IdBuilder.buildId(crossSequenceParts);
String id = plainId;
UUID id = UUID.randomUUID();
List<EntityPositionSequence> result = new ArrayList<>();
int currentPage = -1;
int idDiffentPageSuffix = 1;
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
for (TextPositionSequence textPositionSequence : crossSequenceParts) {
if (currentPage == -1) {
@@ -150,13 +126,9 @@ public class SearchableText {
} else if (currentPage == textPositionSequence.getPage()) {
entityPositionSequence.getSequences().add(textPositionSequence);
} else {
id = plainId + "-" + idDiffentPageSuffix;
idDiffentPageSuffix++;
result.add(entityPositionSequence);
entityPositionSequence = new EntityPositionSequence(id);
entityPositionSequence.setPageNumber(textPositionSequence.getPage());
entityPositionSequence.getSequences().add(textPositionSequence);
currentPage = textPositionSequence.getPage();
}
}
result.add(entityPositionSequence);
@@ -179,7 +151,7 @@ public class SearchableText {
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -191,7 +163,7 @@ public class SearchableText {
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" {2}", " ");
.replaceAll(" ", " ");
}
@@ -203,7 +175,7 @@ public class SearchableText {
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -215,4 +187,4 @@ public class SearchableText {
return sb.append("\n").toString();
}
}
}
@@ -1,38 +1,20 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import static com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary.RECOMMENDATION_PREFIX;
import java.util.Collection;
import java.util.HashMap;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.Map;
import java.util.List;
import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import org.apache.commons.lang3.StringUtils;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
import lombok.Builder;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
@Data
@Slf4j
@Builder
public class Section {
private boolean isLocal;
private Set<String> dictionaryTypes;
@Builder.Default
private Map<String, Set<String>> localDictionaryAdds = new HashMap<>();
private Set<Entity> entities;
// This still contains linebreaks etc.
@@ -45,46 +27,25 @@ public class Section {
private int sectionNumber;
private Map<String, CellValue> tabularData;
private Dictionary dictionary;
private SearchableText searchableText;
public boolean rowEquals(String headerName, String value) {
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
return tabularData != null && tabularData.containsKey(cleanHeaderName) && tabularData.get(cleanHeaderName)
.toString()
.equals(value);
}
public boolean matchesType(String type) {
public boolean contains(String type) {
return entities.stream().anyMatch(entity -> entity.getType().equals(type));
}
public boolean headlineContainsWord(String word) {
public boolean headlineContainsWord(String word){
return StringUtils.containsIgnoreCase(headline, word);
}
public void redact(String type, int ruleNumber, String reason, String legalBasis) {
boolean hasRecommendationDictionary = dictionaryTypes.contains(RECOMMENDATION_PREFIX + type);
public void redact(String type, int ruleNumber, String reason) {
entities.forEach(entity -> {
if (entity.getType().equals(type) || hasRecommendationDictionary && entity.getType()
.equals(RECOMMENDATION_PREFIX + type)) {
if (entity.getType().equals(type)) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
@@ -92,11 +53,8 @@ public class Section {
public void redactNot(String type, int ruleNumber, String reason) {
boolean hasRecommendationDictionary = dictionaryTypes.contains(RECOMMENDATION_PREFIX + type);
entities.forEach(entity -> {
if (entity.getType().equals(type) || hasRecommendationDictionary && entity.getType()
.equals(RECOMMENDATION_PREFIX + type)) {
if (entity.getType().equals(type)) {
entity.setRedaction(false);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
@@ -105,231 +63,86 @@ public class Section {
}
public void redactIfPrecededBy(String prefix, String type, int ruleNumber, String reason, String legalBasis) {
public void redactLineAfter(String start, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(text, start, "\n");
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
}
// TODO No need to iterate
entities.forEach(entity -> {
if (entity.getType().equals(type) && searchText.indexOf(prefix + entity.getWord()) != 1) {
if (entity.getType().equals(asType)) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
}
});
}
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason) {
String value = StringUtils.substringBetween(searchText, start, stop);
if (value != null) {
Set<Entity> found = findEntity(value.trim(), asType);
entities.addAll(found);
}
// TODO No need to iterate
entities.forEach(entity -> {
if (entity.getType().equals(asType)) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
public void addHintAnnotation(String value, String asType) {
private Set<Entity> findEntity(String value, String asType) {
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null);
addNewerToEntities(found);
Set<Entity> found = new HashSet<>();
int startIndex;
int stopIndex = 0;
do {
startIndex = searchText.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText
.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline, sectionNumber));
}
} while (startIndex > -1);
return removeEntitiesContainedInLarger(found);
}
public void redactLineAfter(String start, String asType, int ruleNumber, boolean redactEverywhere, String reason,
String legalBasis) {
private boolean isSeparator(char c) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
public Set<Entity> removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
for (Entity inner : entities) {
if (inner.getWord().length() < word.getWord()
.length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
wordsToRemove.add(inner);
}
}
}
}
public void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
}
}
}
public void addRecommendationByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(text);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>())
.add(match);
}
}
}
public void redactBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere,
String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(searchText, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
}
}
}
}
public void redactLinesBetween(String start, String stop, String asType, int ruleNumber,
boolean redactEverywhere,
String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(text, start, stop);
if (values != null) {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
String[] lines = value.split("\n");
for (String line : lines) {
if (line.trim().length() <= 2) {
return;
}
Set<Entity> found = findEntities(line.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(line.trim());
}
}
}
}
}
}
public void highlightCell(String cellHeader, int ruleNumber, String type) {
annotateCell(cellHeader, ruleNumber, type, false, false, null, null);
}
public void redactCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations, String
reason,
String legalBasis) {
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis);
}
public void redactNotCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations,
String reason) {
annotateCell(cellHeader, ruleNumber, type, false, addAsRecommendations, reason, null);
}
private Set<Entity> findEntities(String value, String asType, boolean caseInsensitive, boolean redacted,
int ruleNumber, String reason, String legalBasis) {
String text = caseInsensitive ? searchText.toLowerCase() : searchText;
String searchValue = caseInsensitive ? value.toLowerCase() : value;
Set<Entity> found = EntitySearchUtils.find(text, Set.of(searchValue), asType, headline, sectionNumber, true);
found.forEach(entity -> {
if (redacted) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
return EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary);
}
private void annotateCell(String cellHeader, int ruleNumber, String type, boolean redact,
boolean addAsRecommendations, String reason, String legalBasis) {
String cleanHeaderName = cellHeader.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
CellValue value = tabularData.get(cleanHeaderName);
if (value == null) {
log.warn("Could not find any data for {}.", cellHeader);
} else {
String word = value.toString();
Entity entity = new Entity(word, type, value.getRowSpanStart(), value.getRowSpanStart() + word.length(), headline, sectionNumber, false);
entity.setRedaction(redact);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setTargetSequences(value.getTextBlocks()
.stream()
.map(TextBlock::getSequences)
.flatMap(Collection::stream)
.collect(Collectors.toList())); // Make sure no other cells with same content are highlighted
entity.setLegalBasis(legalBasis);
Set<Entity> singleEntitySet = new HashSet<>();
singleEntitySet.add(entity);
EntitySearchUtils.clearAndFindPositions(singleEntitySet, searchableText, dictionary);
addNewerToEntities(entity);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
if (addAsRecommendations && !isLocal()) {
String cleanedWord = word.replaceAll(",", " ").replaceAll(" ", " ").trim() + " ";
Pattern pattern = Patterns.AUTHOR_TABLE_SPITTER;
Matcher matcher = pattern.matcher(cleanedWord);
while (matcher.find()) {
String match = matcher.group().trim();
if (match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + type, (x) -> new HashSet<>())
.add(match);
String lastname = match.split(" ")[0];
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + type, (x) -> new HashSet<>())
.add(lastname);
}
}
}
}
}
private void addNewerToEntities(Set<Entity> found) {
// HashSet keeps the older value, but we want the new only.
entities.removeAll(found);
entities.addAll(found);
}
private void addNewerToEntities(Entity found) {
// HashSet keeps the older value, but we want the new only.
entities.remove(found);
entities.add(found);
entities.removeAll(wordsToRemove);
return entities;
}
}
@@ -1,13 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class SectionSearchableTextPair {
private Section section;
private SearchableText searchableText;
}
@@ -1,22 +1,6 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryRepresentation;
import feign.FeignException;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.SerializationUtils;
import org.springframework.stereotype.Service;
import java.awt.Color;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
@@ -24,59 +8,73 @@ import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
@Slf4j
import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import feign.FeignException;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Service
@RequiredArgsConstructor
@Slf4j
public class DictionaryService {
private final DictionaryClient dictionaryClient;
private long dictionaryVersion = -1;
private Map<String, DictionaryRepresentation> dictionariesByRuleSets = new HashMap<>();
@Getter
private Map<String, Set<String>> dictionary = new HashMap<>();
@Getter
private Map<String, float[]> entryColors = new HashMap<>();
@Getter
private List<String> hintTypes = new ArrayList<>();
@Getter
private List<String> caseInsensitiveTypes = new ArrayList<>();
@Getter
private float[] defaultColor;
public void updateDictionary(String ruleSetId) {
public void updateDictionary() {
long version = dictionaryClient.getVersion(ruleSetId);
var foundDictionary = dictionariesByRuleSets.get(ruleSetId);
if (foundDictionary == null || version > foundDictionary.getDictionaryVersion()) {
updateDictionaryEntry(ruleSetId, version);
long version = dictionaryClient.getVersion();
if (version > dictionaryVersion) {
dictionaryVersion = version;
updateDictionaryEntry();
}
}
private void updateDictionaryEntry(String ruleSetId, long version) {
private void updateDictionaryEntry() {
try {
DictionaryRepresentation dictionaryRepresentation = new DictionaryRepresentation();
TypeResponse typeResponse = dictionaryClient.getAllTypes(ruleSetId);
TypeResponse typeResponse = dictionaryClient.getAllTypes();
if (typeResponse != null && CollectionUtils.isNotEmpty(typeResponse.getTypes())) {
List<DictionaryModel> dictionary = typeResponse.getTypes()
entryColors = typeResponse.getTypes()
.stream()
.map(t -> new DictionaryModel(t.getType(), t.getRank(), convertColor(t.getHexColor()), t.isCaseInsensitive(), t
.isHint(), t.isRecommendation(), convertEntries(t), new HashSet<>()))
.sorted(Comparator.comparingInt(DictionaryModel::getRank).reversed())
.collect(Collectors.toMap(TypeResult::getType, TypeResult::getColor));
hintTypes = typeResponse.getTypes()
.stream()
.filter(TypeResult::isHint)
.map(TypeResult::getType)
.collect(Collectors.toList());
dictionary.forEach(dm -> dictionaryRepresentation.getLocalAccessMap().put(dm.getType(), dm));
Colors colors = dictionaryClient.getColors(ruleSetId);
dictionaryRepresentation.setDefaultColor(convertColor(colors.getDefaultColor()));
dictionaryRepresentation.setRequestAddColor(convertColor(colors.getRequestAdd()));
dictionaryRepresentation.setRequestRemoveColor(convertColor(colors.getRequestRemove()));
dictionaryRepresentation.setNotRedactedColor(convertColor(colors.getNotRedacted()));
dictionaryRepresentation.setRuleSetId(ruleSetId);
dictionaryRepresentation.setDictionaryVersion(version);
dictionaryRepresentation.setDictionary(dictionary);
dictionariesByRuleSets.put(ruleSetId, dictionaryRepresentation);
caseInsensitiveTypes = typeResponse.getTypes()
.stream()
.filter(TypeResult::isCaseInsensitive)
.map(TypeResult::getType)
.collect(Collectors.toList());
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, s -> convertEntries(s)));
defaultColor = dictionaryClient.getDefaultColor().getColor();
}
} catch (FeignException e) {
log.warn("Got some unknown feignException", e);
@@ -85,100 +83,15 @@ public class DictionaryService {
}
public void updateExternalDictionary(Dictionary dictionary, String ruleSetId) {
dictionary.getDictionaryModels().forEach(dm -> {
if (dm.isRecommendation() && !dm.getLocalEntries().isEmpty()) {
dictionaryClient.addEntries(dm.getType(), ruleSetId, new ArrayList<>(dm.getLocalEntries()), false);
long externalVersion = dictionaryClient.getVersion(ruleSetId);
if (externalVersion == dictionary.getVersion() + 1) {
dictionary.setVersion(externalVersion);
}
}
});
}
private Set<String> convertEntries(TypeResult t) {
if (t.isCaseInsensitive()) {
return dictionaryClient.getDictionaryForType(t.getType(), t.getRuleSetId())
private Set<String> convertEntries(String s) {
if (caseInsensitiveTypes.contains(s)) {
return dictionaryClient.getDictionaryForType(s)
.getEntries()
.stream()
.map(String::toLowerCase)
.collect(Collectors.toSet());
} else {
return new HashSet<>(dictionaryClient.getDictionaryForType(t.getType(), t.getRuleSetId()).getEntries());
}
return new HashSet<>(dictionaryClient.getDictionaryForType(s).getEntries());
}
private float[] convertColor(String hex) {
Color color = Color.decode(hex);
return new float[]{color.getRed() / 255f, color.getGreen() / 255f, color.getBlue() / 255f};
}
public boolean isCaseInsensitiveDictionary(String type, String ruleSetId) {
DictionaryModel dictionaryModel = dictionariesByRuleSets.get(ruleSetId).getLocalAccessMap().get(type);
if (dictionaryModel != null) {
return dictionaryModel.isCaseInsensitive();
}
return false;
}
public float[] getColor(String type, String ruleSetId) {
DictionaryModel model = dictionariesByRuleSets.get(ruleSetId).getLocalAccessMap().get(type);
if (model != null) {
return model.getColor();
}
return dictionariesByRuleSets.get(ruleSetId).getDefaultColor();
}
public boolean isHint(String type, String ruleSetId) {
DictionaryModel model = dictionariesByRuleSets.get(ruleSetId).getLocalAccessMap().get(type);
if (model != null) {
return model.isHint();
}
return false;
}
public boolean isRecommendation(String type, String ruleSetId) {
DictionaryModel model = dictionariesByRuleSets.get(ruleSetId).getLocalAccessMap().get(type);
if (model != null) {
return model.isRecommendation();
}
return false;
}
public Dictionary getDeepCopyDictionary(String ruleSetId) {
List<DictionaryModel> copy = new ArrayList<>();
var representation = dictionariesByRuleSets.get(ruleSetId);
var dictionary = dictionariesByRuleSets.get(ruleSetId).getDictionary();
dictionary.forEach(dm -> {
copy.add(SerializationUtils.clone(dm));
});
return new Dictionary(copy, representation.getDictionaryVersion());
}
public float[] getRequestRemoveColor(String ruleSetId) {
return dictionariesByRuleSets.get(ruleSetId).getRequestAddColor();
}
public float[] getNotRedactedColor(String ruleSetId) {
return dictionariesByRuleSets.get(ruleSetId).getNotRedactedColor();
}
public float[] getRequestAddColor(String ruleSetId) {
return dictionariesByRuleSets.get(ruleSetId).getRequestAddColor();
}
}
@@ -1,11 +1,9 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
@@ -13,13 +11,14 @@ import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.kie.api.runtime.KieSession;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Service;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.HashMap;
import java.util.Map;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
@@ -27,22 +26,12 @@ public class DroolsExecutionService {
private final RulesClient rulesClient;
private Map<String, KieContainer> kieContainers = new HashMap<>();
@Autowired
private KieContainer kieContainer;
public KieContainer getKieContainer(String ruleSetId) {
KieContainer container = kieContainers.get(ruleSetId);
if (container == null) {
return createOrUpdateKieContainer(ruleSetId);
} else {
return container;
}
}
@Getter
private long rulesVersion = -1;
public Section executeRules(KieContainer kieContainer, Section section) {
public Section executeRules(Section section) {
KieSession kieSession = kieContainer.newKieSession();
kieSession.setGlobal("section", section);
@@ -54,60 +43,34 @@ public class DroolsExecutionService {
}
public KieContainer updateRules(String ruleSetId) {
public void updateRules() {
long version = rulesClient.getVersion(ruleSetId);
long version = rulesClient.getVersion();
if (version > rulesVersion) {
rulesVersion = version;
return createOrUpdateKieContainer(ruleSetId);
updateRules(rulesClient.getRules().getRules());
}
return getKieContainer(ruleSetId);
}
private KieContainer createOrUpdateKieContainer(String ruleSetId) {
public void updateRules(String drlAsString) {
try {
RulesResponse rules = rulesClient.getRules(ruleSetId);
if (rules == null || StringUtils.isEmpty(rules.getRules())) {
if (StringUtils.isEmpty(drlAsString)) {
throw new RuntimeException("Rules cannot be empty.");
}
KieServices kieServices = KieServices.Factory.get();
KieModule kieModule = getKieModule(ruleSetId, rules.getRules(), kieServices);
var container = kieContainers.get(ruleSetId);
if (container != null) {
container.updateToVersion(kieModule.getReleaseId());
return container;
}
container = kieServices.newKieContainer(kieModule.getReleaseId());
kieContainers.put(ruleSetId, container);
return container;
InputStream input = new ByteArrayInputStream(drlAsString.getBytes(StandardCharsets.UTF_8));
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
kieFileSystem.write("src/main/resources/drools/rules.drl", kieServices.getResources().newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
kieContainer.updateToVersion(kieModule.getReleaseId());
} catch (Exception e) {
throw new RulesValidationException("Could not update rules: " + e.getMessage(), e);
}
}
private KieModule getKieModule(String ruleSetId, String rules, KieServices kieServices) {
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
InputStream input = new ByteArrayInputStream(rules.getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/main/resources/drools/rules" + ruleSetId + ".drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
return kieBuilder.getKieModule();
}
public void testRules(String rules) {
KieServices kieServices = KieServices.Factory.get();
KieModule kieModule = getKieModule("test-rules", rules, kieServices);
var container = kieServices.newKieContainer(kieModule.getReleaseId());
container.newKieSession();
container.dispose();
}
}
@@ -6,72 +6,81 @@ import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.concurrent.atomic.AtomicInteger;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import java.util.regex.Pattern;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.runtime.KieContainer;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.redaction.model.SectionSearchableTextPair;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class EntityRedactionService {
private final DictionaryService dictionaryService;
private final DroolsExecutionService droolsExecutionService;
private final SurroundingWordsService surroundingWordsService;
public void processDocument(Document classifiedDoc, String ruleSetId, ManualRedactions manualRedactions) {
public void processDocument(Document classifiedDoc) {
dictionaryService.updateDictionary(ruleSetId);
KieContainer container = droolsExecutionService.updateRules(ruleSetId);
long rulesVersion = droolsExecutionService.getRulesVersion();
dictionaryService.updateDictionary();
droolsExecutionService.updateRules();
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(ruleSetId);
Set<Entity> documentEntities = new HashSet<>();
int sectionNumber = 1;
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
Set<Entity> documentEntities = new HashSet<>(findEntities(classifiedDoc, container, manualRedactions, dictionary, false, null));
SearchableText searchableText = paragraph.getSearchableText();
if (dictionary.hasLocalEntries()) {
List<Table> tables = paragraph.getTables();
Map<Integer, Set<Entity>> hintsPerSectionNumber = new HashMap<>();
documentEntities.stream().forEach(entity -> {
if (dictionary.isHint(entity.getType())) {
hintsPerSectionNumber.computeIfAbsent(entity.getSectionNumber(), (x) -> new HashSet<>())
.add(entity);
for (Table table : tables) {
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
for (Cell column : row) {
if (column == null || column.getTextBlocks() == null) {
continue;
}
for (TextBlock textBlock : column.getTextBlocks()) {
searchableRow.addAll(textBlock.getSequences());
}
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber);
Section analysedRowSection = droolsExecutionService.executeRules(Section.builder()
.entities(rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber)
.build());
documentEntities.addAll(clearAndFindPositions(analysedRowSection.getEntities(), searchableRow));
sectionNumber++;
}
});
sectionNumber++;
}
Set<Entity> foundByLocal = findEntities(classifiedDoc, container, manualRedactions, dictionary, true, hintsPerSectionNumber);
// HashSet keeps the older value, but we want the new only.
documentEntities.removeAll(foundByLocal);
documentEntities.addAll(foundByLocal);
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline(), sectionNumber);
Section analysedSection = droolsExecutionService.executeRules(Section.builder()
.entities(entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(paragraph.getHeadline())
.sectionNumber(sectionNumber)
.build());
EntitySearchUtils.removeEntitiesContainedInLarger(documentEntities);
documentEntities.addAll(clearAndFindPositions(analysedSection.getEntities(), searchableText));
sectionNumber++;
}
for (Entity entity : documentEntities) {
@@ -85,237 +94,87 @@ public class EntityRedactionService {
classifiedDoc.getEntities()
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber(), entity
.getLegalBasis(), entity.isDictionaryEntry(), entity.getTextBefore(), entity.getTextAfter()));
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber()));
}
}
dictionaryService.updateExternalDictionary(dictionary, ruleSetId);
classifiedDoc.setDictionaryVersion(dictionary.getVersion());
classifiedDoc.setRulesVersion(rulesVersion);
}
private Set<Entity> findEntities(Document classifiedDoc, KieContainer kieContainer,
ManualRedactions manualRedactions, Dictionary dictionary, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
private Set<Entity> clearAndFindPositions(Set<Entity> entities, SearchableText text) {
Set<Entity> documentEntities = new HashSet<>();
Set<Entity> cleanEntities = removeEntitiesContainedInLarger(entities);
AtomicInteger sectionNumber = new AtomicInteger(1);
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
List<Table> tables = paragraph.getTables();
for (Table table : tables) {
if (table.getColCount() == 2) {
sectionSearchableTextPairs.addAll(processTableAsOneText(table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
} else {
sectionSearchableTextPairs.addAll(processTablePerRow(table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
}
sectionNumber.incrementAndGet();
}
sectionSearchableTextPairs.add(processText(paragraph, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionNumber.incrementAndGet();
}
sectionSearchableTextPairs.forEach(sectionSearchableTextPair -> {
Section analysedRowSection = droolsExecutionService.executeRules(kieContainer, sectionSearchableTextPair.getSection());
documentEntities.addAll(analysedRowSection.getEntities());
analysedRowSection.getLocalDictionaryAdds().keySet().forEach(key -> {
if (dictionary.isRecommendation(key)) {
analysedRowSection.getLocalDictionaryAdds().get(key).forEach(value -> {
if (!dictionary.containsValue(key, value)) {
dictionary.getLocalAccessMap().get(key).getLocalEntries().add(value);
}
});
} else {
analysedRowSection.getLocalDictionaryAdds().get(key).forEach(value -> {
if (dictionary.getLocalAccessMap().get(key) == null) {
log.warn("Dictionary {} is null", key);
}
if (dictionary.getLocalAccessMap().get(key).getLocalEntries() == null) {
log.warn("Dictionary {} localEntries is null", key);
}
dictionary.getLocalAccessMap().get(key).getLocalEntries().add(value);
});
}
});
});
return documentEntities;
}
private List<SectionSearchableTextPair> processTablePerRow(Table table, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
Map<String, CellValue> tabularData = new HashMap<>();
int start = 0;
List<Integer> cellStarts = new ArrayList<>();
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
addSectionToManualRedactions(cell.getTextBlocks(), manualRedactions, table.getHeadline(), sectionNumber.intValue());
int cellStart = start;
cell.getHeaderCells().forEach(headerCell -> {
StringBuilder headerBuilder = new StringBuilder();
headerCell.getTextBlocks().forEach(textBlock -> headerBuilder.append(textBlock.getText()));
String headerName = headerBuilder.toString()
.replaceAll("\n", "")
.replaceAll(" ", "")
.replaceAll("-", "");
tabularData.put(headerName, new CellValue(cell.getTextBlocks(), cellStart));
});
for (TextBlock textBlock : cell.getTextBlocks()) {
// TODO avoid cell overlap merging.
searchableRow.addAll(textBlock.getSequences());
}
cellStarts.add(cellStart);
start = start + cell.toString().trim().length() + 1;
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, searchableRow, dictionary, cellStarts);
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(rowEntities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : rowEntities)
.text(searchableRow.getAsStringWithLinebreaks())
.searchText(searchableRow.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber.intValue())
.tabularData(tabularData)
.searchableText(searchableRow)
.dictionary(dictionary)
.build(), searchableRow));
sectionNumber.incrementAndGet();
}
return sectionSearchableTextPairs;
}
private List<SectionSearchableTextPair> processTableAsOneText(Table table, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
SearchableText entireTableText = new SearchableText();
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
for (TextBlock textBlock : cell.getTextBlocks()) {
entireTableText.addAll(textBlock.getSequences());
}
addSectionToManualRedactions(cell.getTextBlocks(), manualRedactions, table.getHeadline(), sectionNumber.intValue());
for (Entity entity : cleanEntities) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entity.getType())) {
entity.setPositionSequences(text.getSequences(entity.getWord(), true));
} else {
entity.setPositionSequences(text.getSequences(entity.getWord(), false));
}
}
Set<Entity> rowEntities = findEntities(entireTableText, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, entireTableText, dictionary);
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(rowEntities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : rowEntities)
.text(entireTableText.getAsStringWithLinebreaks())
.searchText(entireTableText.toString())
.headline(table.getHeadline())
.sectionNumber(sectionNumber.intValue())
.searchableText(entireTableText)
.dictionary(dictionary)
.build(), entireTableText));
return sectionSearchableTextPairs;
return cleanEntities;
}
private SectionSearchableTextPair processText(Paragraph paragraph, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber) {
SearchableText searchableText = paragraph.getSearchableText();
addSectionToManualRedactions(paragraph.getTextBlocks(), manualRedactions, paragraph.getHeadline(), sectionNumber
.intValue());
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(entities, searchableText, dictionary);
return new SectionSearchableTextPair(Section.builder()
.isLocal(local)
.dictionaryTypes(dictionary.getTypes())
.entities(hintsPerSectionNumber != null && hintsPerSectionNumber.containsKey(sectionNumber.intValue()) ? Stream
.concat(entities.stream(), hintsPerSectionNumber.get(sectionNumber.intValue()).stream())
.collect(Collectors.toSet()) : entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(paragraph.getHeadline())
.sectionNumber(sectionNumber.intValue())
.searchableText(searchableText)
.dictionary(dictionary)
.build(), searchableText);
}
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber,
Dictionary dictionary, boolean local) {
String inputString = searchableText.toString();
String lowercaseInputString = inputString.toLowerCase();
Set<Entity> found = new HashSet<>();
String searchableString = searchableText.toString();
if (StringUtils.isEmpty(searchableString)) {
return found;
}
for (Map.Entry<String, Set<String>> entry : dictionaryService.getDictionary().entrySet()) {
String lowercaseInputString = searchableString.toLowerCase();
for (DictionaryModel model : dictionary.getDictionaryModels()) {
if (model.isCaseInsensitive()) {
found.addAll(EntitySearchUtils.find(lowercaseInputString, model.getValues(local), model.getType(), headline, sectionNumber, local));
if (dictionaryService.getCaseInsensitiveTypes().contains(entry.getKey())) {
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
} else {
found.addAll(EntitySearchUtils.find(searchableString, model.getValues(local), model.getType(), headline, sectionNumber, local));
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
}
}
return EntitySearchUtils.clearAndFindPositions(found, searchableText, dictionary);
return removeEntitiesContainedInLarger(found);
}
private void addSectionToManualRedactions(List<TextBlock> textBlocks, ManualRedactions manualRedactions,
String section, int sectionNumber) {
private Set<Entity> find(String inputString, Set<String> values, String type, String headline, int sectionNumber) {
if (manualRedactions == null || manualRedactions.getEntriesToAdd().isEmpty()) {
return;
Set<Entity> found = new HashSet<>();
for (String value : values) {
int startIndex;
int stopIndex = 0;
do {
startIndex = inputString.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(inputString.charAt(startIndex - 1)) || isSeparator(inputString
.charAt(startIndex - 1))) && (stopIndex == inputString.length() || isSeparator(inputString.charAt(stopIndex)))) {
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline, sectionNumber));
}
} while (startIndex > -1);
}
return found;
}
for (TextBlock textBlock : textBlocks) {
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
for (Rectangle rectangle : manualRedactionEntry.getPositions()) {
if (textBlock.contains(rectangle)) {
manualRedactionEntry.setSection(section);
manualRedactionEntry.setSectionNumber(sectionNumber);
}
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
public Set<Entity> removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
for (Entity inner : entities) {
if (inner.getWord().length() < word.getWord()
.length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner) {
wordsToRemove.add(inner);
}
}
}
entities.removeAll(wordsToRemove);
return entities;
}
}
@@ -1,140 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import java.util.List;
import java.util.Set;
@Slf4j
@Service
@RequiredArgsConstructor
public class SurroundingWordsService {
private final RedactionServiceSettings redactionServiceSettings;
public void addSurroundingText(Set<Entity> entities, SearchableText searchableText, Dictionary dictionary) {
if (entities.isEmpty()) {
return;
}
try {
for (Entity entity : entities) {
if (dictionary.isHint(entity.getType())) {
continue;
}
findSurroundingWords(entity, searchableText.toString(), entity.getStart(), entity.getEnd());
}
} catch (Exception e) {
log.warn("Could not get surrounding text!");
}
}
public void addSurroundingText(Set<Entity> entities, SearchableText searchableText, Dictionary dictionary,
List<Integer> cellstarts) {
if (entities.isEmpty()) {
return;
}
try {
String searchableString = searchableText.toString();
if (cellstarts != null) {
for (int i = 0; i < cellstarts.size(); i++) {
int startOffset = cellstarts.get(i);
int endOffset = -1;
if (i + 1 < cellstarts.size()) {
endOffset = cellstarts.get(i + 1);
} else {
endOffset = searchableString.length() - 1;
}
String text = searchableString.substring(startOffset, endOffset);
for (Entity entity : entities) {
if (dictionary.isHint(entity.getType())) {
continue;
}
if (entity.getStart() >= startOffset && entity.getEnd() <= endOffset) {
int entityStartOffset = entity.getStart() - startOffset;
int entityEndOffset = entity.getEnd() - startOffset;
findSurroundingWords(entity, text, entityStartOffset, entityEndOffset);
}
}
}
}
} catch (Exception e) {
log.warn("Could not get surrounding text!");
}
}
private void findSurroundingWords(Entity entity, String text, int entityStartOffset, int entityEndOffset) {
int offsetBefore = entityStartOffset - redactionServiceSettings.getSurroundingWordsOffsetWindow() < 0 ? 0 : entityStartOffset - redactionServiceSettings
.getSurroundingWordsOffsetWindow();
String textBefore = text.substring(offsetBefore, entityStartOffset);
if (!textBefore.isBlank()) {
String[] wordsBefore = textBefore.split(" ");
int numberOfWordsBefore = wordsBefore.length > redactionServiceSettings.getNumberOfSurroundingWords() ? redactionServiceSettings
.getNumberOfSurroundingWords() : wordsBefore.length;
if (wordsBefore.length > 0) {
entity.setTextBefore(concatWordsBefore(wordsBefore, numberOfWordsBefore, textBefore.endsWith(" ")));
}
}
int endOffset = entityEndOffset + redactionServiceSettings.getSurroundingWordsOffsetWindow() > text.length() ? text
.length() : entityEndOffset + redactionServiceSettings.getSurroundingWordsOffsetWindow();
String textAfter = text.substring(entityEndOffset, endOffset);
if (!textAfter.isBlank()) {
String[] wordsAfter = textAfter.split(" ");
int numberOfWordsAfter = wordsAfter.length > redactionServiceSettings.getNumberOfSurroundingWords() ? redactionServiceSettings
.getNumberOfSurroundingWords() : wordsAfter.length;
if (wordsAfter.length > 0) {
entity.setTextAfter(concatWordsAfter(wordsAfter, numberOfWordsAfter, textAfter.startsWith(" ")));
}
}
}
private String concatWordsBefore(String[] words, int number, boolean endWithSpace) {
StringBuilder sb = new StringBuilder();
int startNumber = words.length > number ? words.length - number : 0;
for (int i = startNumber; i < words.length; i++) {
sb.append(words[i]).append(" ");
}
String result = sb.toString().trim();
return endWithSpace ? result + " " : result;
}
private String concatWordsAfter(String[] words, int number, boolean startWithSpace) {
StringBuilder sb = new StringBuilder();
for (int i = 0; i < number; i++) {
sb.append(words[i]).append(" ");
}
String result = sb.toString().trim();
return startWithSpace ? " " + result : result;
}
}
@@ -1,107 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.experimental.UtilityClass;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@UtilityClass
public class EntitySearchUtils {
public Set<Entity> find(String inputString, Set<String> values, String type, String headline, int sectionNumber,
boolean local) {
Set<Entity> found = new HashSet<>();
for (String value : values) {
String cleanValue = value.trim();
if (cleanValue.length() <= 2) {
continue;
}
int startIndex;
int stopIndex = 0;
do {
startIndex = inputString.indexOf(cleanValue, stopIndex);
stopIndex = startIndex + cleanValue.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(inputString.charAt(startIndex - 1)) || isSeparator(inputString
.charAt(startIndex - 1))) && (stopIndex == inputString.length() || isSeparator(inputString.charAt(stopIndex)))) {
found.add(new Entity(inputString.substring(startIndex, stopIndex), type, startIndex, stopIndex, headline, sectionNumber, !local));
}
} while (startIndex > -1);
}
return found;
}
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
public Set<Entity> clearAndFindPositions(Set<Entity> entities, SearchableText text, Dictionary dictionary) {
Map<String, List<Entity>> entitiesByWord = new HashMap<>();
for (Entity entity : entities) {
entitiesByWord.computeIfAbsent(entity.getWord(), (x) -> new ArrayList<>()).add(entity);
}
for (String word : entitiesByWord.keySet()) {
List<Entity> orderedEntities = entitiesByWord.get(word)
.stream()
.sorted(Comparator.comparing(Entity::getStart))
.collect(Collectors.toList());
Entity firstEntity = orderedEntities.get(0);
List<EntityPositionSequence> positionSequences = text.getSequences(firstEntity.getWord().trim(), dictionary.isCaseInsensitiveDictionary(firstEntity
.getType()), firstEntity.getTargetSequences());
for (int i = 0; i <= orderedEntities.size() - 1; i++) {
try {
orderedEntities.get(i).setPositionSequences(List.of(positionSequences.get(i)));
} catch (Exception e) {
log.warn("Mismatch between EntityPositionSequence and found Entity!");
}
}
}
removeEntitiesContainedInLarger(entities);
return entities;
}
public void removeEntitiesContainedInLarger(Set<Entity> entities) {
List<Entity> wordsToRemove = new ArrayList<>();
for (Entity word : entities) {
for (Entity inner : entities) {
if (inner.getWord().length() < word.getWord()
.length() && inner.getStart() >= word.getStart() && inner.getEnd() <= word.getEnd() && word != inner && word
.getSectionNumber() == inner.getSectionNumber()) {
wordsToRemove.add(inner);
}
}
}
entities.removeAll(wordsToRemove);
}
}
@@ -1,26 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import java.nio.charset.StandardCharsets;
import java.util.List;
import com.google.common.hash.HashFunction;
import com.google.common.hash.Hashing;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.experimental.UtilityClass;
@UtilityClass
public class IdBuilder {
private final HashFunction hashFunction = Hashing.murmur3_128();
public String buildId(List<TextPositionSequence> crossSequenceParts) {
StringBuilder sb = new StringBuilder();
crossSequenceParts.forEach(sequencePart -> sequencePart.getTextPositions().forEach(textPosition -> {
sb.append(textPosition.getTextMatrix()).append(sequencePart.getPage());
}));
return hashFunction.hashString(sb.toString(), StandardCharsets.UTF_8).toString();
}
}
@@ -1,28 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import lombok.experimental.UtilityClass;
import java.util.HashMap;
import java.util.Map;
import java.util.regex.Pattern;
@UtilityClass
public class Patterns {
public static Map<String, Pattern> patternCache = new HashMap<>();
public static Pattern AUTHOR_TABLE_SPITTER = Pattern.compile("((((di)|(van)) )|[A-Z]’)?[A-ZÄÖÜ][\\wäöüéèê]{2,}( ?[A-ZÄÖÜ]{1,2}\\.)+|((((di)|(van)) )|[A-Z]’)?[A-ZÄÖÜ][\\wäöüéèê]{2,}( ?[A-ZÄÖÜ]{1,2} )+");
public Pattern getCompiledPattern(String pattern, boolean caseInsensitive) {
String patternKey = pattern + caseInsensitive;
if (patternCache.containsKey(patternKey)) {
return patternCache.get(patternKey);
}
Pattern compiledPattern = Pattern.compile(pattern, caseInsensitive ? Pattern.CASE_INSENSITIVE : 0);
patternCache.put(patternKey, compiledPattern);
return compiledPattern;
}
}
@@ -11,7 +11,7 @@ public class TextNormalizationUtilities {
* @return Text without line-break hyphenation.
*/
public static String removeHyphenLineBreaks(String text) {
return text.replaceAll("([^\\s\\d\\-]{2,})[\\-\\u00AD]\\R|\n\r(.+ )", "$1$2");
return text.replaceAll("\\s(\\S+)[\\-\\u00AD]\\R|\n\r(.+ )", "\n$1$2");
}
}
@@ -21,6 +21,7 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractT
import com.iqser.red.service.redaction.v1.server.tableextraction.model.CleanRulings;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.RulingCleaningService;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.TableExtractionService;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@@ -28,6 +29,7 @@ import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@SuppressWarnings("PMD")
public class PdfSegmentationService {
private final RulingCleaningService rulingCleaningService;
@@ -36,7 +38,6 @@ public class PdfSegmentationService {
private final ClassificationService classificationService;
private final SectionsBuilderService sectionsBuilderService;
public Document parseDocument(PDDocument pdDocument) throws IOException {
Document document = new Document();
@@ -56,21 +57,19 @@ public class PdfSegmentationService {
int rotation = pdPage.getRotation();
boolean isRotated = rotation != 0 && rotation != 360;
ParsedElements parsedElements = ParsedElements.builder()
ParsedElements parsedElements = ParsedElements
.builder()
.rulings(stripper.getRulings())
.sequences(stripper.getTextPositionSequences())
.maxCharWidth(stripper.getMaxCharWidths())
.maxCharHeight(stripper.getMaxCharWidths())
.minCharWidth(Utils.round(stripper.getMinCharWidth(), 2))
.minCharHeight(Utils.round(stripper.getMinCharHeight(), 2))
.landscape(isLandscape)
.rotated(isRotated)
.build();
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), parsedElements
.getMaxCharWidth(), parsedElements.getMaxCharHeight());
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), parsedElements.getMinCharWidth(), parsedElements.getMinCharHeight());
Page page = blockificationService.blockify(parsedElements.getSequences(), cleanRulings.getHorizontal(), cleanRulings
.getVertical());
Page page = blockificationService.blockify(parsedElements.getSequences(), cleanRulings.getHorizontal(), cleanRulings.getVertical());
page.setRotation(rotation);
tableExtractionService.extractTables(cleanRulings, page);
@@ -93,10 +92,7 @@ public class PdfSegmentationService {
}
private void increaseDocumentStatistics(Page page, Document document) {
if (!page.isLandscape()) {
document.getFontSizeCounter().addAll(page.getFontSizeCounter().getCountPerValue());
}
@@ -105,7 +101,6 @@ public class PdfSegmentationService {
document.getFontStyleCounter().addAll(page.getFontStyleCounter().getCountPerValue());
}
private void buildPageStatistics(Page page) {
// Collect all statistics for the page, except from blocks inside tables, as tables will always be added to BodyTextFrame.
@@ -1,12 +1,9 @@
package com.iqser.red.service.redaction.v1.server.segmentation;
import java.util.ArrayList;
import java.util.Collections;
import java.util.Iterator;
import java.util.List;
import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
@@ -14,10 +11,10 @@ import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@Service
@SuppressWarnings("all")
public class SectionsBuilderService {
public void buildSections(Document document) {
@@ -28,7 +25,6 @@ public class SectionsBuilderService {
AbstractTextContainer prev = null;
String lastHeadline = "";
Table previousTable = null;
for (Page page : document.getPages()) {
for (AbstractTextContainer current : page.getTextBlocks()) {
@@ -39,68 +35,32 @@ public class SectionsBuilderService {
current.setPage(page.getPageNumber());
if (prev != null && current.getClassification().startsWith("H ") && !prev.getClassification().startsWith("H ") || !document.isHeadlines()) {
if (prev != null && current.getClassification().startsWith("H ") || !document.isHeadlines()) {
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
if(document.isHeadlines()) {
lastHeadline = current.getText();
}
lastHeadline = current.getText();
chunkBlockList.add(chunkBlock);
chunkWords = new ArrayList<>();
if (CollectionUtils.isNotEmpty(chunkBlock.getTables())) {
previousTable = chunkBlock.getTables().get(chunkBlock.getTables().size() - 1);
}
}
if (current instanceof Table) {
Table table = (Table) current;
// Distribute header information for subsequent tables
mergeTableMetadata(table, previousTable);
previousTable = table;
}
chunkWords.add(current);
prev = current;
}
}
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
chunkBlockList.add(chunkBlock);
if (chunkBlock != null) {
chunkBlockList.add(chunkBlock);
chunkBlock.setHeadline(lastHeadline);
}
document.setParagraphs(chunkBlockList);
}
private void mergeTableMetadata(Table currentTable, Table previousTable) {
// Distribute header information for subsequent tables
if (previousTable != null && hasInvalidHeaderInformation(currentTable) && hasValidHeaderInformation(previousTable)) {
List<Cell> previousTableNonHeaderRow = getRowWithNonHeaderCells(previousTable);
List<Cell> tableNonHeaderRow = getRowWithNonHeaderCells(currentTable);
// Allow merging of tables if header row is separated from first logical non-header row
if (previousTableNonHeaderRow.isEmpty() && previousTable.getRowCount() == 1 && previousTable.getRows()
.get(0)
.size() == tableNonHeaderRow.size()) {
previousTableNonHeaderRow = previousTable.getRows().get(0).stream().map(cell -> {
Cell fakeCell = new Cell(cell.getPoints()[0], cell.getPoints()[2]);
fakeCell.setHeaderCells(Collections.singletonList(cell));
return fakeCell;
}).collect(Collectors.toList());
}
if (previousTableNonHeaderRow.size() == tableNonHeaderRow.size()) {
for (int i = currentTable.getRowCount() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = currentTable.getRows().get(i);
if (row.size() == tableNonHeaderRow.size() && row.stream()
.allMatch(cell -> cell.getHeaderCells().isEmpty())) {
for (int j = 0; j < row.size(); j++) {
row.get(j).setHeaderCells(previousTableNonHeaderRow.get(j).getHeaderCells());
}
}
}
}
}
}
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline) {
Paragraph paragraph = new Paragraph();
@@ -116,20 +76,19 @@ public class SectionsBuilderService {
AbstractTextContainer container = itty.next();
if (container instanceof Table) {
Table table = (Table) container;
splitByTable = true;
if (previous != null && previous.getText().startsWith("Table ")) {
table.setHeadline(previous.getText());
if (previous != null && previous instanceof TextBlock && previous.getText().startsWith("Table ")) {
((Table) container).setHeadline(previous.getText());
} else {
table.setHeadline("Table in: " + lastHeadline);
((Table) container).setHeadline("Table in: " + lastHeadline);
}
if (textBlock != null && !alreadyAdded) {
paragraph.getPageBlocks().add(textBlock);
alreadyAdded = true;
}
paragraph.getPageBlocks().add(table);
paragraph.getPageBlocks().add(container);
continue;
}
@@ -166,42 +125,4 @@ public class SectionsBuilderService {
return paragraph;
}
private boolean hasValidHeaderInformation(Table table) {
return !hasInvalidHeaderInformation(table);
}
private boolean hasInvalidHeaderInformation(Table table) {
return table.getRows().stream()
.flatMap(row -> row.stream()
.filter(cell -> CollectionUtils.isNotEmpty(cell.getHeaderCells())))
.findAny()
.isEmpty();
}
private List<Cell> getRowWithNonHeaderCells(Table table) {
for (int i = table.getRowCount() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = table.getRows().get(i);
boolean allNonHeader = true;
for (Cell cell : row) {
if (cell.isHeaderCell()) {
allNonHeader = false;
break;
}
}
if (allNonHeader) {
return row;
}
}
return Collections.emptyList();
}
}
}
@@ -7,9 +7,12 @@ import lombok.Data;
@Data
@ConfigurationProperties("redaction-service")
public class RedactionServiceSettings {
private int numberOfSurroundingWords = 3;
private int surroundingWordsOffsetWindow = 100;
/**
* Tenant used in single tenant mode.
*/
private String defaultTenant = "iqser-id";
private int flattenImageDpi = 100;
}
@@ -1,7 +1,5 @@
package com.iqser.red.service.redaction.v1.server.tableextraction.model;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -24,16 +22,4 @@ public abstract class AbstractTextContainer {
return this.minX <= other.minX && this.maxX >= other.maxX && this.minY >= other.minY && this.maxY <= other.maxY;
}
public boolean contains(Rectangle other) {
return page == other.getPage() && this.minX <= other.getTopLeft().getX() && this.maxX >= other.getTopLeft().getX() + other.getWidth() && this.minY <= other.getTopLeft().getY() && this.maxY >= other.getTopLeft().getY() + other.getHeight();
}
public float getHeight() {
return maxY - minY;
}
public float getWidth() {
return maxX - minX;
}
}
@@ -2,12 +2,9 @@ package com.iqser.red.service.redaction.v1.server.tableextraction.model;
import java.awt.geom.Point2D;
import java.util.ArrayList;
import java.util.Iterator;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import lombok.Data;
import lombok.EqualsAndHashCode;
@@ -19,57 +16,11 @@ public class Cell extends Rectangle {
private List<TextBlock> textBlocks = new ArrayList<>();
private List<Cell> headerCells = new ArrayList<>();
private boolean isHeaderCell;
public Cell(Point2D topLeft, Point2D bottomRight) {
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight
.getY() - topLeft.getY()));
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight.getY() - topLeft.getY()));
}
public void addTextBlock(TextBlock textBlock) {
textBlocks.add(textBlock);
}
@Override
public String toString() {
StringBuilder sb = new StringBuilder();
Iterator<TextBlock> itty = textBlocks.iterator();
TextPositionSequence previous = null;
while (itty.hasNext()) {
TextBlock textBlock = itty.next();
for (TextPositionSequence word : textBlock.getSequences()) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
}
}
sb.append(word.toString());
previous = word;
}
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" {2}", " ");
}
}
}
@@ -1,22 +0,0 @@
package com.iqser.red.service.redaction.v1.server.tableextraction.model;
import lombok.RequiredArgsConstructor;
import lombok.Value;
@Value
@RequiredArgsConstructor
public class CellPosition implements Comparable<CellPosition> {
int row;
int col;
@Override
public int compareTo(CellPosition other) {
int rowDiff = row - other.row;
return rowDiff != 0 ? rowDiff : col - other.col;
}
}
@@ -8,28 +8,25 @@ import org.locationtech.jts.index.strtree.STRtree;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@SuppressWarnings("all")
public class RectangleSpatialIndex<T extends Rectangle> {
private final STRtree si = new STRtree();
private final List<T> rectangles = new ArrayList<>();
public void add(T te) {
rectangles.add(te);
si.insert(new Envelope(te.getLeft(), te.getRight(), te.getBottom(), te.getTop()), te);
}
public List<T> contains(Rectangle rectangle) {
List<T> intersection = si.query(new Envelope(rectangle.getLeft(), rectangle.getRight(), rectangle.getTop(), rectangle
.getBottom()));
public List<T> contains(Rectangle r) {
List<T> intersection = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
List<T> rv = new ArrayList<T>();
for (T ir : intersection) {
if (rectangle.contains(ir)) {
for (T ir: intersection) {
if (r.contains(ir)) {
rv.add(ir);
}
}
@@ -37,22 +34,18 @@ public class RectangleSpatialIndex<T extends Rectangle> {
Utils.sort(rv, Rectangle.ILL_DEFINED_ORDER);
return rv;
}
public List<T> intersects(Rectangle r) {
List rv = si.query(new Envelope(r.getLeft(), r.getRight(), r.getTop(), r.getBottom()));
return rv;
}
/**
* Minimum bounding box of all the Rectangles contained on this RectangleSpatialIndex
*
*
* @return a Rectangle
*/
public Rectangle getBounds() {
return Rectangle.boundingBoxOf(rectangles);
}
@@ -9,38 +9,31 @@ import java.util.List;
import java.util.Map;
import java.util.TreeMap;
import org.apache.commons.collections4.CollectionUtils;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.Getter;
import lombok.Setter;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@SuppressWarnings("all")
public class Table extends AbstractTextContainer {
private final TreeMap<CellPosition, Cell> cells = new TreeMap<>();
private final RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
private RectangleSpatialIndex<Cell> si = new RectangleSpatialIndex<>();
@Getter
@Setter
private String headline;
private int unrotatedRowCount;
@Getter
private int rowCount = 0;
@Getter
private int colCount = 0;
private int unrotatedColCount;
private int rowCount = -1;
private int colCount = -1;
private final int rotation;
private List<List<Cell>> rows;
private int rotation = 0;
private List<List<Cell>> memoizedRows = null;
public Table(List<Cell> cells, Rectangle area, int rotation) {
@@ -54,129 +47,43 @@ public class Table extends AbstractTextContainer {
}
public List<List<Cell>> getRows() {
if (rows == null) {
rows = computeRows();
computeHeaders();
if (memoizedRows == null) {
memoizedRows = computeRows();
}
return rows;
return memoizedRows;
}
public int getRowCount() {
if (rowCount == -1) {
rowCount = getRows().size();
}
return rowCount;
}
public int getColCount() {
if (colCount == -1) {
colCount = getRows().stream().mapToInt(List::size).max().orElse(0);
}
return colCount;
}
/**
* Detect header cells (either first row or first column):
* Column is marked as header if cell text is bold and row cell text is not bold.
* Defaults to row.
*/
private void computeHeaders() {
if (rows == null) {
rows = computeRows();
}
// A bold cell is a header cell as long as every cell to the left/top is bold, too
// we move from left to right and top to bottom
for (int rowIndex = 0; rowIndex < rows.size(); rowIndex++) {
List<Cell> rowCells = rows.get(rowIndex);
for (int colIndex = 0; colIndex < rowCells.size(); colIndex++) {
Cell cell = rowCells.get(colIndex);
List<Cell> cellsToTheLeft = rowCells.subList(0, colIndex);
Cell lastHeaderCell = null;
for (Cell leftCell : cellsToTheLeft) {
if (leftCell.isHeaderCell()) {
lastHeaderCell = leftCell;
} else {
break;
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
List<Cell> cellsToTheTop = new ArrayList<>();
for (int i = 0; i < rowIndex; i++) {
try {
cellsToTheTop.add(rows.get(i).get(colIndex));
} catch (IndexOutOfBoundsException e) {
log.debug("No cell {} in row {}, ignoring.", colIndex, rowIndex);
}
}
for (Cell topCell : cellsToTheTop) {
if (topCell.isHeaderCell()) {
lastHeaderCell = topCell;
} else {
break;
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
if (CollectionUtils.isNotEmpty(cell.getTextBlocks()) && cell.getTextBlocks()
.get(0)
.getMostPopularWordStyle()
.equals("bold")) {
cell.setHeaderCell(true);
}
}
}
}
private List<List<Cell>> computeRows() {
List<List<Cell>> rows = new ArrayList<>();
if (rotation == 90) {
for (int i = 0; i < unrotatedColCount; i++) { // rows
for (int i = 0; i < colCount; i++) { // rows
List<Cell> lastRow = new ArrayList<>();
for (int j = unrotatedRowCount - 1; j >= 0; j--) { // cols
for (int j = rowCount - 1; j >= 0; j--) { // cols
Cell cell = cells.get(new CellPosition(j, i));
if (cell != null) {
lastRow.add(cell);
}
lastRow.add(cell);
}
rows.add(lastRow);
}
} else if (rotation == 270) {
for (int i = unrotatedColCount - 1; i >= 0; i--) { // rows
for (int i = colCount - 1; i >= 0; i--) { // rows
List<Cell> lastRow = new ArrayList<>();
for (int j = 0; j < unrotatedRowCount; j++) { // cols
for (int j = 0; j < rowCount; j++) { // cols
Cell cell = cells.get(new CellPosition(i, j));
if (cell != null) {
lastRow.add(cell);
}
lastRow.add(cell);
}
rows.add(lastRow);
}
} else {
for (int i = 0; i < unrotatedRowCount; i++) {
for (int i = 0; i < rowCount; i++) {
List<Cell> lastRow = new ArrayList<>();
for (int j = 0; j < unrotatedColCount; j++) {
for (int j = 0; j < colCount; j++) {
Cell cell = cells.get(new CellPosition(i, j)); // JAVA_8 use getOrDefault()
if (cell != null) {
lastRow.add(cell);
}
lastRow.add(cell);
}
rows.add(lastRow);
}
@@ -186,18 +93,16 @@ public class Table extends AbstractTextContainer {
}
public void add(Cell chunk, int row, int col) {
private void add(Cell chunk, int row, int col) {
unrotatedRowCount = Math.max(unrotatedRowCount, row + 1);
unrotatedColCount = Math.max(unrotatedColCount, col + 1);
rowCount = Math.max(rowCount, row + 1);
colCount = Math.max(colCount, col + 1);
CellPosition cp = new CellPosition(row, col);
cells.put(cp, chunk);
}
private void addCells(List<Cell> cells) {
if (cells.isEmpty()) {
@@ -226,21 +131,25 @@ public class Table extends AbstractTextContainer {
while (rowCells.hasNext()) {
Cell cell = rowCells.next();
if (i > 0) {
Rectangle rectangle = new Rectangle(cell.getBottom(),
si.getBounds().getLeft(),
cell.getLeft() - si.getBounds().getLeft() + 1,
si.getBounds().getBottom() - cell.getBottom());
List<List<Cell>> others = rowsOfCells(si.contains(rectangle));
List<List<Cell>> others = rowsOfCells(
si.contains(
new Rectangle(cell.getBottom(),
si.getBounds().getLeft(),
cell.getLeft() - si.getBounds().getLeft() + 1,
si.getBounds().getBottom() - cell.getBottom()
)
));
for (List<Cell> r : others) {
jumpToColumn = Math.max(jumpToColumn, r.size());
}
while (startColumn != jumpToColumn) {
add(previousNonNullCellForColumnIndex.get(startColumn), i, startColumn);
startColumn++;
}
}
while (startColumn != jumpToColumn) {
add(previousNonNullCellForColumnIndex.get(startColumn), i, startColumn);
startColumn++;
}
add(cell, i, startColumn);
previousNonNullCellForColumnIndex.put(startColumn, cell);
startColumn++;
@@ -249,24 +158,34 @@ public class Table extends AbstractTextContainer {
}
}
private List<List<Cell>> rowsOfCells(List<Cell> cells) {
private static List<List<Cell>> rowsOfCells(List<Cell> cells) {
Cell c;
float lastTop;
List<List<Cell>> rv = new ArrayList<>();
List<Cell> lastRow;
if (cells.isEmpty()) {
return rv;
}
cells.sort(Comparator.comparingDouble(Rectangle::getLeft));
cells.sort(Collections.reverseOrder((arg0, arg1) -> Float.compare(Utils.round(arg0.getBottom(), 2),
Utils.round(arg1
.getBottom(), 2))));
Collections.sort(cells, new Comparator<Cell>() {
@Override
public int compare(Cell arg0, Cell arg1) {
return Double.compare(arg0.getLeft(), arg1.getLeft());
}
});
Collections.sort(cells, Collections.reverseOrder(new Comparator<Cell>() {
@Override
public int compare(Cell arg0, Cell arg1) {
return Float.compare(Utils.round(arg0.getBottom(), 2), Utils.round(arg1.getBottom(),2));
}
}));
Iterator<Cell> iter = cells.iterator();
Cell c = iter.next();
float lastTop = c.getBottom();
List<Cell> lastRow = new ArrayList<>();
c = iter.next();
lastTop = c.getBottom();
lastRow = new ArrayList<>();
lastRow.add(c);
rv.add(lastRow);
@@ -282,7 +201,6 @@ public class Table extends AbstractTextContainer {
return rv;
}
@Override
public String getText() {
@@ -319,7 +237,6 @@ public class Table extends AbstractTextContainer {
return sb.toString();
}
public String getTextAsHtml() {
StringBuilder sb = new StringBuilder();
@@ -353,4 +270,41 @@ public class Table extends AbstractTextContainer {
return sb.toString();
}
class CellPosition implements Comparable<CellPosition> {
CellPosition(int row, int col) {
this.row = row;
this.col = col;
}
final int row, col;
@Override
public int hashCode() {
return row + 101 * col;
}
@Override
public boolean equals(Object obj) {
if (this == obj) {
return true;
}
if (obj == null) {
return false;
}
if (getClass() != obj.getClass()) {
return false;
}
CellPosition other = (CellPosition) obj;
return row == other.row && col == other.col;
}
@Override
public int compareTo(CellPosition other) {
int rowdiff = row - other.row;
return rowdiff != 0 ? rowdiff : col - other.col;
}
}
}
@@ -18,9 +18,9 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@Service
public class RulingCleaningService {
public CleanRulings getCleanRulings(List<Ruling> rulings, float maxCharWidth, float maxCharHeight){
public CleanRulings getCleanRulings(List<Ruling> rulings, float minCharWidth, float minCharHeight){
if (!rulings.isEmpty()) {
snapPoints(rulings, maxCharWidth , maxCharHeight);
snapPoints(rulings, minCharWidth , minCharHeight);
}
List<Ruling> vrs = new ArrayList<>();
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.server.tableextraction.service;
import java.awt.geom.Point2D;
import java.util.ArrayList;
import java.util.Collections;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
@@ -24,29 +25,26 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@Service
@SuppressWarnings("all")
public class TableExtractionService {
public void extractTables(CleanRulings cleanRulings, Page page) {
public void extractTables(CleanRulings cleanRulings, Page page){
List<Cell> cells = findCells(cleanRulings.getHorizontal(), cleanRulings.getVertical());
List<TextBlock> toBeRemoved = new ArrayList<>();
for (AbstractTextContainer abstractTextContainer : page.getTextBlocks()) {
TextBlock textBlock = (TextBlock) abstractTextContainer;
Iterator<AbstractTextContainer> itty = page.getTextBlocks().iterator();
while (itty.hasNext()) {
TextBlock textBlock = (TextBlock) itty.next();
for (Cell cell : cells) {
if (cell.intersects(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight())) {
cell.addTextBlock(textBlock);
toBeRemoved.add(textBlock);
break;
}
}
}
cells = new ArrayList<>(new HashSet<>(cells));
Utils.sort(cells, Rectangle.ILL_DEFINED_ORDER);
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells).stream()
List<Rectangle> spreadsheetAreas = findSpreadsheetsFromCells(cells)
.stream()
.filter(r -> r.getWidth() > 0f && r.getHeight() > 0f)
.collect(Collectors.toList());
@@ -65,32 +63,33 @@ public class TableExtractionService {
for (Table table : tables) {
int position = -1;
Iterator<AbstractTextContainer> itty = page.getTextBlocks().iterator();
itty = page.getTextBlocks().iterator();
while (itty.hasNext()) {
AbstractTextContainer textBlock = itty.next();
if (table.contains(textBlock) && position == -1) {
position = page.getTextBlocks().indexOf(textBlock);
AbstractTextContainer textBlock = (AbstractTextContainer) itty.next();
if (table.contains(textBlock)) {
if (position == -1) {
position = page.getTextBlocks().indexOf(textBlock);
}
itty.remove();
}
}
if (position != -1) {
page.getTextBlocks().add(position, table);
}
}
page.getTextBlocks().removeAll(toBeRemoved);
}
public List<Cell> findCells(List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
List<Cell> cellsFound = new ArrayList<>();
Map<Point2D, Ruling[]> intersectionPoints = Ruling.findIntersections(horizontalRulingLines, verticalRulingLines);
List<Point2D> intersectionPointsList = new ArrayList<>(intersectionPoints.keySet());
intersectionPointsList.sort(POINT_COMPARATOR);
Collections.sort(intersectionPointsList, POINT_COMPARATOR);
boolean doBreak;
for (int i = 0; i < intersectionPointsList.size(); i++) {
Point2D topLeft = intersectionPointsList.get(i);
Ruling[] hv = intersectionPoints.get(topLeft);
doBreak = false;
// CrossingPointsDirectlyBelow( topLeft );
List<Point2D> xPoints = new ArrayList<>();
@@ -107,6 +106,10 @@ public class TableExtractionService {
}
outer:
for (Point2D xPoint : xPoints) {
if (doBreak) {
break;
}
// is there a vertical edge b/w topLeft and xPoint?
if (!hv[1].equals(intersectionPoints.get(xPoint)[1])) {
continue;
@@ -117,9 +120,11 @@ public class TableExtractionService {
continue;
}
Point2D btmRight = new Point2D.Float((float) yPoint.getX(), (float) xPoint.getY());
if (intersectionPoints.containsKey(btmRight) && intersectionPoints.get(btmRight)[0].equals(intersectionPoints
.get(xPoint)[0]) && intersectionPoints.get(btmRight)[1].equals(intersectionPoints.get(yPoint)[1])) {
if (intersectionPoints.containsKey(btmRight)
&& intersectionPoints.get(btmRight)[0].equals(intersectionPoints.get(xPoint)[0])
&& intersectionPoints.get(btmRight)[1].equals(intersectionPoints.get(yPoint)[1])) {
cellsFound.add(new Cell(topLeft, btmRight));
doBreak = true;
break outer;
}
}
@@ -134,7 +139,7 @@ public class TableExtractionService {
}
private List<Rectangle> findSpreadsheetsFromCells(List<? extends Rectangle> cells) {
public List<Rectangle> findSpreadsheetsFromCells(List<? extends Rectangle> cells) {
// via: http://stackoverflow.com/questions/13746284/merging-multiple-adjacent-rectangles-into-one-polygon
List<Rectangle> rectangles = new ArrayList<>();
Set<Point2D> pointSet = new HashSet<>();
@@ -142,6 +147,10 @@ public class TableExtractionService {
Map<Point2D, Point2D> edgesV = new HashMap<>();
int i = 0;
cells = new ArrayList<>(new HashSet<>(cells));
Utils.sort(cells, Rectangle.ILL_DEFINED_ORDER);
for (Rectangle cell : cells) {
for (Point2D pt : cell.getPoints()) {
if (pointSet.contains(pt)) { // shared vertex, remove it
@@ -154,10 +163,10 @@ public class TableExtractionService {
// X first sort
List<Point2D> pointsSortX = new ArrayList<>(pointSet);
pointsSortX.sort(X_FIRST_POINT_COMPARATOR);
Collections.sort(pointsSortX, X_FIRST_POINT_COMPARATOR);
// Y first sort
List<Point2D> pointsSortY = new ArrayList<>(pointSet);
pointsSortY.sort(POINT_COMPARATOR);
Collections.sort(pointsSortY, POINT_COMPARATOR);
while (i < pointSet.size()) {
float currY = (float) pointsSortY.get(i).getY();
@@ -194,12 +203,13 @@ public class TableExtractionService {
nextVertex = edgesV.get(curr.point);
edgesV.remove(curr.point);
lastAddedVertex = new PolygonVertex(nextVertex, Direction.VERTICAL);
polygon.add(lastAddedVertex);
} else {
nextVertex = edgesH.get(curr.point);
edgesH.remove(curr.point);
lastAddedVertex = new PolygonVertex(nextVertex, Direction.HORIZONTAL);
polygon.add(lastAddedVertex);
}
polygon.add(lastAddedVertex);
if (lastAddedVertex.equals(polygon.get(0))) {
// closed polygon
@@ -217,10 +227,10 @@ public class TableExtractionService {
// calculate grid-aligned minimum area rectangles for each found polygon
for (List<PolygonVertex> poly : polygons) {
float top = Float.MAX_VALUE;
float left = Float.MAX_VALUE;
float bottom = Float.MIN_VALUE;
float right = Float.MIN_VALUE;
float top = java.lang.Float.MAX_VALUE;
float left = java.lang.Float.MAX_VALUE;
float bottom = java.lang.Float.MIN_VALUE;
float right = java.lang.Float.MIN_VALUE;
for (PolygonVertex pt : poly) {
top = (float) Math.min(top, pt.point.getY());
left = (float) Math.min(left, pt.point.getX());
@@ -234,66 +244,69 @@ public class TableExtractionService {
}
private static final Comparator<Point2D> X_FIRST_POINT_COMPARATOR = (arg0, arg1) -> {
private static final Comparator<Point2D> X_FIRST_POINT_COMPARATOR = new Comparator<Point2D>() {
@Override
public int compare(Point2D arg0, Point2D arg1) {
int rv = 0;
float arg0X = Utils.round(arg0.getX(), 2);
float arg0Y = Utils.round(arg0.getY(), 2);
float arg1X = Utils.round(arg1.getX(), 2);
float arg1Y = Utils.round(arg1.getY(), 2);
int rv = 0;
float arg0X = Utils.round(arg0.getX(), 2);
float arg0Y = Utils.round(arg0.getY(), 2);
float arg1X = Utils.round(arg1.getX(), 2);
float arg1Y = Utils.round(arg1.getY(), 2);
if (arg0X > arg1X) {
rv = 1;
} else if (arg0X < arg1X) {
rv = -1;
} else if (arg0Y > arg1Y) {
rv = 1;
} else if (arg0Y < arg1Y) {
rv = -1;
if (arg0X > arg1X) {
rv = 1;
} else if (arg0X < arg1X) {
rv = -1;
} else if (arg0Y > arg1Y) {
rv = 1;
} else if (arg0Y < arg1Y) {
rv = -1;
}
return rv;
}
return rv;
};
private static final Comparator<Point2D> POINT_COMPARATOR = (arg0, arg1) -> {
int rv = 0;
float arg0X = Utils.round(arg0.getX(), 2);
float arg0Y = Utils.round(arg0.getY(), 2);
float arg1X = Utils.round(arg1.getX(), 2);
float arg1Y = Utils.round(arg1.getY(), 2);
private static final Comparator<Point2D> POINT_COMPARATOR = new Comparator<Point2D>() {
@Override
public int compare(Point2D arg0, Point2D arg1) {
int rv = 0;
float arg0X = Utils.round(arg0.getX(), 2);
float arg0Y = Utils.round(arg0.getY(), 2);
float arg1X = Utils.round(arg1.getX(), 2);
float arg1Y = Utils.round(arg1.getY(), 2);
if (arg0Y > arg1Y) {
rv = 1;
} else if (arg0Y < arg1Y) {
rv = -1;
} else if (arg0X > arg1X) {
rv = 1;
} else if (arg0X < arg1X) {
rv = -1;
if (arg0Y > arg1Y) {
rv = 1;
} else if (arg0Y < arg1Y) {
rv = -1;
} else if (arg0X > arg1X) {
rv = 1;
} else if (arg0X < arg1X) {
rv = -1;
}
return rv;
}
return rv;
};
private enum Direction {
HORIZONTAL, VERTICAL
HORIZONTAL,
VERTICAL
}
static class PolygonVertex {
Point2D point;
Direction direction;
PolygonVertex(Point2D point, Direction direction) {
public PolygonVertex(Point2D point, Direction direction) {
this.direction = direction;
this.point = point;
}
@Override
public boolean equals(Object other) {
if (this == other) {
return true;
}
@@ -303,21 +316,15 @@ public class TableExtractionService {
return this.point.equals(((PolygonVertex) other).point);
}
@Override
public int hashCode() {
return this.point.hashCode();
}
@Override
public String toString() {
return String.format("%s[point=%s,direction=%s]", this.getClass()
.getName(), this.point.toString(), this.direction.toString());
return String.format("%s[point=%s,direction=%s]", this.getClass().getName(), this.point.toString(), this.direction.toString());
}
}
}
@@ -1,111 +0,0 @@
package com.iqser.red.service.redaction.v1.server.tableextraction.utils;
import java.util.ArrayDeque;
import java.util.Comparator;
import java.util.Deque;
import java.util.List;
/**
* Copied and minimal modified from PDFBox.
*/
public final class QuickSort {
private QuickSort() {
}
private static final Comparator<? extends Comparable> OBJCOMP = new Comparator<Comparable>() {
@Override
public int compare(Comparable object1, Comparable object2) {
return object1.compareTo(object2);
}
};
/**
* Sorts the given list using the given comparator.
*
* @param <T> type of the objects to be sorted.
* @param list list to be sorted
* @param cmp comparator used to compare the objects within the list
*/
public static <T> void sort(List<T> list, Comparator<? super T> cmp) {
int size = list.size();
if (size < 2) {
return;
}
quicksort(list, cmp);
}
/**
* Sorts the given list using compareTo as comparator.
*
* @param <T> type of the objects to be sorted.
* @param list list to be sorted
*/
public static <T extends Comparable> void sort(List<T> list) {
sort(list, (Comparator<T>) OBJCOMP);
}
private static <T> void quicksort(List<T> list, Comparator<? super T> cmp) {
Deque<Integer> stack = new ArrayDeque<Integer>();
stack.push(0);
stack.push(list.size());
while (!stack.isEmpty()) {
int right = stack.pop();
int left = stack.pop();
if (right - left < 2) {
continue;
}
int p = left + ((right - left) / 2);
p = partition(list, cmp, p, left, right);
stack.push(p + 1);
stack.push(right);
stack.push(left);
stack.push(p);
}
}
private static <T> int partition(List<T> list, Comparator<? super T> cmp, int p, int start, int end) {
int l = start;
int h = end - 2;
T piv = list.get(p);
swap(list, p, end - 1);
while (l < h) {
if (cmp.compare(list.get(l), piv) <= 0) {
l++;
} else if (cmp.compare(piv, list.get(h)) <= 0) {
h--;
} else {
swap(list, l, h);
}
}
int idx = h;
if (cmp.compare(list.get(h), piv) < 0) {
idx++;
}
swap(list, end - 1, idx);
return idx;
}
private static <T> void swap(List<T> list, int i, int j) {
T tmp = list.get(i);
list.set(i, list.get(j));
list.set(j, tmp);
}
}
@@ -1,6 +1,7 @@
package com.iqser.red.service.redaction.v1.server.tableextraction.utils;
import java.math.BigDecimal;
import java.util.Collections;
import java.util.Comparator;
import java.util.List;
@@ -12,29 +13,23 @@ public class Utils {
private final static float EPSILON = 0.1f;
public static boolean feq(double f1, double f2) {
return (Math.abs(f1 - f2) < EPSILON);
}
public static float round(double d, int decimalPlace) {
BigDecimal bd = BigDecimal.valueOf(d);
bd = bd.setScale(decimalPlace, BigDecimal.ROUND_HALF_UP);
return bd.floatValue();
}
public static <T> void sort(List<T> list, Comparator<? super T> comparator) {
public static <T> void sort(List<T> list, Comparator<? super T> comparator) {
try {
QuickSort.sort(list, comparator);
} catch (IllegalArgumentException e) {
// This should not happen since we use QuickSort from PDFBox
Collections.sort(list, comparator);
} catch (IllegalArgumentException e){
//TODO Figure out why this happens.
log.warn(e.getMessage());
}
}
}
}
@@ -1,16 +1,24 @@
package com.iqser.red.service.redaction.v1.server.visualization.service;
import com.iqser.red.service.redaction.v1.model.CellRectangle;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactionType;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import java.awt.Color;
import java.io.IOException;
import java.util.List;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.graphics.color.PDColor;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.SectionRectangle;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
@@ -21,30 +29,11 @@ import com.iqser.red.service.redaction.v1.server.redaction.service.DictionarySer
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.graphics.color.PDColor;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationText;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.apache.pdfbox.text.TextPosition;
import org.springframework.stereotype.Service;
import java.awt.Color;
import java.io.IOException;
import java.util.ArrayList;
import java.util.GregorianCalendar;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class AnnotationHighlightService {
@@ -52,331 +41,139 @@ public class AnnotationHighlightService {
private final DictionaryService dictionaryService;
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions, String ruleSetId) throws IOException {
Set<Integer> manualRedactionPages = getManualRedactionPages(manualRedactions);
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction) throws IOException {
for (int page = 1; page <= document.getNumberOfPages(); page++) {
PDPage pdPage = document.getPage(page - 1);
drawSectionFrames(document, classifiedDoc, flatRedaction, pdPage, page);
if (!flatRedaction) {
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
if (classifiedDoc.getEntities().get(page) != null) {
addAnnotations(pdPage, classifiedDoc, flatRedaction, manualRedactions, page, ruleSetId);
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
for (int i = 0; i <= paragraph.getPageBlocks().size() - 1; i++) {
AbstractTextContainer textBlock = paragraph.getPageBlocks().get(i);
if (textBlock.getPage() != page) {
continue;
}
if (textBlock instanceof TextBlock) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
visualizeTextBlock((TextBlock) textBlock, contentStream);
} else if (textBlock instanceof Table) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
visualizeTable((Table) textBlock, contentStream);
}
}
}
contentStream.close();
}
if (manualRedactionPages.contains(page)) {
addManualAnnotations(pdPage, classifiedDoc, manualRedactions, page, ruleSetId);
}
}
}
private Set<Integer> getManualRedactionPages(ManualRedactions manualRedactions) {
Set<Integer> manualRedactionPages = new HashSet<>();
if (manualRedactions == null) {
return manualRedactionPages;
}
manualRedactions.getEntriesToAdd().forEach(entry -> {
entry.getPositions().forEach(pos -> {
manualRedactionPages.add(pos.getPage());
});
});
return manualRedactionPages;
}
private void addAnnotations(PDPage pdPage, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions, int page, String ruleSetId) throws IOException {
List<PDAnnotation> annotations = pdPage.getAnnotations();
// Duplicates can exist due table extraction colums over multiple rows.
Set<String> processedIds = new HashSet<>();
entityLoop:
for (Entity entity : classifiedDoc.getEntities().get(page)) {
if (flatRedaction && !isRedactionType(entity, ruleSetId)) {
if (classifiedDoc.getEntities().get(page) == null) {
continue;
}
boolean requestedToRemove = false;
List<Comment> comments = null;
for (Entity entity : classifiedDoc.getEntities().get(page)) {
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
RedactionLogEntry redactionLogEntry = new RedactionLogEntry();
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(entity, ruleSetId);
if (processedIds.contains(entityPositionSequence.getId())) {
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
// TODO refactor this outer loop jump as soon as we have the time.
continue entityLoop;
} else {
processedIds.add(entityPositionSequence.getId());
}
if (flatRedaction && !isRedactionType(entity)) {
continue;
}
if (manualRedactions != null && !manualRedactions.getIdsToRemove().isEmpty()) {
for (IdRemoval manualRemoval : manualRedactions.getIdsToRemove()) {
if (manualRemoval.getId().equals(entityPositionSequence.getId())) {
comments = manualRedactions.getComments().get(manualRemoval.getId());
String manualOverrideReason = null;
if (manualRemoval.getStatus().equals(Status.APPROVED)) {
entity.setRedaction(false);
redactionLogEntry.setRedacted(false);
redactionLogEntry.setStatus(Status.APPROVED);
manualOverrideReason = entity.getRedactionReason() + ", removed by manual override";
} else if (manualRemoval.getStatus().equals(Status.REQUESTED)) {
requestedToRemove = true;
manualOverrideReason = entity.getRedactionReason() + ", requested to remove";
redactionLogEntry.setStatus(Status.REQUESTED);
} else {
redactionLogEntry.setStatus(Status.DECLINED);
}
for (TextPositionSequence textPositions : entityPositionSequence.getSequences()) {
entity.setRedactionReason(manualOverrideReason != null ? manualOverrideReason : entity.getRedactionReason());
redactionLogEntry.setReason(manualOverrideReason);
redactionLogEntry.setManual(true);
redactionLogEntry.setManualRedactionType(ManualRedactionType.REMOVE);
float height = textPositions.getTextPositions().get(0).getHeightDir() + 2;
float posXInit;
float posXEnd;
float posYInit;
float posYEnd;
if (textPositions.getTextPositions().get(0).getRotation() == 90) {
posXEnd = textPositions.getTextPositions().get(0).getYDirAdj() + 2;
posXInit = textPositions.getTextPositions().get(0).getYDirAdj() - height;
posYInit = textPositions.getTextPositions().get(0).getXDirAdj();
posYEnd = textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getXDirAdj() - height + 4;
} else {
posXInit = textPositions.getTextPositions().get(0).getXDirAdj();
posXEnd = textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getXDirAdj() + textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getWidth() + 1;
posYInit = textPositions.getTextPositions()
.get(0)
.getPageHeight() - textPositions.getTextPositions().get(0).getYDirAdj() - 2;
posYEnd = textPositions.getTextPositions()
.get(0)
.getPageHeight() - textPositions.getTextPositions()
.get(textPositions.getTextPositions().size() - 1)
.getYDirAdj() + 2;
}
Rectangle textHighlightRectangle = new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
List<PDAnnotation> annotations = pdPage.getAnnotations();
PDAnnotationTextMarkup highlight = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
highlight.constructAppearances();
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(posXInit);
annotationPosition.setLowerLeftY(posYEnd);
annotationPosition.setUpperRightX(posXEnd);
annotationPosition.setUpperRightY(posYEnd + height);
highlight.setRectangle(annotationPosition);
if (!flatRedaction && !isHint(entity)) {
highlight.setAnnotationName(entityPositionSequence.getId().toString());
highlight.setTitlePopup(entityPositionSequence.getId().toString());
highlight.setContents("\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\n" + "In Section : \"" + entity
.getHeadline() + "\"");
}
highlight.setQuadPoints(toQuadPoints(textHighlightRectangle));
PDColor color;
if (flatRedaction) {
color = new PDColor(new float[]{0, 0, 0}, PDDeviceRGB.INSTANCE);
} else {
color = new PDColor(getColor(entity), PDDeviceRGB.INSTANCE);
}
highlight.setColor(color);
annotations.add(highlight);
redactionLogEntry.getPositions().add(textHighlightRectangle);
}
redactionLogEntry.setId(entityPositionSequence.getId().toString());
}
if (CollectionUtils.isNotEmpty(entityPositionSequence.getSequences())) {
List<Rectangle> rectanglesPerLine = getRectanglesPerLine(entityPositionSequence.getSequences()
.stream()
.flatMap(seq -> seq.getTextPositions().stream())
.collect(Collectors.toList()), page);
if (manualRedactions != null) {
comments = manualRedactions.getComments().get(entityPositionSequence.getId());
}
redactionLogEntry.getPositions().addAll(rectanglesPerLine);
annotations.addAll(createAnnotation(rectanglesPerLine, entityPositionSequence.getId(), createAnnotationContent(entity), getColor(entity, ruleSetId, requestedToRemove), comments, !isHint(entity, ruleSetId)));
}
redactionLogEntry.setId(entityPositionSequence.getId());
// FIXME ids should never be null. Figure out why this happens.
if (redactionLogEntry.getId() != null) {
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
}
}
private List<Rectangle> getRectanglesPerLine(List<TextPosition> textPositions, int page) {
List<Rectangle> rectangles = new ArrayList<>();
if (textPositions.size() == 1) {
rectangles.add(new TextPositionSequence(textPositions, page).getRectangle());
} else {
float y = textPositions.get(0).getYDirAdj();
int startIndex = 0;
for (int i = 1; i < textPositions.size(); i++) {
float yDirAdj = textPositions.get(i).getYDirAdj();
if (yDirAdj != y) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, i), page).getRectangle());
y = yDirAdj;
startIndex = i;
}
}
if (startIndex != textPositions.size() - 1) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, textPositions.size()), page).getRectangle());
}
}
return rectangles;
}
private void addManualAnnotations(PDPage pdPage, Document classifiedDoc, ManualRedactions manualRedactions,
int page, String ruleSetId) throws IOException {
if (manualRedactions == null) {
return;
}
List<PDAnnotation> annotations = pdPage.getAnnotations();
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
String id = manualRedactionEntry.getId();
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(manualRedactionEntry, id, ruleSetId);
List<Rectangle> rectanglesOnPage = new ArrayList<>();
for (Rectangle rectangle : manualRedactionEntry.getPositions()) {
if (page == rectangle.getPage()) {
rectanglesOnPage.add(rectangle);
redactionLogEntry.getPositions().add(rectangle);
}
}
if (!rectanglesOnPage.isEmpty() && !approvedAndShouldBeInDictionary(manualRedactionEntry)) {
annotations.addAll(createAnnotation(rectanglesOnPage, id, createAnnotationContent(manualRedactionEntry), getColorForManualAdd(manualRedactionEntry
.getType(), ruleSetId, manualRedactionEntry.getStatus()), manualRedactions.getComments().get(id), true));
redactionLogEntry.setColor(getColor(entity));
redactionLogEntry.setReason(entity.getRedactionReason());
redactionLogEntry.setValue(entity.getWord());
redactionLogEntry.setType(entity.getType());
redactionLogEntry.setRedacted(entity.isRedaction());
redactionLogEntry.setSection(entity.getHeadline());
redactionLogEntry.setHint(isHint(entity));
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
}
private boolean approvedAndShouldBeInDictionary(ManualRedactionEntry manualRedactionEntry) {
return manualRedactionEntry.getStatus().equals(Status.APPROVED) && manualRedactionEntry.isAddToDictionary();
}
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry, String id, String ruleSetId) {
return RedactionLogEntry.builder()
.id(id)
.color(getColor(manualRedactionEntry.getType(), ruleSetId))
.reason(manualRedactionEntry.getReason())
.legalBasis(manualRedactionEntry.getLegalBasis())
.value(manualRedactionEntry.getValue())
.type(manualRedactionEntry.getType())
.redacted(true)
.isHint(false)
.section(manualRedactionEntry.getSection())
.sectionNumber(manualRedactionEntry.getSectionNumber())
.manual(true)
.status(manualRedactionEntry.getStatus())
.manualRedactionType(ManualRedactionType.ADD)
.isDictionaryEntry(false)
.build();
}
private RedactionLogEntry createRedactionLogEntry(Entity entity, String ruleSetId) {
return RedactionLogEntry.builder()
.color(getColor(entity, ruleSetId, false))
.reason(entity.getRedactionReason())
.legalBasis(entity.getLegalBasis())
.value(entity.getWord())
.type(entity.getType())
.redacted(entity.isRedaction())
.isHint(isHint(entity, ruleSetId))
.isRecommendation(isRecommendation(entity, ruleSetId))
.section(entity.getHeadline())
.sectionNumber(entity.getSectionNumber())
.matchedRule(entity.getMatchedRule())
.isDictionaryEntry(entity.isDictionaryEntry())
.textAfter(entity.getTextAfter())
.textBefore(entity.getTextBefore())
.build();
}
private List<PDAnnotation> createAnnotation(List<Rectangle> rectangles, String id, String content, float[] color,
List<Comment> comments, boolean popup) {
List<PDAnnotation> annotations = new ArrayList<>();
PDAnnotationTextMarkup annotation = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
annotation.constructAppearances();
PDRectangle pdRectangle = toPDRectangle(rectangles);
annotation.setRectangle(pdRectangle);
annotation.setQuadPoints(toQuadPoints(rectangles));
if (popup) {
annotation.setContents(content);
}
annotation.setTitlePopup(id);
annotation.setAnnotationName(id);
annotation.setColor(new PDColor(color, PDDeviceRGB.INSTANCE));
annotations.add(annotation);
if (comments != null) {
for (Comment comment : comments) {
PDAnnotationText txtAnnot = new PDAnnotationText();
txtAnnot.setAnnotationName(comment.getId());
txtAnnot.setInReplyTo(annotation); // Reference to highlight annotation
txtAnnot.setName(PDAnnotationText.NAME_COMMENT);
txtAnnot.setCreationDate(GregorianCalendar.from(comment.getDate().toZonedDateTime()));
txtAnnot.setTitlePopup(comment.getUser());
txtAnnot.setContents(comment.getText());
txtAnnot.setRectangle(pdRectangle);
annotations.add(txtAnnot);
}
}
return annotations;
}
private String createAnnotationContent(Entity entity) {
return "\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\nLegal basis:" + entity
.getLegalBasis() + "\n\nIn section: \"" + entity.getHeadline() + "\"";
}
private String createAnnotationContent(ManualRedactionEntry entry) {
return "\nManual Redaction\n\nIn Section : \"" + entry.getSection() + "\"";
}
private PDRectangle toPDRectangle(List<Rectangle> rectangles) {
float lowerLeftX = Float.MAX_VALUE;
float upperRightX = 0;
float lowerLeftY = 0;
float upperRightY = Float.MAX_VALUE;
for (Rectangle rectangle : rectangles) {
if (rectangle.getTopLeft().getX() < lowerLeftX) {
lowerLeftX = rectangle.getTopLeft().getX();
}
if (rectangle.getTopLeft().getX() + rectangle.getWidth() > upperRightX) {
upperRightX = rectangle.getTopLeft().getX() + rectangle.getWidth();
}
if (rectangle.getTopLeft().getY() + rectangle.getHeight() > lowerLeftY) {
lowerLeftY = rectangle.getTopLeft().getY() + rectangle.getHeight();
}
if (rectangle.getTopLeft().getY() < upperRightY) {
upperRightY = rectangle.getTopLeft().getY();
}
}
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(lowerLeftX);
annotationPosition.setLowerLeftY(lowerLeftY);
annotationPosition.setUpperRightX(upperRightX);
annotationPosition.setUpperRightY(upperRightY);
return annotationPosition;
}
private float[] toQuadPoints(List<Rectangle> rectangles) {
float[] quadPoints = new float[rectangles.size() * 8];
int i = 0;
for (Rectangle rectangle : rectangles) {
float[] quadPoint = toQuadPoint(rectangle);
for (int j = 0; j <= 7; j++) {
quadPoints[i + j] = quadPoint[j];
}
i += 8;
}
return quadPoints;
}
private float[] toQuadPoint(Rectangle rectangle) {
private float[] toQuadPoints(Rectangle rectangle) {
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
@@ -387,119 +184,64 @@ public class AnnotationHighlightService {
}
private boolean isRedactionType(Entity entity, String ruleSetId) {
private boolean isRedactionType(Entity entity) {
if (!entity.isRedaction()) {
return false;
}
return !isHint(entity, ruleSetId);
}
private float[] getColor(Entity entity, String ruleSetId, boolean requestedToRemove) {
if (requestedToRemove) {
return dictionaryService.getRequestRemoveColor(ruleSetId);
if (isHint(entity)) {
return false;
}
if (!entity.isRedaction() && !isHint(entity, ruleSetId)) {
return dictionaryService.getNotRedactedColor(ruleSetId);
}
return dictionaryService.getColor(entity.getType(), ruleSetId);
return true;
}
private float[] getColorForManualAdd(String type, String ruleSetId, Status status) {
private float[] getColor(Entity entity) {
if (status.equals(Status.REQUESTED)) {
return dictionaryService.getRequestAddColor(ruleSetId);
} else if (status.equals(Status.DECLINED)) {
return dictionaryService.getNotRedactedColor(ruleSetId);
}
return getColor(type, ruleSetId);
}
private float[] getColor(String type, String ruleSetId) {
return dictionaryService.getColor(type, ruleSetId);
}
private boolean isHint(Entity entity, String ruleSetId) {
return dictionaryService.isHint(entity.getType(), ruleSetId);
}
private boolean isRecommendation(Entity entity, String ruleSetId) {
return dictionaryService.isRecommendation(entity.getType(), ruleSetId);
}
private void drawSectionFrames(PDDocument document, Document classifiedDoc, boolean flatRedaction, PDPage pdPage,
int page) throws IOException {
if (flatRedaction) {
return;
if (!entity.isRedaction() && !isHint(entity)) {
return new float[]{0.627f, 0.627f, 0.627f};
}
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
for (int i = 0; i <= paragraph.getPageBlocks().size() - 1; i++) {
AbstractTextContainer textBlock = paragraph.getPageBlocks().get(i);
if (textBlock.getPage() != page) {
continue;
}
if (textBlock instanceof TextBlock) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
visualizeTextBlock((TextBlock) textBlock, contentStream);
classifiedDoc.getSectionGrid()
.getRectanglesPerPage()
.computeIfAbsent(page, (x) -> new ArrayList<>())
.add(new SectionRectangle(new Point(textBlock.getMinX(), textBlock.getMinY()), textBlock.getWidth(), textBlock
.getHeight(), i + 1, paragraph.getPageBlocks().size()));
} else if (textBlock instanceof Table) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
List<CellRectangle> cellRectangles = visualizeTable((Table) textBlock, contentStream);
classifiedDoc.getSectionGrid()
.getRectanglesPerPage()
.computeIfAbsent(page, (x) -> new ArrayList<>())
.add(new SectionRectangle(new Point(textBlock.getMinX(), textBlock.getMinY()), textBlock.getWidth(), textBlock
.getHeight(), i + 1, paragraph.getPageBlocks().size(), cellRectangles));
}
}
if (!dictionaryService.getEntryColors().containsKey(entity.getType())) {
return dictionaryService.getDefaultColor();
}
contentStream.close();
return dictionaryService.getEntryColors().get(entity.getType());
}
private boolean isHint(Entity entity) {
List<String> hintTypes = dictionaryService.getHintTypes();
if (CollectionUtils.isNotEmpty(hintTypes) && hintTypes.contains(entity.getType())) {
return true;
}
return false;
}
private void visualizeTextBlock(TextBlock textBlock, PDPageContentStream contentStream) throws IOException {
contentStream.setStrokingColor(Color.LIGHT_GRAY);
contentStream.setLineWidth(0.5f);
contentStream.addRect(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight());
contentStream.stroke();
if (textBlock.getClassification() != null) {
contentStream.beginText();
contentStream.setNonStrokingColor(Color.DARK_GRAY);
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
contentStream.newLineAtOffset(textBlock.getMinX(), textBlock.getMaxY());
contentStream.showText(textBlock.getClassification());
contentStream.endText();
}
}
private List<CellRectangle> visualizeTable(Table table, PDPageContentStream contentStream) throws IOException {
private void visualizeTable(Table table, PDPageContentStream contentStream) throws IOException {
List<CellRectangle> cellRectangles = new ArrayList<>();
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
@@ -509,22 +251,28 @@ public class AnnotationHighlightService {
contentStream.addRect((float) cell.getX(), (float) cell.getY(), (float) cell.getWidth(), (float) cell
.getHeight());
contentStream.stroke();
cellRectangles.add(new CellRectangle(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight()));
// contentStream.setStrokingColor(Color.GREEN);
// for (TextBlock textBlock : cell.getTextBlocks()) {
// contentStream.addRect(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight());
// contentStream.stroke();
// }
}
}
}
if (table.getClassification() != null) {
contentStream.beginText();
contentStream.setNonStrokingColor(Color.DARK_GRAY);
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
contentStream.newLineAtOffset(table.getMinX(), table.getMinY());
contentStream.showText(table.getClassification());
contentStream.endText();
}
return cellRectangles;
}
}
}
@@ -0,0 +1,68 @@
package com.iqser.red.service.redaction.v1.server.visualization.service;
import java.awt.image.BufferedImage;
import java.io.IOException;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.graphics.image.LosslessFactory;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.rendering.ImageType;
import org.apache.pdfbox.rendering.PDFRenderer;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class PdfFlattenService {
private final RedactionServiceSettings settings;
public PDDocument flattenPDF(PDDocument sourceDoc) throws IOException {
PDDocument destDoc = new PDDocument();
PDFRenderer pdfRenderer = new PDFRenderer(sourceDoc);
final int pageCount = sourceDoc.getDocumentCatalog().getPages().getCount();
log.info(pageCount + " page" + (pageCount == 1 ? "" : "s") + " to flatten.");
for (int i = 0; i < pageCount; i += 1) {
log.info("Flattening page " + (i + 1) + " of " + pageCount + "...");
BufferedImage img = pdfRenderer.renderImageWithDPI(i, settings.getFlattenImageDpi(), ImageType.RGB);
log.info("Image rendered in memory (" + img.getWidth() + "x" + img.getHeight() + " " + settings.getFlattenImageDpi() + "DPI). Adding to PDF...");
PDPage imagePage = new PDPage(new PDRectangle(img.getWidth(), img.getHeight()));
destDoc.addPage(imagePage);
PDImageXObject imgObj = LosslessFactory.createFromImage(destDoc, img);
PDPageContentStream imagePageContentStream = new PDPageContentStream(destDoc, imagePage);
imagePageContentStream.drawImage(imgObj, 0, 0);
log.info("Image added successfully.");
imagePageContentStream.close();
img.flush();
}
log.info("New flattened PDF created in memory.");
sourceDoc.close();
return destDoc;
}
}
@@ -1,4 +0,0 @@
server:
port: 8083
configuration-service.url: "http://localhost:8081"
@@ -1,8 +1,6 @@
info:
description: Redaction Service Server V1
configuration-service.url: "http://configuration-service-v1:8080"
server:
port: 8080
@@ -10,10 +8,15 @@ spring:
profiles:
active: kubernetes
platform.multi-tenancy:
enabled: ${multitenancy.enabled:false}
tenantFilter:
urlPatterns: /redact
urlPatternsToIgnore:
management:
endpoint:
metrics.enabled: ${monitoring.enabled:false}
prometheus.enabled: ${monitoring.enabled:false}
health.enabled: true
endpoints.web.exposure.include: prometheus, health
metrics.export.prometheus.enabled: ${monitoring.enabled:false}
@@ -2,6 +2,10 @@ spring:
application:
name: redaction-service-v1
management.endpoints:
web.base-path: /
enabled-by-default: false
management:
endpoints:
web:
base-path: /
path-mapping:
health: "health"
@@ -1,25 +1,22 @@
package com.iqser.red.service.redaction.v1.server;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.DEFINED_PORT;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
import org.junit.Before;
import org.junit.Test;
@@ -37,51 +34,30 @@ import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.UUID;
import java.util.stream.Collectors;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.RANDOM_PORT;
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
@RunWith(SpringRunner.class)
@SpringBootTest(webEnvironment = RANDOM_PORT)
@SpringBootTest(webEnvironment = DEFINED_PORT)
public class RedactionIntegrationTest {
private static final String RULES = loadFromClassPath("drools/rules.drl");
private static final String VERTEBRATE = "vertebrate";
private static final String ADDRESS = "CBI_address";
private static final String AUTHOR = "CBI_author";
private static final String SPONSOR = "CBI_sponsor";
private static final String VERTEBRATES_CODE = "vertebrate";
private static final String ADDRESS_CODE = "address";
private static final String NAME_CODE = "name";
private static final String NO_REDACTION_INDICATOR = "no_redaction_indicator";
private static final String REDACTION_INDICATOR = "redaction_indicator";
private static final String HINT_ONLY = "hint_only";
private static final String MUST_REDACT = "must_redact";
private static final String PUBLISHED_INFORMATION = "published_information";
private static final String TEST_METHOD = "test_method";
private static final String RECOMMENDATION_AUTHOR = "recommendation_CBI_author";
private static final String RECOMMENDATION_ADDRESS = "recommendation_CBI_address";
private static final String FALSE_POSITIVE = "false_positive";
private static final String PII = "PII";
@Autowired
private RedactionController redactionController;
@@ -93,13 +69,9 @@ public class RedactionIntegrationTest {
private DictionaryClient dictionaryClient;
private final Map<String, List<String>> dictionary = new HashMap<>();
private final Map<String, String> typeColorMap = new HashMap<>();
private final Map<String, float[]> typeColorMap = new HashMap<>();
private final Map<String, Boolean> hintTypeMap = new HashMap<>();
private final Map<String, Boolean> caseInSensitiveMap = new HashMap<>();
private final Map<String, Boolean> recommendationTypeMap = new HashMap<>();
private final Colors colors = new Colors();
private final static String TEST_RULESET_ID = "123";
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@@ -124,52 +96,39 @@ public class RedactionIntegrationTest {
@Before
public void stubClients() {
public void stubRulesClient() {
when(rulesClient.getVersion(TEST_RULESET_ID)).thenReturn(0L);
when(rulesClient.getRules(TEST_RULESET_ID)).thenReturn(new RulesResponse(RULES));
when(rulesClient.getVersion()).thenReturn(0L);
when(rulesClient.getRules()).thenReturn(new RulesResponse(RULES));
loadDictionaryForTest();
loadTypeForTest();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(0L);
when(dictionaryClient.getAllTypes(TEST_RULESET_ID)).thenReturn(TypeResponse.builder().types(getTypeResponse()).build());
when(dictionaryClient.getDictionaryForType(VERTEBRATE, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(VERTEBRATE));
when(dictionaryClient.getDictionaryForType(ADDRESS, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(ADDRESS));
when(dictionaryClient.getDictionaryForType(AUTHOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(AUTHOR));
when(dictionaryClient.getDictionaryForType(SPONSOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(SPONSOR));
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(REDACTION_INDICATOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(HINT_ONLY, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(HINT_ONLY));
when(dictionaryClient.getDictionaryForType(MUST_REDACT, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(MUST_REDACT));
when(dictionaryClient.getDictionaryForType(PUBLISHED_INFORMATION, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(PUBLISHED_INFORMATION));
when(dictionaryClient.getDictionaryForType(TEST_METHOD, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(TEST_METHOD));
when(dictionaryClient.getDictionaryForType(PII, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(PII));
when(dictionaryClient.getDictionaryForType(RECOMMENDATION_AUTHOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(RECOMMENDATION_AUTHOR));
when(dictionaryClient.getDictionaryForType(RECOMMENDATION_ADDRESS, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(RECOMMENDATION_ADDRESS));
when(dictionaryClient.getDictionaryForType(FALSE_POSITIVE, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(FALSE_POSITIVE));
when(dictionaryClient.getColors(TEST_RULESET_ID)).thenReturn(colors);
when(dictionaryClient.getVersion()).thenReturn(0L);
when(dictionaryClient.getAllTypes()).thenReturn(TypeResponse.builder().types(getTypeResponse()).build());
when(dictionaryClient.getDictionaryForType(VERTEBRATES_CODE)).thenReturn(getDictionaryResponse(VERTEBRATES_CODE));
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(getDictionaryResponse(ADDRESS_CODE));
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(getDictionaryResponse(NAME_CODE));
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(HINT_ONLY)).thenReturn(getDictionaryResponse(HINT_ONLY));
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor(new float[]{1f, 0.502f, 0f}));
}
private void loadDictionaryForTest() {
dictionary.computeIfAbsent(AUTHOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/CBI_author.txt")
dictionary.computeIfAbsent(NAME_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/names.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(SPONSOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/CBI_sponsor.txt")
dictionary.computeIfAbsent(VERTEBRATES_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/vertebrates.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(VERTEBRATE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/vertebrate.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(ADDRESS, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/CBI_address.txt")
dictionary.computeIfAbsent(ADDRESS_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/addresses.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
@@ -188,41 +147,6 @@ public class RedactionIntegrationTest {
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(MUST_REDACT, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/must_redact.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(PUBLISHED_INFORMATION, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/published_information.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(TEST_METHOD, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/test_method.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(PII, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/PII.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(RECOMMENDATION_AUTHOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/recommendation_CBI_author.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(RECOMMENDATION_ADDRESS, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/recommendation_CBI_address.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(FALSE_POSITIVE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/false_positive.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
}
@@ -234,70 +158,26 @@ public class RedactionIntegrationTest {
private void loadTypeForTest() {
typeColorMap.put(VERTEBRATE, "#ff85f7");
typeColorMap.put(ADDRESS, "#ffe187");
typeColorMap.put(AUTHOR, "#ffe187");
typeColorMap.put(SPONSOR, "#85ebff");
typeColorMap.put(NO_REDACTION_INDICATOR, "#be85ff");
typeColorMap.put(REDACTION_INDICATOR, "#caff85");
typeColorMap.put(HINT_ONLY, "#abc0c4");
typeColorMap.put(MUST_REDACT, "#fab4c0");
typeColorMap.put(PUBLISHED_INFORMATION, "#85ebff");
typeColorMap.put(TEST_METHOD, "#91fae8");
typeColorMap.put(PII, "#66ccff");
typeColorMap.put(RECOMMENDATION_AUTHOR, "#8df06c");
typeColorMap.put(RECOMMENDATION_ADDRESS, "#8df06c");
typeColorMap.put(FALSE_POSITIVE, "#ffffff");
typeColorMap.put(VERTEBRATES_CODE, new float[]{0, 1, 0});
typeColorMap.put(ADDRESS_CODE, new float[]{0, 1, 1});
typeColorMap.put(NAME_CODE, new float[]{1, 1, 0});
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{0.8f, 0, 0.8f});
typeColorMap.put(REDACTION_INDICATOR, new float[]{1, 0.502f, 0.1f});
typeColorMap.put(HINT_ONLY, new float[]{0.8f, 1, 0.8f});
hintTypeMap.put(VERTEBRATE, true);
hintTypeMap.put(ADDRESS, false);
hintTypeMap.put(AUTHOR, false);
hintTypeMap.put(SPONSOR, false);
hintTypeMap.put(VERTEBRATES_CODE, true);
hintTypeMap.put(ADDRESS_CODE, false);
hintTypeMap.put(NAME_CODE, false);
hintTypeMap.put(NO_REDACTION_INDICATOR, true);
hintTypeMap.put(REDACTION_INDICATOR, true);
hintTypeMap.put(HINT_ONLY, true);
hintTypeMap.put(MUST_REDACT, true);
hintTypeMap.put(PUBLISHED_INFORMATION, true);
hintTypeMap.put(TEST_METHOD, true);
hintTypeMap.put(PII, false);
hintTypeMap.put(RECOMMENDATION_AUTHOR, false);
hintTypeMap.put(RECOMMENDATION_ADDRESS, false);
hintTypeMap.put(FALSE_POSITIVE, true);
caseInSensitiveMap.put(VERTEBRATE, true);
caseInSensitiveMap.put(ADDRESS, false);
caseInSensitiveMap.put(AUTHOR, false);
caseInSensitiveMap.put(SPONSOR, false);
caseInSensitiveMap.put(VERTEBRATES_CODE, true);
caseInSensitiveMap.put(ADDRESS_CODE, false);
caseInSensitiveMap.put(NAME_CODE, false);
caseInSensitiveMap.put(NO_REDACTION_INDICATOR, true);
caseInSensitiveMap.put(REDACTION_INDICATOR, true);
caseInSensitiveMap.put(HINT_ONLY, true);
caseInSensitiveMap.put(MUST_REDACT, true);
caseInSensitiveMap.put(PUBLISHED_INFORMATION, true);
caseInSensitiveMap.put(TEST_METHOD, false);
caseInSensitiveMap.put(PII, false);
caseInSensitiveMap.put(RECOMMENDATION_AUTHOR, false);
caseInSensitiveMap.put(RECOMMENDATION_ADDRESS, false);
caseInSensitiveMap.put(FALSE_POSITIVE, false);
recommendationTypeMap.put(VERTEBRATE, false);
recommendationTypeMap.put(ADDRESS, false);
recommendationTypeMap.put(AUTHOR, false);
recommendationTypeMap.put(SPONSOR, false);
recommendationTypeMap.put(NO_REDACTION_INDICATOR, false);
recommendationTypeMap.put(REDACTION_INDICATOR, false);
recommendationTypeMap.put(HINT_ONLY, false);
recommendationTypeMap.put(MUST_REDACT, false);
recommendationTypeMap.put(PUBLISHED_INFORMATION, false);
recommendationTypeMap.put(TEST_METHOD, false);
recommendationTypeMap.put(PII, false);
recommendationTypeMap.put(RECOMMENDATION_AUTHOR, true);
recommendationTypeMap.put(RECOMMENDATION_ADDRESS, true);
recommendationTypeMap.put(FALSE_POSITIVE, false);
colors.setDefaultColor("#acfc00");
colors.setNotRedacted("#cccccc");
colors.setRequestAdd("#04b093");
colors.setRequestRemove("#04b093");
}
@@ -307,11 +187,9 @@ public class RedactionIntegrationTest {
.stream()
.map(typeColor -> TypeResult.builder()
.type(typeColor.getKey())
.ruleSetId(TEST_RULESET_ID)
.hexColor(typeColor.getValue())
.color(typeColor.getValue())
.isHint(hintTypeMap.get(typeColor.getKey()))
.isCaseInsensitive(caseInSensitiveMap.get(typeColor.getKey()))
.isRecommendation(recommendationTypeMap.get(typeColor.getKey()))
.build())
.collect(Collectors.toList());
@@ -321,168 +199,27 @@ public class RedactionIntegrationTest {
private DictionaryResponse getDictionaryResponse(String type) {
return DictionaryResponse.builder()
.hexColor(typeColorMap.get(type))
.color(typeColorMap.get(type))
.entries(dictionary.get(type))
.isHint(hintTypeMap.get(type))
.isCaseInsensitive(caseInSensitiveMap.get(type))
.isRecommendation(recommendationTypeMap.get(type))
.build();
}
@Test
public void noExceptionShouldBeThrownForAnyFiles() throws IOException {
long start = System.currentTimeMillis();
System.out.println("noExceptionShouldBeThrownForAnyFiles");
ClassLoader loader = getClass().getClassLoader();
URL url = loader.getResource("files");
File[] files = new File(url.getPath()).listFiles();
List<File> input = new ArrayList<>();
for (File file : files) {
input.addAll(getPathsRecursively(file));
}
for (File path : input) {
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(new FileInputStream(path)))
.build();
System.out.println("Redacting file : " + path.getName());
RedactionResult result = redactionController.redact(request);
Map<String, List<RedactionLogEntry>> duplicates = new HashMap<>();
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
duplicates.computeIfAbsent(entry.getId(), v -> new ArrayList<>()).add(entry);
});
duplicates.entrySet().forEach(entry -> {
assertThat(entry.getValue().size()).isEqualTo(1);
});
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
}
private List<File> getPathsRecursively(File path) {
List<File> result = new ArrayList<>();
if (path == null || path.listFiles() == null) {
return result;
}
for (File f : path.listFiles()) {
if (f.isFile()) {
result.add(f);
} else {
result.addAll(getPathsRecursively(f));
}
}
return result;
}
@Test
public void redactionTest() throws IOException {
System.out.println("redactionTest");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Applicant Producer Table.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
// result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
// if(!entry.isHint()){
// System.out.println(entry.getPositions().get(0).getPage() +":"+ entry.getTextBefore() +"--->"+ entry.getValue() + "--->" + entry.getTextAfter());
// }
// });
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
}
@Test
public void testTableRedaction() throws IOException {
System.out.println("testTableRedaction");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.redact(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
}
@Test
public void testManualRedaction() throws IOException {
System.out.println("testManualRedaction");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
ManualRedactions manualRedactions = new ManualRedactions();
String manualAddId = UUID.randomUUID().toString();
Comment comment = Comment.builder()
.date(OffsetDateTime.now())
.user("TEST_USER")
.text("This is a comment test")
.build();
manualRedactions.setIdsToRemove(Set.of(IdRemoval.builder()
.id("0836727c3508a0b2ea271da69c04cc2f")
.status(Status.REQUESTED)
.build()));
manualRedactions.getComments().put("e5be0f1d941bbb92a068e198648d06c4", List.of(comment));
manualRedactions.getComments().put("0836727c3508a0b2ea271da69c04cc2f", List.of(comment));
manualRedactions.getComments().put(manualAddId, List.of(comment));
ManualRedactionEntry manualRedactionEntry = new ManualRedactionEntry();
manualRedactionEntry.setId(manualAddId);
manualRedactionEntry.setStatus(Status.REQUESTED);
manualRedactionEntry.setType("name");
manualRedactionEntry.setValue("O'Loughlin C.K.");
manualRedactionEntry.setReason("Manual Redaction");
manualRedactionEntry.setPositions(List.of(new Rectangle(new Point(375.61096f, 241.282f), 7.648041f, 43.72262f, 1), new Rectangle(new Point(384.83517f, 241.282f), 7.648041f, 17.043358f, 1)));
manualRedactions.getEntriesToAdd().add(manualRedactionEntry);
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.manualRedactions(manualRedactions)
.build();
RedactionResult result = redactionController.redact(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
@@ -496,8 +233,7 @@ public class RedactionIntegrationTest {
@Test
public void classificationTest() throws IOException {
System.out.println("classificationTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Trinexapac/93 Trinexapac-ethyl_RAR_03_Volume_3CA_B-1_2017-03-31.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -514,8 +250,7 @@ public class RedactionIntegrationTest {
@Test
public void sectionsTest() throws IOException {
System.out.println("sectionsTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " + "Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -532,8 +267,7 @@ public class RedactionIntegrationTest {
@Test
public void htmlTablesTest() throws IOException {
System.out.println("htmlTablesTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/52 Fludioxonil_RAR_07_Volume_3CA_B-5_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -550,7 +284,6 @@ public class RedactionIntegrationTest {
@Test
public void htmlTableRotationTest() throws IOException {
System.out.println("htmlTableRotationTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
@@ -565,51 +298,6 @@ public class RedactionIntegrationTest {
}
@Test
public void phantomCellsDocumentTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Phantom Cells.pdf");
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
if (!entry.isHint()) {
assertThat(entry.getReason()).isEqualTo("Not redacted because row is not a vertebrate study");
}
});
}
@Test
public void sponsorCompanyTest() throws IOException {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/sponsor_companies.pdf");
RedactionRequest request = RedactionRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.flatRedaction(false)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.redact(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
}
private static String loadFromClassPath(String path) {
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
@@ -1,518 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.Before;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
import org.kie.api.builder.KieFileSystem;
import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.test.context.SpringBootTest;
import org.springframework.boot.test.context.TestConfiguration;
import org.springframework.boot.test.mock.mockito.MockBean;
import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Collections;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.concurrent.atomic.AtomicLong;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
@SpringBootTest
@RunWith(SpringRunner.class)
public class EntityRedactionServiceTest {
private static final String DEFAULT_RULES = loadFromClassPath("drools/rules.drl");
private static final String AUTHOR_CODE = "author";
private static final String ADDRESS_CODE = "address";
private static final String SPONSOR_CODE = "sponsor";
private static final AtomicLong DICTIONARY_VERSION = new AtomicLong();
private static final AtomicLong RULES_VERSION = new AtomicLong();
@MockBean
private DictionaryClient dictionaryClient;
@MockBean
private RulesClient rulesClient;
@Autowired
private EntityRedactionService entityRedactionService;
@Autowired
private PdfSegmentationService pdfSegmentationService;
@Autowired
private DroolsExecutionService droolsExecutionService;
private final static String TEST_RULESET_ID = "123";
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@Bean
public KieContainer kieContainer() {
KieServices kieServices = KieServices.Factory.get();
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
InputStream input = new ByteArrayInputStream(DEFAULT_RULES.getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/test/resources/drools/rules.drl", kieServices.getResources()
.newInputStreamResource(input));
KieBuilder kieBuilder = kieServices.newKieBuilder(kieFileSystem);
kieBuilder.buildAll();
KieModule kieModule = kieBuilder.getKieModule();
return kieServices.newKieContainer(kieModule.getReleaseId());
}
}
@Test
public void testNestedEntitiesRemoval() {
Set<Entity> entities = new HashSet<>();
Entity nested = new Entity("nested", "fake type", 10, 16, "fake headline", 0, false);
Entity nesting = new Entity("nesting nested", "fake type", 2, 16, "fake headline", 0, false);
entities.add(nested);
entities.add(nesting);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
assertThat(entities.size()).isEqualTo(1);
assertThat(entities).contains(nesting);
}
@Test
public void testTableRedaction() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
RedactionRequest redactionRequest = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Casey, H.W.", "O’Loughlin, C.K.", "Salamon, C.M.", "Smith, S.H."))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Toxigenics, Inc., Decatur, IL 62526, USA"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1)).hasSize(7);// 3 author cells, 1 address, 1 Y and 2 N entities
}
}
@Test
public void testNestedRedaction() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/nested_redaction.pdf");
RedactionRequest redactionRequest = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Casey, H.W.", "O’Loughlin, C.K.", "Salamon, C.M.", "Smith, S.H."))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Toxigenics, Inc., Decatur, IL 62526, USA"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1)).hasSize(7);// 3 author cells, 1 address, 1 Y and 2 N entities
}
}
@Test
public void testTrueNegativesInTable() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Cyprodinil/40 Cyprodinil - EU AIR3 - LCA Section 1" +
" Supplement - Identity of the active substance - Reference list.pdf");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_author.txt")))
.build();
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_address.txt")))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()
.entrySet()
.stream()
.noneMatch(entry -> entry.getValue().stream().anyMatch(e -> e.getMatchedRule() == 9))).isTrue();
}
pdfFileResource = new ClassPathResource("files/Compounds/27 A8637C - EU AIR3 - MCP Section 1 - Identity of " +
"the plant protection product.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()
.entrySet()
.stream()
.noneMatch(entry -> entry.getValue().stream().anyMatch(e -> e.getMatchedRule() == 9))).isTrue();
}
}
@Test
public void testFalsePositiveInWrongCell() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Row With Ambiguous Redaction.pdf");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_author.txt")))
.build();
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_address.txt")))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_sponsor.txt")))
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 9)
.count()).isEqualTo(10);
}
}
@Test
public void testApplicantInTableRedaction() throws IOException {
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" +
"rule \"6: Redact contact information if applicant is found\"\n" +
" when\n" +
" eval(section.headlineContainsWord(\"applicant\") || section.getText().contains(\"Applicant\"));\n" +
" then\n" +
" section.redactLineAfter(\"Name:\", \"address\", 6,true, \"Applicant information was found\", \"Reg" +
" (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"Address:\", \"Contact\", \"address\", 6,true, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Contact point:\", \"address\", 6,true, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Phone:\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Fax:\", \"address\", 6,true, \"Applicant information was found\", \"Reg " +
"(EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Tel.:\", \"address\", 6,true, \"Applicant information was found\", \"Reg" +
" (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Tel:\", \"address\", 6,true, \"Applicant information was found\", \"Reg " +
"(EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"E-mail:\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Email:\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Contact:\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Telephone number:\", \"address\", 6,true, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Fax number:\", \"address\", 6,true, \"Applicant information was found\"," +
" \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Telephone:\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"No:\", \"Fax\", \"address\", 6,true, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"Contact:\", \"Tel.:\", \"address\", 6,true, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" end";
when(rulesClient.getVersion(TEST_RULESET_ID)).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules(TEST_RULESET_ID)).thenReturn(new RulesResponse(tableRules));
droolsExecutionService.updateRules(TEST_RULESET_ID);
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Applicant Producer Table.pdf");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_author.txt")))
.build();
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_address.txt")))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 6)
.count()).isEqualTo(13);
}
}
@Test
public void testSponsorInCell() throws IOException {
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" + "rule \"11: Redact sponsor company\"\n" + " when\n" + " " +
"Section(searchText.toLowerCase().contains(\"batches produced at\"))\n" + " then\n" + " section" +
".redactIfPrecededBy(\"batches produced at\", \"sponsor\", 11, \"Redacted because it represents a " +
"sponsor company\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" + " end";
when(rulesClient.getVersion(TEST_RULESET_ID)).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules(TEST_RULESET_ID)).thenReturn(new RulesResponse(tableRules));
droolsExecutionService.updateRules(TEST_RULESET_ID);
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/batches_new_line.pdf");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse authorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(authorResponse);
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/CBI_sponsor.txt")))
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 11)
.count()).isEqualTo(1);
}
}
@Test
public void headerPropagation() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Header Propagation.pdf");
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Bissig R.", "Thanei P."))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(2); // two pages
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(8);
assertThat(classifiedDoc.getEntities().get(2).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(5); // 2 names, 1 address, 2 Y
}
pdfFileResource = new ClassPathResource("files/Minimal Examples/Header Propagation2.pdf");
dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Tribolet, R.", "Muir, G.", "Kühne-Thu, H.", "Close, C."))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(3);
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 8).count()).isEqualTo(9);
}
}
@Test
public void testNGuideline() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Empty Tabular Data.pdf");
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Aldershof S."))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.entries(Collections.emptyList())
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, TEST_RULESET_ID, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 8).count()).isEqualTo(6);
}
}
@Before
public void stubRedaction() {
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" +
"rule \"8: Not redacted because Vertebrate Study = N\"\n" +
" when\n" +
" Section(rowEquals(\"Vertebrate study Y/N\", \"N\") || rowEquals(\"Vertebrate study Y/N\", \"No\"))\n" +
" then\n" +
" section.redactNotCell(\"Author(s)\", 8, \"name\", false, \"Not redacted because row is not a vertebrate study\");\n" +
" section.redactNot(\"address\", 8, \"Not redacted because row is not a vertebrate study\");\n" +
" section.highlightCell(\"Vertebrate study Y/N\", 8, \"hint_only\");\n" +
" end\n" +
"rule \"9: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study\"\n" +
" when\n" +
" Section(rowEquals(\"Vertebrate study Y/N\", \"Y\") || rowEquals(\"Vertebrate study Y/N\", " +
"\"Yes\"))\n" +
" then\n" +
" section.redactCell(\"Author(s)\", 9, \"name\", false, \"Redacted because row is a vertebrate study\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redact(\"address\", 9, \"Redacted because row is a vertebrate study\", \"Reg (EC) No" +
" 1107/2009 Art. 63 (2g)\");\n" +
" section.highlightCell(\"Vertebrate study Y/N\", 9, \"must_redact\");\n" +
" end";
when(rulesClient.getVersion(TEST_RULESET_ID)).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules(TEST_RULESET_ID)).thenReturn(new RulesResponse(tableRules));
TypeResponse typeResponse = TypeResponse.builder()
.types(Arrays.asList(
TypeResult.builder().ruleSetId(TEST_RULESET_ID).type(AUTHOR_CODE).hexColor("#ffff00").build(),
TypeResult.builder().ruleSetId(TEST_RULESET_ID).type(ADDRESS_CODE).hexColor("#ff00ff").build(),
TypeResult.builder().ruleSetId(TEST_RULESET_ID).type(SPONSOR_CODE).hexColor("#00ffff").build()))
.build();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getAllTypes(TEST_RULESET_ID)).thenReturn(typeResponse);
// Default empty return to prevent NPEs
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(AUTHOR_CODE, TEST_RULESET_ID)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE, TEST_RULESET_ID)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE, TEST_RULESET_ID)).thenReturn(sponsorResponse);
Colors colors = new Colors();
colors.setDefaultColor("#acfc00");
colors.setNotRedacted("#cccccc");
colors.setRequestAdd("#04b093");
colors.setRequestRemove("#04b093");
when(dictionaryClient.getColors(TEST_RULESET_ID)).thenReturn(colors);
}
private static String loadFromClassPath(String path) {
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
StringBuilder sb = new StringBuilder();
String str;
while ((str = br.readLine()) != null) {
sb.append(str).append("\n");
}
return sb.toString();
} catch (IOException e) {
throw new IllegalArgumentException("could not load classpath resource: " + path, e);
}
}
}
@@ -1,95 +0,0 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import org.junit.Test;
import java.util.ArrayList;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
public class RegExPatternTest {
@Test
public void testEmailRegEx(){
String text = "Address: Schwarzwaldalle " +
"P.O.Box\n" +
"CH-4002 Basel\n" +
"Switzerland\n" +
"Contact: Christian Warmers\n" +
"Tel: +41 (61) 323 8044\n" +
"christian.warmers@syngenta.com";
Pattern p = Pattern.compile("\\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,4}\\b", Pattern.CASE_INSENSITIVE);
Matcher matcher = p.matcher(text);
while (matcher.find()) {
String match = matcher.group(0);
System.out.println(match);
}
}
@Test
public void testEtAlRegEx() {
String text = "To assess the potential of S-metolachlor to cause endocrine disruption (ED) a review (Charlton 2014,\n" +
"ASB2016-762) was submitted that summarises results from regulatory and open scientific literature\n" +
"studies covering in vitro and in vivo studies (level 2-5 of the OECD Conceptual Framework). According to this information metolachlor increased (1.5-fold) aromatase activity in JEG-3 cells (Laville et al.\n" +
"2006, ASB2010-14391) and induced weak anti-androgenic activity in the MDA-kb2 reporter cell line\n" +
"with a IC50 of 9.92 µM (IC50 of positive control flutamide: 0.51 µM) (Aït-Aïssa et al. 2010, ASB2015-\n" +
"9562). Data from the Tox21 high throughput screening revealed just few postive findings in assays to\n" +
"identify antagonists of the androgen receptor. An isolated result of this screening showed agonistic\n" +
"activity on the thyroid stimulating hormone receptor, while Dalton et al. (2003, ASB2018-2832)\n" +
"demonstrated that metolachlor induced CYP2B1/2 and CYP3A1/2 but did not affect T4, T3 or TSH.\n" +
"After prepubertal exposure of male Wistar rats to metolachlor (Mathias et al. 2012, ASB2016-9890) a\n" +
"statistically significant increase of serum hormone concentration was observed for testosterone (at the\n" +
"dose 50 mg/kg) as well as a statistically significant decrease in the age of preputial separation at a dose\n" +
"of 5 and 50 mg/kg. Furthermore a statistically significant increase for estradiol at a dose of 50 mg/kg\n" +
"and for FSH at a dose of 5 and 50 mg/kg and morphological alterations of the seminiferous epithelium\n" +
"were observed. Relative testicular weight was not altered. A statistically significant increase of relative\n" +
"weights was observed in long-term studies with rats (Tisdel et al. 1983, TOX9800328 ). This finding\n" +
"was attributed to lower terminal body weight. In mice a statistically significant decrease of the weight\n" +
"seminal vesicle (Tisdel et al. 1982, TOX9800327) was shown after 24 month treatment with\n" +
"metolachlor. In a mouse preimplantation embryo assay from open literature metolachlor increased the\n" +
"percentage of apoptosis significantly and reduced the mean number of cells per embryo significantly\n" +
"while the percentage of developing blastocytes was unaltered (Grennlee et al. 2004, ASB2016-9889).\n" +
"In reproduvtive toxicity studies a retarded body weight development of the pups was observed, while\n" +
"survival and normal morphological and functional development were not altered. No adverse effects\n" +
"on male fertility were seen, however important parameters to assess effects on female fertility like\n" +
"cyclicity, ovarian follicles as well as developmental landmarks in the offspring have not been investigated.";
Pattern p = Pattern.compile("([^\\s(]*?( \\w\\.?)?) et al\\.?");
Matcher matcher = p.matcher(text);
while (matcher.find()) {
String match = matcher.group(1);
System.out.println(match);
}
}
@Test
public void testAuthorSplitting(){
String word = "Porch JR, " + "Kendall TZ, " + "Krueger HO";
word.replaceAll(",", " ").replaceAll(" ", " ");
Pattern pattern = Pattern.compile("[A-ZÄÖÜ][\\wäöüéèê]{2,}( [A-ZÄÖÜ]{1,2}\\.)+");
Matcher matcher = pattern.matcher(word);
List<String> allMatches = new ArrayList<>();
while (matcher.find()) {
allMatches.add(matcher.group());
}
for(String name: allMatches) {
if(name.length() >= 3) {
System.out.println(name);
}
}
}
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import lombok.experimental.UtilityClass;
@UtilityClass
public class TextNormalizationUtilities {
/**
* Revert hyphenation due to line breaks.
* @param text Text to be processed.
* @return Text without line-break hyphenation.
*/
public static String removeHyphenLineBreaks(String text) {
return text.replaceAll("\\s(\\S+)[\\-\\u00AD]\\R|\n\r(.+ )", "\n$1$2");
}
}
@@ -10,11 +10,11 @@ public class TextNormalizationUtilitiesTest {
String test = "Without these peo-\nple, this conference would not happen";
Assertions.assertThat(TextNormalizationUtilities.removeHyphenLineBreaks(test))
.contains("people");
.contains("\npeople");
test = "Die\t\nFreiwillige\t Versicherung\t endet\t zudem\t für\t den\t ein\u00AD\nzelnen\tVersicherten\tmit\tder\tAufhebung\tdes\tVertra-\nges,\t seiner\t Unterstellung\t unter\t die\t obligatorische\t\nVersicherung\t oder\t seinem\t Ausschluss.";
Assertions.assertThat(TextNormalizationUtilities.removeHyphenLineBreaks(test))
.contains("einzelnen", "Vertrages");
.contains("\neinzelnen", "\nVertrages");
}
@@ -1,189 +0,0 @@
package com.iqser.red.service.redaction.v1.server.segmentation;
import static org.assertj.core.api.Assertions.assertThat;
import java.io.IOException;
import java.util.Collections;
import java.util.List;
import java.util.stream.Collectors;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.test.context.SpringBootTest;
import org.springframework.boot.test.mock.mockito.MockBean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.service.BlockificationService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.RulingCleaningService;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.TableExtractionService;
@SpringBootTest
@RunWith(SpringRunner.class)
public class PdfSegmentationServiceTest {
@Autowired
private PdfSegmentationService pdfSegmentationService;
@Autowired
private RulingCleaningService rulingCleaningService;
@Autowired
private TableExtractionService tableExtractionService;
@Autowired
private BlockificationService blockificationService;
@MockBean
private KieContainer kieContainer;
@Test
public void testPDFSegmentationWithComplexTable() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Spanning Cells.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table table = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(table.getColCount()).isEqualTo(6);
assertThat(table.getRowCount()).isEqualTo(13);
assertThat(table.getRows().stream().mapToInt(List::size).sum()).isEqualTo(6 * 13);
}
}
@Test
public void testTableExtraction() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Merge Table.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table firstTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(firstTable.getColCount()).isEqualTo(8);
assertThat(firstTable.getRowCount()).isEqualTo(1);
Table secondTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(1);
assertThat(secondTable.getColCount()).isEqualTo(8);
assertThat(secondTable.getRowCount()).isEqualTo(2);
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(0)
.stream()
.map(Collections::singletonList)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
.allMatch(row -> row.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList())
.equals(firstTableHeaderCells)))
.isTrue();
}
}
@Test
public void testMultiPageMetadataPropagation() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Merge Multi Page Table.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table firstTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(firstTable.getColCount()).isEqualTo(9);
assertThat(firstTable.getRowCount()).isEqualTo(5);
Table secondTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(1);
assertThat(secondTable.getColCount()).isEqualTo(9);
assertThat(secondTable.getRowCount()).isEqualTo(6);
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(firstTable.getRowCount() - 1)
.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
.allMatch(row -> row.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList())
.equals(firstTableHeaderCells)))
.isTrue();
}
}
@Test
public void testHeaderCellsForRotatedTable() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Rotated Table Headers.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table firstTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(firstTable.getColCount()).isEqualTo(8);
assertThat(firstTable.getRowCount()).isEqualTo(1);
Table secondTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(1);
assertThat(secondTable.getColCount()).isEqualTo(8);
assertThat(secondTable.getRowCount()).isEqualTo(6);
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(0)
.stream()
.map(Collections::singletonList)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
.allMatch(row -> row.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList())
.equals(firstTableHeaderCells)))
.isTrue();
}
}
}
@@ -1,9 +1,9 @@
configuration-service.url: "http://configuration-service-v1:8080"
ribbon:
ConnectTimeout: 600000
ReadTimeout: 600000
spring:
main:
allow-bean-definition-overriding: true
@@ -1,8 +0,0 @@
Crumrine
Fine Organics Limited, Middlesbrough, United Kingdom
Hunan Haili Chemical Industry Co., Ltd., Hunan, China
Monthey Syngenta Crop Protection AG, Basel, Switzerland
Syngenta Crop Protection, Monthey, Switzerland
Syngenta Monthey Switzerland
Syngenta Nantong, China
Syngenta, Switzerland
@@ -0,0 +1,796 @@
Aquatic BioSystems Inc, Fort Collins, Colorado, USA
Aquatic BioSystems, Inc., Ft. Collins, Colorado, USA.
Biological Research Laboratory (BRL), Füllinsdorf, Switzerland.
Biological Serviced Section, Alderley Park, Macclesfield, Cheshire
Harlan Laboratories Ltd., Itingen,
Jealott’s Hill, International Research Station, Bracknell,
Jealott’s Hill, International Research Station, Bracknell, RG42 6EY, United Kingdom
Jealott’s Hill, International Research Station, Bracknell, RG42 6EY, United Kingdom.
Obtained from P. Hohler, trout breeding station Zeiningen, CH-4314 Zeiningen, Switzerland
P. Hohler, Forellenzucht Zeiningen, CH-4314 Zeiningen Switzerland
P.Hohler trout breeding station Zeiningen, CH-4314 Zeiningen, Swit-zerland, and held in the test facility for more than 2 weeks
RCC Biotechnology & Animal Breeding Division, Füllinsdorf,
RCC Biotechnology & Animal Breeding Division, Füllinsdorf, Switzerland
Sequani Limited, Ledbury, United Kingdom, BFI0274
Springborn Laboratories Inc., 790 Main St., Wareham, Massachusetts, 02571-1075, USA.
Syngenta, Jealott’s Hill, International Research Station, Bracknell, RG42 6EY, United Kingdom
adama max rudong 2014 - huifeng
animal metabolism, dietary exposure, product safety, research and development, ciba-geigy limited, basle, switzerland
aquatic bio systems, inc., fort collins, colorado.
aquatic bioassay laboratory, baton rouge, louisiana
arysta lifescience north america, llc, cary, nc, usa
arysta lifescience sas, noguères, france
bayer crop-science
bayer crop-science ag
bc potter, rosedean, woodhurst, cambridgeshire, england
biospheric inc., rockville, usa
birds obtained from m & m quail farm, 4090 campbell road, gillsville, ga 30543 u.s.a
brixham environmental laboratory, astrazeneca uk limited, brixham, uk
brixham environmental laboratory, brixham, uk
brixham environmental laboratory, brixham, united kingdom
brood stock maintained at springborn laboratories
buffalo creek quail farm, po box 579, ellerbe, nc
bybrook bass hatchery, connecticut
c.i.t, miserey, france
celsius property b.v., amsterdam, netherlands
central toxicology laboratory
central toxicology laboratory (ctl), cheshire, united kingdom
central toxicology laboratory (ctl), cheshire, united kingdom, hr2464
central toxicology laboratory, alderley park, macclesfield, cheshire uk
centre international de toxicologie (c.i.t.), miserey, 27005 evreux, france
charles river
charles river (uk) limited
charles river (uk) limited, margate, kent, ct9 4lt, england.
charles river aquaria, margate, uk
charles river breeding laboratories, raleigh, nc, usa
charles river deutschland gmbh, stolzenseeweg 32-36, d-88353 kisslegg / germany
charles river france
charles river laboratories edinburgh ltd, tranent, eh33 2ne
charles river laboratories edinburgh ltd, tranent, eh33 2ne, uk
charles river laboratories france, bp 0109, f-69592 l’arbresle
charles river laboratories, edinburgh, united kingdom
charles river laboratories, edinburgh, united kingdom, 38674
charles river laboratories, portage, mi
charles river laboratories, raleigh, nc, usa
charles river uk limited, margate, kent.
charles river, 76410, saint-aubin-les-elbeuf, france
cheshire, united kingdom,
china agricultural university, no.2, yuan ming yuan west road, haidian district, beijing, 100193, p.r. china
ciba-geigy agricultural division, 410 swing road, p.o. box 18300, greensboro, north carolina 27419
ciba-geigy basel, oekotoxikologie, basel, switzerland, 953609
ciba-geigy corp. environmental health centre, farmington, ct, usa.
ciba-geigy corp., greensboro, us
ciba-geigy corp., vero beach, us
ciba-geigy corporation agricultural division, environmental health centre (ehc), 400 farmington avenue, farmington, ct 06032
ciba-geigy limited, animal production unit, basle, switzerland.
ciba-geigy limited, animal production unit, stein, switzerland.
ciba-geigy limited, animal production, 4332 stein, switzerland
ciba-geigy limited, basle, switzerland, toxicology ii. laboratories, animal facilities of toxicology ii. laboratories of residue analysis unit, agricultural division ciba-geigy limited, basle.
ciba-geigy limited, metabolism and ecology department, r&d plant protection agricultural division, basle, switzerland
ciba-geigy limited, plant protection division, ch-4002 basle, switzerland
ciba-geigy limited, research and development department, product safety, safety evaluation, basle, switzerland.
ciba-geigy limited, tierfarm, 4334 sisseln, switzerland
ciba-geigy ltd. ch-4002 basle, switzerland
ciba-geigy ltd., basel, switzerland
ciba-geigy ltd., basel, switzerland,
ciba-geigy ltd., basle, ch
ciba-geigy ltd., genetic toxicology, basel, switzerland
ciba-geigy,greensboro, united states
citoxlab france
covance laboratories inc.9200 leesburg pike, vienna, virginia 22182
covance laboratories limited, harrogate, uk
covance laboratories ltd., north yorkshire, uk.
covance laboratories, harrogate, united kingdom
cultures maintained at wildlife international ltd. laboratories
division of toxicology, institute of environmental toxicology
eba inc.
eba inc., snow camp, usa
eg&g bionomics
epl inc., research triangle
eurofins agroscience services chem sas, vergèze, france
experimental toxicology, ciba-geigy limited, 4332 stein, switzerland
fine organics limited, seal sands, middlesbrough ts2 1ub, uk
genetic toxicology, novartis crop protection ag, ch-4002 basel, switzerland
granja perrone, são bernardo do campo - sp – brazil
harlan (ad zeist, the netherlands).
harlan france, zi le malcourlet, 03800 gannat / france
harlan laboratories b.v. kreuzelweg 53 5961 nm horst / the netherlands
harlan laboratories b.v. postbus 6174 5960 ad horst / the netherlands
harlan laboratories b.v., kreuzelweg 53, 5961 nm horst / the netherlands, postbus 6174, 5960 ad horst / the netherlands
harlan laboratories ltd., itingen, switzerland, d24665
harlan sprague dawley, inc., madison, wi.
harlan uk, shaw’s farm, blackthorn, bicester, oxon, ox6 0tp
harlan winkelmann gmbh, d-33178 borchen, germany
hazleton wisconsin
hazleton wisconsin, inc.
hazleton wisconsin, inc., 3301 kinsman boulevard, madison, wisconsin
houghton springs fish farm, dorset, uk
huntingdon research centre ltd, cambridgeshire, england
huntingdon research centre ltd., huntingdon, united kingdom
huntingdon research centre ltd., p.o. box 2, huntingdon, cambridgeshire, pe18 6es, england
ibc manufacturing co., memphis, tn, usa
j. cole, the county game farms, ashford, kent, england
jealott’s hill international, bracknell, berkshire, united kingdom
jiangsu huifeng agrochemicals co. ltd.
kleintierfarm madoerin ag, ch-4414 fuellinsdorf
m & m quail farm, 4090 campbell road, gillsville, ga 30543, u.s.a.
maryland exotic birds of pasadena, maryland usa
max (rudong) chemical co ltd
morse laboratories llc, 1525 fulton avenue, sacramento, ca 95825 usa
mount lassen trout farms, california
mr j. coles, the country game farms, ashford, kent, england.
mt. lassen trout farm, rt. 5, box 36, red bluff, california 98080
nichols rabbitry inc. ; lumberton, tx
nichols rabbitry inc; lumberton, tx., us
notox b.v., hertogenbosch, netherlands
novartis crop protection ag, basel, switzerland ciba-geigy ltd., basel, switzerland
novartis crop protection ag, product portfolio management, environmental safety, ecotoxicology, ch-4002 basel, switzerland
organics limited, middlesbrough, united kingdom
osage catfish./box 222/missouri 65065/usa
osage catfisheries inc., lake road 54-56, route 4, box 1500, osage beach, mo65065, usa
p. hohler / ch-4341 zeiningen, switzerland
p. hohler, trout breeding station zeiningen, switzerland
park, nc, usa
plant protection division ciba-geigy limited basle, switzerland. genetic toxicology cibageigy limited basle, switzerland
product safety laboratories, east brunswick, new jersey 08816-3206, usa
product safety labs, east brunswick, usa
rcc - biological research laboratories, füllinsdorf, switzerland,
rcc cytotest cell research gmbh, rossdorf, germany
rcc ltd, environmental chemistry & pharmanalytics, ch-4452 itingen / switzerland
rcc ltd, itingen, switzerland
rcc ltd, laboratory animal services, wölferstrasse 4, 4414 füllinsdorf, switzerland
rcc ltd., itingen, switzerland,
rcc ltd., itingen, switzerland, b18966, t009636-06
rcc ltd., laboratory animal services, ch-4414 füllinsdorf, switzerland
rcc ltd., toxicology, wölferstrasse 4, ch-4414 füllinsdorf, switzerland
rcc ltd., zelgliweg 1, 4452 itingen, switzerland
rcc, cytotest cell research gmbh (rcc-ccr), in den leppsteinwiesen19, 64380 rossdorf, germany
research department, pharmaceuticals division, ciba-geigy corporation, 556 morris avenue, summit, new jersey 07901
ricerca, inc., ohio, usa
rodent breeding unit, alderley park, macclesfield, uk
sequani limited, bromyard road, ledbury, herefordshire, hr8 1lh, united kingdom
sequani limited, ledbury, united kingdom
sequani limited, ledbury, united kingdom,
sipcamadvan, durham, nc, usa
smithers viscient, 790 main street, wareham, ma 02571-1037 usa
smithers viscient, 790 main street, wareham, ma, usa
smithers viscient, 790 main street, wareham, massachusetts 02571 usa
smithers viscient, 790 main street, wareham, massachusetts 02571-1037, usa
source tierfarm sisseln, switzerland
southwest bio-labs, inc.401 n. 17th street, suite 11, las cruces, nm 88005 usa.
spring creek trout hatchery, lewistown, montana, usa
springborn (europe) ag, horn, switzerland
springborn laboratories inc., wareham, usa
springborn laboratories, inc. 790 main street wareham, massachusetts 02571
springborn laboratories, inc. environmental sciences division, 790 main street, wareham, 02571, usa massachusetts
springborn laboratories, inc.,
springborn laboratories, inc., health and environmental sciences, 790 main street, wareham, massachusetts, 02571-1075, usa
springborn life sciences inc.,
springborn smithers laboratories, wareham, usa
stillmeadow inc. study number 9062-05,
stillmeadow inc., sugar land, united states,
stillmeadow inc., sugarland tx, usa
stillmeadow inc., sugarland tx, usa, 8065-04 8321-03
stillmeadow, inc, 12852 park one drive, sugar land, tx 77478, us
stillmeadow, inc., 12852 park one drive, sugar land, tx 77478, usa
syngenta - jealott’s hill, bracknell, united kingdom
syngenta -jealott’s hill international research centre, uk
syngenta central toxicology laboratory, alderley park, macclesfield, cheshire, uk
syngenta crop protection, llc, greensboro, nc, usa
syngenta crop protection, llc, greensboro, usa
syngenta crop protection, monthey, switzerland
syngenta ctl, alderley park, macclesfield, cheshire, sk10 4tj, uk
syngenta – jealott’s hill international, bracknell, berkshire, united kingdom
syngenta, jealott’s hill, international research station, bracknell, rg42 6ey, united kingdom
texas animal specialties, humble, tx
texas animal specialties, humble, tx, us
toxigeneticsinc. decatur, il, us
uk. charles river
veterinary health research pty ltd, nsw, australia
vischim srl, c/o lewis & harrison, llc, washington, dc, usa
vischim srl, milano, italy
wil research laboratories, llc, 1407 george road.ashland, oh, usa
wil research laboratories, llc, ashland, oh, usa
wil research laboratories, llc, ashland, oh, usa,
wil research, 1407 george road, ashland, oh, 44805-8946, usa
wil research, llc, 1407 george road, ashland, oh 44805-8946, usa
wildlife international a division of eag inc. 8598 commerce drive easton, md 21601
wildlife international ltd. cultures, 8651 brooks drive, easton, maryland 21601
wildlife international ltd., 8598 commerce drive, easton, maryland 21601, usa
wildlife international ltd., 8598 commerce drive, maryland 21601, usa
wildlife international ltd., easton md, usa
wildlife international ltd., easton, maryland 21601, usa
wildlife international ltd., easton, usa
wildlife international ltd., maryland, us
wildlife international ltd., maryland, usa
wildlife international, 8598 commerce drive, easton, md 21601 usa
wildlife international, a division of eag inc., 8598 commerce drive, easton, md 21601 usa
wise d.r. & wise r.e., monkfield, bourn, cambridgeshire, england
zeneca agrochemicals, jealott’s hill, united kingdom
zentralinstitut fur versuchstierzucht gmbh, hannover, germany",
Syngenta Ltd., Jealott’s Hill International Research Centre, Bracknell, Berkshire, RG42 6EY, UK.
Sequani Limited, Bromyard Road, Ledbury, Herefordshire, HR8 1LH, UK.
Harlan Cytotest Cell Research GmbH (Harlan CCR), In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
Harlan Laboratories Ltd, Itingen, Switzerland.
Bioassay Labor fuer biologische Analytik GmbH INF 515, 69120 Heidelberg, Germany
Syngenta Crop Protection Ltd.
Syngenta, Jealott’s Hill, Bracknell, United Kingdom
Charles River Laboratories, Preclinical Services, Tranent (PCS-EDI) Edinburgh, EH33 2NE, UK
CXR Biosciences, 2, James Lindsay Place, Dundee Technopole, Dundee, DD1 5JJ, Scotland, UK
CiToxLAB Hungary Ltd. H-8200 Veszprém, Szabadságpuszta Hungary
Charles River, Tranent, Edinburgh, EH33 2NE, UK
Charles River Laboratories Edinburgh Ltd., Tranent, Edinburgh, EH33 2NE, UK
BASF SE; Ludwigshafen/Rhein; Germany Fed.Rep.
Leatherhead Food Research (LFR), Molecular Sciences Department, Randalls Road, Leatherhead, Surrey, KT22 7RY, UK
Syngenta, Jealott’s Hill, Bracknell, United Kingdom
Department of Veterinary & Biomedical Sciences, 101 Life Sciences Building, Penn State University, University Park, PA 16802, USA
CiToxLAB Hungary Ltd., H-8200 Veszprém, Szabadságpuszta, Hungary
SafePharm Laboratories Ltd, Shardlow Business Park, Shardlow, Derbyshire, UK
Harlan Laboratories Ltd., Zelgliweg 1, 4452 Itingen, Switzerland
RCC, Cytotest Cell Research GmbH (RCC-CCR), In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
Harlan, Cytotest Cell Research GmbH (Harlan CCR), In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
Harlan Laboratories Ltd. Zelgliweg 1, CH-4452 Itingen / Switzerland
Quotient Bioresearch (Rushden) Ltd., Pegasus Way, Crown Business Park, Rushden, Northamptonshire, NN10 6ER, UK
Charles River Laboratories Edinburgh, Ltd., Elphinstone Research Centre, Tranent, East Lothian, EH33 2NE, United Kingdom
CiToxLAB Hungary Ltd. H-8200 Veszprém, Szabadságpuszta, Hungary
Harlan Cytotest Cell Research GmbH, In den Leppsteinswiesen 19, 64380 Rossdorf Germany
Charles River, Tranent, Edinburgh, EH32 2NE, UK
Charles River Laboratories Edinburgh Ltd, Tranent, Edinburgh, EH33 2NE, UK
Harlan Cytotest Cell Research GmbH, (Harlan CCR), In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
Charles River UK Limited, Margate, Kent, UK
RCC Ltd., Biotechnology & Animal Breeding Division, 4414 Fuellinsdorf, Switzerland
Charles River (UK) Ltd., Margate, Kent, CT9 4LT, England
Charles River Ltd., Margate, Kent, United Kingdom
Charles River UK Ltd, Manston Road, Margate, Kent CT9 4LT, England, UK
Syngenta Crop Protection, Toxicology, 4332 Stein, Switzerland
Safepharm Laboratories Limited, Shardlow Business Park, Shardlow, Derbyshire, DE72 2GD, United Kingdom
Sequani Ltd, Bromyard Road, Ledbury, Herefordshire, HR8 1LH, United Kingdom
Central Toxicology Laboratory, Alderley Park, Macclesfield, Cheshire, SK10 4TJ, UK
Charles River UK
Department of Veterinary & Biomedical Sciences, Penn State University
Syngenta Ltd. Jealott’s Hill International Research, Bracknell, Berks RG42 6EY
Charles River Laboratories, Research Models and Services Germany GmbH; Sandhofer Weg 7, 97633 Sulzfeld, Germany
Novartis Crop Protection AG, Toxicology, 4332 Stein, Switzerland
BRL Biological Research Laboratories Ltd., Wölferstrasse 4, 4414 Füllinsdorf, Switzerland
B&K Universal Ltd, Grimston, Aldbrough, Hull, HU11 4QE, East Yorkshire, UK
B&K Universal Ltd, Grimston, Aldborough, Hull, UK
Nunc GmbH & Co. KG, 65203 Wiesbaden, Germany
Fluka, 89203 Neu-Ulm, Germany
MERCK, 64293 Darmstadt, Germany
Charles River Laboratories, Research Models and Services Germany GmbH; Sandhofer Weg 7, 97633 Sulzfeld, Germany
Animal Production, Novartis Pharma AG, 4332 Stein, Switzerland
RCC Ltd., Biotechnology & Animal Breeding Division, 4414 Fuellinsdorf, Switzerland.
SYSTAT Software, Inc., 501, Canal Boulevard, Suite C, Richmond, CA 94804, USA
Safepharm Laboratories Limited, Shardlow Business Park, Shardlow, Derbyshire, DE72 2GD, United Kingdom
Charles River (UK) Limited, Margate, Kent, CT9 4LT, England
CXR Biosciences, 2 James Lindsay Place, Dundee Technopole, Dundee, DD1 5JJ, Scotland, UK
Granja Perrone, São Bernardo do Campo - SP – Brazil
Harlan Sprague-Dawley, Inc. Houston/Texas
P. Hohler, trout breeding station Zeiningen, 4314 Zeiningen, Switzerland
Spring Creek trout hatchery, Lewistown, Montana, USA
Springborn laboratories culture facility
Springborn culture
University of Texas
Institute for Plant Physiology, University of Göttingen, 37073 Göttingen, Germany
Bayer CropScience AG, 40789 Monheim, Germany
Koppert B. V. Berkel en Rodenrijs, Nederland
Bio-Test Labor GmbH, Sagerheide, Germany
Ciba-Geigy
Ciba-Geigy Ltd.
Harlan Laboratories Ltd., Itingen, Switzerland, D24643
Springborn Laboratories Inc., Wareham, USA
Springborn Laboratories (Europe) AG
Syngenta Eurofins - GAB, Niefern Öschelbronn, Germany
Syngenta Eurofins Agroscience Services EcoChem GmbH, N-Osch., Germany
Novartis Crop Protection AG, Basel, CH
Springborn (Europe) AG, Horn, Switzerland
Springborn Smithers Laboratories (Europe) AG, Horn, Switzerland
Syngenta Crop Protection AG, Basel, Switzerland
GAB Biotechnologie GmbH, Niefern, Germany
BioChem Agrar, Gerichshain, Germany
AgroChemex Ltd, Manningtree, United Kingdom
Ciba-Geigy Ltd., Basel, Switzerland
Ciba-Geigy Muenchwilen AG, Muenchwilen, Switzerland
Novartis Crop Protection Münchwilen AG, Münchwilen, Switzerland
Novartis Crop Protection AG, Basel, Switzerland
Ciba-Geigy Muenchwilen AG, Muenchwilen, Switzerland
Charles River Laboratories, Research Models and Services Germany GmbH; Sandhofer Weg 7, 97633 Sulzfeld, Germany
Alderley Park
Alderley Park Swiss
Stillmeadow, Inc., 12852 Park One Drive, Sugar Land, TX 77478, USA
Texas Animal Specialties, Humble, TX
Nichols Rabbitry Inc. ; Lumberton, TX
Charles River Laboratories., Wilmington, MA
Charles River Laboratories Edinburgh Ltd., Elphinstone Research Centre, Tranent, East Lothian, EH33 2NE
Syngenta Crop Protection, Monthey, Switzerland
Syngenta Crop Protection, Münchwilen, Switzerland
Fine Organics Limited, Middlesbrough, United Kingdom
Fine Organics Limited, Seal Sands, Middlesbrough TS2 1UB, UK
Syngenta Crop Protection, Inc., Greensboro, USA
Syngenta Technology & Projects, Huddersfield, United Kingdom
Syngenta Biosciences Pvt. Ltd., Ilhas Goa, India
Syngenta - Process Hazards Section, Huddersfield, United Kingdom
Syngenta Walloon Agricultural Research Centre, Gembloux, Belgium , 21764
Syngenta Crop Protection, Münchwilen, Switzerland, 300052719
Syngenta Crop Protection Münchwilen AG, Münchwilen, Switzerland, 109747
Syngenta Crop Protection, Münchwilen, Switzerland, 300073294
Syngenta - Jealott’s Hill, Bracknell, United Kingdom RCC Ltd., Itingen, Switzerland, B18977, T003446-06
Syngenta - Jealott’s Hill, Bracknell, United Kingdom RCC Ltd., Itingen, Switzerland, B18966, T009636-06
RCC Cytotest Cell Research GmbH, Rossdorf, Germany, RCC 107662
Syngenta Syngenta - Jealott’s Hill, Bracknell, United Kingdom,
RCC Cytotest Cell Research GmbH, Rossdorf, Germany
WIL Research Laboratories, LLC, Ashland, OH, USA
Charles River Laboratories, Edinburgh, United Kingdom, 36955
Syngenta Crop Protection AG, Basel, Switzerland Stillmeadow Inc., Sugarland TX, USA
Novartis Crop Protection Inc., Greensboro, USA
Syngenta - Jealott’s Hill, Bracknell, United Kingdom
Eurofins - ADME Bioanalyses, Vergeze, France
BioChem GmbH, Cunnersdorf, Germany
Syngenta Syngenta Crop Protection, LLC, Greensboro, NC, USA
Syngenta Eurofins Agroscience Services Chem SAS, Vergèze, France
Syngenta Innovative Environmental Services, Witterswil, Switzerland
Ricerca Biosciences, LLC, Concord, OH, USA
Dr Knoell Consult GmbH, Mannheim, Germany
RCC Umweltchemie GmbH & Co. KG, Rossdorf, Germany
JSC International Ltd., Harrogate, United Kingdom
Wildlife International Ltd., Easton, Maryland 21601, USA
Syngenta Crop Protection, LLC, Greensboro, NC, USA
Novartis - Greensboro, Greensboro, USA
Smithers Viscient, 790 Main Street, Wareham, MA, USA
Syngenta Cambridge Environmental Assessments, United Kingdom
Ciba-Geigy Basel, Oekotoxikologie, Basel, Switzerland
RCC Ltd., Itingen, Switzerland
IBACON GmbH, Rossdorf, Germany
Envigo Research Limited, Shardlow, UK
Syngenta Crop Protection Münchwilen AG, Münchwilen, Switzerland
Ciba-Geigy Münchwilen AG, Münchwilen, Switzerland
Huntingdon Research Centre Ltd., Huntingdon, United Kingdom
Syngenta Technology & Projects, Huddersfield, United Kingdom
Harlan Laboratories Ltd., Shardlow, Derbyshire, UK
Dr. Specht & Partner Chem. Laboratorien GmbH, Hamburg, Germany
Institut Fresenius, Taunusstein, Germany
Syngenta - Jealott’s Hill International, Bracknell, Berkshire, United Kingdom
Ciba-Geigy Corp., Greensboro, USA
CIP Chemisches Institut Pforzheim GmbH, Pforzheim, Germany
Charles River Laboratories Edinburgh Ltd, Tranent, EH33 2NE, UK
Hazleton Laboratories, Madison, USA
Eurofins BioPharma, Planegg, Germany, 150556
Syngenta Environ. Health Center, Farmington, USA
Centre International de Toxicologie C.I.T., Evreux, France
Toxalim, Research Centre in Food Toxicology, F- 31027 Toulouse, France
Harlan Laboratories Ltd., Shardlow, Derbyshire, UK
CRS GmbH GmbH, In den Leppsteinswies en 19, 64380 Rossdorf Germany
Environ. Health Center, Farmington, USA
Ciba-Geigy Corp., Summit, USA
Ciba-Geigy Basel, Genetische Toxikologie, Basel, Switzerland
Ciba-Geigy Ltd., Stein, Switzerland
Novartis Crop Protection AG, Stein, Switzerland
Central Toxicology Laboratory (CTL), Cheshire, United Kingdom
Sequani Limited, Bromyard Road, Ledbury, Herefordshire, HR8 1LH, United Kingdom
Brixham Environmental Laboratory, Brixham, United Kingdom
Springborn Smithers Laboratories, Horn, Switzerland
Huntingdon Research Centre, Cambridgeshire, United Kingdom
Mambo-Tox Ltd., Southampton, United Kingdom
MITOX Consultants, Amsterdam, Netherlands
Charles River Aquaria, Margate, UK
Brixham Environmental Laboratory, Brixham, UK
O.Keller, Mörschwil, CH
Huntingdon Life Sciences Ltd., Huntingdon, UK
BTL Bio-Test Labor GmbH, Sagerheide, Germany
Mambo-Tox Ltd., Southampton, UK
Mambo-Tox Ltd. 2 Venture Road, University Science Park, Southampton SO16 7NP, United Kingdom
BioChem GmbH, Germany
PK Nützlingszuchten, Welzheim, Germany
BioChem agrar, Germany
Sautter & Stepper, Ammerbuch, Germany
Koppert, The Netherlands
Kraut & Rubeen (Doris Haber), Zeilstraße 40, 64367 Mühltal-Frankenhausen, Germany
Springborn Laboratories (Europe) AG, Seestrasse 21, CH-9326 Horn, Switzerland
Biologische Bundesanstalt (BBA), Braunschweig, Germany
Institut für Biologische Analytik und Consulting, IBACON GmbH, Arheilger Weg 17, 64380 Rossdorf, Germany
Abandoned vineyard, Northern Italy
Syngenta Limited, Cheshire, United Kingdom
Agrochemex, Lawford, United Kingdom
Staphyt, Inchy en Artois, France
Dermal Technology Laboratory Ltd., Staffordshire, UK
Ciba Agriculture, Whittlesford, United Kingdom
Bayer Crop Science AG, Monheim, Germany
tier3 solutions GmbH, Leichlingen, Germany
Mambo-Tox. Ltd., Southampton, United Kingdom
Syngenta Crop Protection AG, Stein, Switzerland
Stillmeadow Inc, Sugar Land, TX 77478, US
Texas Animal Specialities, Humble, TX, US
CiToxLAB, 8200 Veszprem, Szabadsagpuszta, Hungary
Syngenta Ltd, Jealott's Hill International Research Centre, Bracknell, Berkshire, RG42 6EY, United Kingdom
Stillmeadow, Inc, 12852 Park One Drive, Sugar Land
Syngenta Central Toxicology Laboratory, Alderley Park, Macclesfield, Cheshire, UK
Syngenta Limited, Alderley Park, Macclesfield, Cheshire, SK10 4TJ
Nichols Rabbitry Inc; Lumberton, TX., US
AgroChemex International Ltd, Aldhams Farm Research Station, Lawford, Essex, UK
Ciba Agriculture, Whittlesford, Cambridge, UK
Ricerca Inc., Department of Residue Analysis, Painesville OH, USA
Staphyt, 23 rue de Moeuvres, F-62860 Inchy en Artois, France
Dermal Technology Laboratory Ltd., Med IC4, Keele University Science and Business Park, Keele, Staffordshire, ST5 5NL, United Kingdom
Tier3 solutions GmbH, Kolberger Strasse 61-63 51381 Leverkusen, Germany
RCC Ltd, Environmental Chemistry & Pharmanalytics, CH-4452 Itingen / Switzerland
GAB Biotechnologie GmbH & IFU Umweltanalytik GmbH, Niefern-Öschelbronn, Germany
Biochem agrar, Germany
Bienenfarm Kern GmbH, Am Rehbacher Anger 10, 04249 Leipzig, Germany
Joaquin Cordero, Paseo de Colón No. 19, 41370 Cazalla (Sevilla), Spain
Mambo-Tox Ltd, Southampton, UK
GAB Biotechnologie GmbH & IFU Umweltanalytik GmbH, Niefern-Öschelbronn, Germany
Innovative Environmental Services (IES), Benkenstrasse 260, 4108 Witterswil, Switzerland
BioChem agrar GmbH, Kupferstraße 6, 04827 Gerichshain, Germany
RCC - Biological Research Laboratories, Füllinsdorf, Switzerland, 859442
RCC Ltd., Toxicology, Wölferstrasse 4, CH-4414 Füllinsdorf, Switzerland
RCC Ltd., Laboratory Animal Services, CH-4414 Füllinsdorf, Switzerland
Charles River Laboratories France, BP 0109, F-69592 L’Arbresle
Charles River Deutschland GmbH, Stolzenseeweg 32-36, D-88353 Kisslegg / Germany
Syngenta CTL, Alderley Park, Macclesfield, Cheshire, SK10 4TJ, UK
Harlan UK, Shaw’s Farm, Blackthorn, Bicester, Oxon, OX6 0TP
Syngenta Central Toxicology Laboratory, UK
RCC Ltd., Toxicology, Wölferstrasse 4, CH- 4414 Füllinsdorf, Switzerland
RCC Ltd, Itingen, Switzerland
P. Hohler, trout breeding station Zeiningen, Switzerland
SAG, Institute for Plant Physiology, University of Göttingen, Germany
GAB Biotechnologie GmbH, Niefern-Öschelbronn, Germany
Beekeeper Mr. Berthold Nengel, Brückenstraße 12, 56348 Dahlheim, Germany
Syngenta Crop Protection, Münchwilen, Switzerland, CHMU140561
Syngenta Crop Protection, Münchwilen, Switzerland
Sequani Limited, Ledbury, United Kingdom, BFI0516
PTRL Europe, Ulm, Germany
SGS Institut Fresenius GmbH, Taunusstein, Germany
CEM Analytical Services Ltd (CEMAS) - Berkshire, UK
PTRL Europe, Ulm, Germany
Sequani Limited, Ledbury, United Kingdom
SGS Institut Fresenius GmbH, Taunusstein, Germany
CEM Analytical Services, UK
Eurofins Agroscience Services Chem SAS, Vergà ̈ze, France
Novartis Services AG, Basel, Switzerland
BSL Bioservice Scientific, Planegg, Germany
Envigo CRS GmbH, Rossdorf, Germany
Envigo CRS GmbH, In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
BASF Ltd., Ludwigshafen, Germany
ALS Laboratory Group, Edmonton, Alberta, Canada
Syngenta Crop Protection, Inc., Greensboro, USA
ADME - Bioanalyses, Vergeze, France
Battelle UK Ltd., Ongar, United Kingdom
SGS Institut Fresenius GmbH
Novartis Agro GmbH, Frankfurt, Germany
Supervision & Test Center Pesticide Safety Evaluation, China
T. R. Wilbury Laboratories, Inc., Marblehead, MA, USA
CEMAS, North Ascot, United Kingdom
EAG Laboratories PTRL Europe GmbH, Germany
Syngenta Crop Protection Inc., USA
Syngenta Crop Protection Inc., 410 Swing Road, Greensboro, NC 27409, USA
Huntingdon Research Centre Ltd., UK
Huntingdon Research Centre Ltd., England
T.R. Wilbury Laboratories, Inc., USA
Wildlife International Ltd., USA
RCC Ltd, Switzerland
RCC Ltd. Environmental Chemistry & Pharmanalytics Division CH-4452 Itingen/Switzerland
Harlan Laboratories Ltd., Switzerland
CIBA-GEIGY Ltd., Switzerland
Syngenta Crop Protection AG, Basel , Switzerland
Syngenta Crop Protection LLC, Greensboro, USA
PTRL Europe GmbH, Helmholtzstr. 22, Science Park, Ulm, Germany
PTRL Europe GmbH, Germany
CEM Analytical Services Ltd (CEMAS), Imperial House, Oaklands Business Centre, Oaklands Park, Wokingham, Berkshire, RG41 2FD UK
SGS INSTITUT FRESENIUS GmbH
Syngenta Ltd, Jealott’s Hill International Research Centre, Bracknell, Berkshire, RG42 6EY, UK
Fraunhofer Institute for Molecular Biology and Applied Ecology, IME, Auf dem Aberg 1, 57392 Schmallenberg, Germany
Eurofins Agroscience Services Chem SAS, 75B, Avenue du Pascalet, 30310 Vergèze, France
Innovative Environmental Services (IES) Ltd, Benkenstrasse 260, 4108 Witterswil, Switzerland
BSL Bioservice, Scientific Laboratories GmbH, Behringstrasse 6/8, 82152 Planegg, Germany
RCC Ltd, Zelgliweg 1, CH-4452 Itingen, Switzerland
RCC Ltd, Laboratory Animal Services, CH-4414 Fuellinsdorf
Harlan Cytotest Cell Research GmbH, In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
Ciba-Geigy Limited, Basel, Switzerland
BASF SE, Experimental Toxicology and Ecology, 67056 Ludwigshafen, Germany
Ciba-Geigy Limited, Animal Production, 4332 Stein, Switzerland
RCC Ltd. Biotechnology & Animal Breeding Division, 4414 Füllinsdorf, Switzerland
Syngenta Ltd. Jealott’s Hill International Research Centre, Bracknell, Berks RG42 6EY
WIL Research Laboratories, LLC, 1407 George Road, Ashland, Ohio 44805-8946, USA
Charles River Laboratories Inc., Kingston, New York, USA
Syngenta, Jealott’s Hill International Research Centre, Bracknell, United Kingdom
Battelle UK Ltd
D.R. & R.E. Wise, Monkfield, Bourn, Cambridgeshire, England
Wildlife International. 8598 Commerce Drive, Easton, MD 21601 USA
Maryland Exotic Birds of Pasadena, MD 21122
Mr D. R. Wise, Monkfield, Bourn, Cambridgeshire, England
Cambridge Environmental Assessments, Battlegate Road, Boxworth, Cambridgeshire, CB23 4NN, UK
J. Coles, The County Game Farms, Ashford, Kent, England
Osage Catfisheries, MO 65 065, USA
Supervision and Test Center for Pesticide Safety Evaluation and Quality Control, 600 Shenliao Road, Tiexi District, Shengyang 110141, Liaoning Province, P.R. China
Syngenta, Jealott’s Hill International Research Centre, Bracknell, Berkshire, RG42 6EY
Harlan Laboratories Ltd., 4452 Itingen, Switzerland
Ciba-Geigy Ltd., Product Safety, Ecotoxicology, CH-4002 Basel, Switzerland
P. Hohler, CH-4314 Zeiningen
Cambridge Environmental Assessments, Battlegate Road, Boxworth, Cambridgeshire, CB23 4NN/UK
Wildlife International, A Division of EAG Inc. 8598 Commerce Drive Easton, MD 21601 USA
Novartis Crop Protection AG, Kanton Aargau, Switzerland.
RCC Ltd, CH-4452 Itingen, Switzerland
CEMAS, North Ascot, Berkshire, UK
Wilbury Laboratories Inc, 40 Doaks Lane, Marblehead, Massachusetts
P. Cummins Oyster Company, Pasadena, Maryland
Harlan Laboratories Ltd, Zelgliweg 1, 4452 Itingen/Switzerland
PK Nützlingszuchten, D-73642 Welzheim, Germany
Institut für Biologische Analytik und Consulting IBACON GmbH Arheilger Weg 17, 64380 Rossdorf, Germany
ABC Laboratories Inc., Analytical Chemistry and Field Services, 7200 E. ABC Lane, Columbia, Missouri
Ciba-Geigy Corporation, Farmington, CT, USA
Syngenta Ltd. Jealott’s Hill, Bracknell, United Kingdom
Eurofins Agroscience Services EcoChem GmbH, N- Osch., Germany
Ciba-Geigy Limited, Animal Production Unit, Basle, Switzerland
Ciba-Geigy Limited, Basle, Switzerland
Charles River Laboratories, Raleigh, NC, USA
Charles River (UK) Limited
Harlan Sprague Dawley, Inc., Madison, WI
CIBA-GEIGY Limited, Animal Production, 4332 Stein, Switzerland
CIBA-GEIGY Limited, 4332 Stein, Switzerland
Kleintierfarm Madoerin AG, CH-4414 Fuellinsdorf
CIBA-GEIGY Limited, Tierfarm, 4334 Sisseln, Switzerland
Animal production, CIBA-GEIGY Limited, 4332 Stain/Switzerland
Environmental Health Centre (EHC), 400 Farmington Avenue, Farmington, CT 06032
Charles River Laboratories, Kingston, NY
Harlan (Ad Zeist, the Netherlands)
Animal Production CIBA-GEIGY Limited 4332 Stein / Switzerland
Tierfarm, Sisseln, Switzerland
Zen-tralinstitut fur Versuchstier-zucht GmbH, Hannover, Germany
Charles River Laboratories, Portage, MI
CIBA-GEIGY Limited, Basel, Switzerland
Novartis Crop Protection AG, CH-4002 Basel, Switzerland
RCC Ltd., Biotechnology and animal breeding division, Fullinsdorf, Switzerland
Tierfarm Sisseln, Switzerland
Charles River Breeding Laboratories, Raleigh, NC, USA
Ciba-Geigy Corporation, Plant Protection Division, Environmental Health Center, 400 Farmington Avenue, Farmington, Connecticut 06032, USA
Charles River Breeding Laboratories, Inc., Raleigh, North Carolina USA
Charles River, 76410, Saint-Aubin-les-Elbeuf, France
Charles River Laboratories, Inc., Raleigh, NC, USA
WIL Research Laboratories, LLC, 1407 George Road, Ashland, OH 44805-8946 USA
RCC Ltd., Biotechnology & Animal Breeding Division, 4414 Fȕllinsdorf, Switzerland
Alderley Park, Macclesfield, Cheshire UK
Rodent Breeding Unit, Alderley Park, Macclesfield, UK
Harlan Winkelmann GmbH, D-33178 Borchen, Germany
WIL Research Laboratories, LLC, 1407 George Road.Ashland, OH 44805-8946 USA
Centre d’Elevage Charles River
CIBA-GEIGY Limited, Experimental Toxicology, 4332 Stein/Switzerland
Centre International de Toxicologie (C.I.T.), Miserey, 27005 Evreux, France
Centre Internationale de Toxicologie, Miserey, 27005 Evreux, France
CIBA-GEIGY Limited, Basle, Switzerland
Harlan Laboratories Ltd, Shardlow Business Park, Shardlow, Derbyshire, DE72 2GD, UK
Envigo CRS GmbH GmbH, In den Leppsteinswiesen 19, 64380 Rossdorf Germany
Ciba-Geigy Ltd., Genetic Toxicology, Basel, Switzerland
Toxalim, Research Centre in Food Toxicology, F-31027 Toulouse, France
Ciba-Geigy Corp, Plant Protection Division, Environmental Health Center, 400 Farmington Avenue, Farmington, Connecticut 06032, USA
Charles River France
Charles River US
WIL Research, LLC, 1407 George Road, Ashland, OH 44805-8946, USA
Novartis Crop Protection AG, Toxicology, 4332 Stein Switzerland
Syngenta Crop Protection, Health Assessment 2 Stein, 4332 Stein, Switzerland
RCC Ltd. Biotechnology and Animal Breeding Division, 4414 Füllinsdorf, Switzerland
Genetic Toxicology, Novartis Crop Protection AG, CH-40002 Basel, Switzerland
RCC - Cytotest Cell Research GmbH In den Leppsteinswiesen 19, D- 64380 Roβdorf, Germany
RCC - Cytotest Cell Research GmbH, In den Leppsteinswiesen 19, D-64380 Rofldorf, Germany
Ciba-Geigy Limited, Animal production, 4332 Stein, Switzerland
RCC Ltd., Zelgliweg 1, 4452 Itingen, Switzerland
RCC Ltd, Laboratory Animal Services, Wölferstrasse 4, 4414 Füllinsdorf, Switzerland
RCC Ltd, Laboratory Animal Services, 4414 Füllinsdorf, Switzerland
CIBA-GEIGY Limited, Basle, Switzerland
RCC Cytotest Cell Research GmbH (RCC-CCR), In den Leppsteinswiesen 19, 64380 Rossdorf, Germany
RCC Cytotest cell Research GmbH, In den Leppsteinwiesen 19, Rossdorf, Germany
Centre International de Toxicologie (CIT)
C iba-Geigy
Ciba-Geigy, Greensboro, North Carolina
Ciba-Geigy Corp., Greensboro, United States
Ciba-Geigy Vero Beach Research Center, Florida, USA
Ciba-Geigy Corporation, Environ. Health Center, Farmington, United States
Ciba-Geigy GmbH, Frankfurt a.Main, Germany
Ciba-Geigy Corp., Greensboro, United States
Wise D.R. & Wise R.E., Monkfield, Bourn, Cambridgeshire, England
Mr J. Coles, The Country Game Farms, Ashford, Kent, England
Maryland Exotic Birds of Pasadena, Maryland USA
J. Cole, The County Game Farms, Ashford, Kent, England
BC Potter, Rosedean, Woodhurst, Cambridgeshire, England
M & M Quail Farm, 4090 Campbell Road, Gillsville, GA 30543, U.S.A
Wildlife International A Division of EAG Inc. 8598 Commerce Drive Easton, MD 21601 USA
M & M Quail Farm, 4090 Campbell Road, Gillsville, GA 30543 U.S.A
China Agricultural University, No.2, Yuan Ming Yuan West Road, Haidian District, Beijing, 100193, P.R. China
Mt. Lassen Trout Farm, Rt. 5, Box 36, Red Bluff, California 98080
Bybrook Bass Hatchery, Connecticut
CIBA-GEIGY Ltd. CH-4002 Basle, Switzerland
Wildlife International Ltd. Cultures, 8651 Brooks Drive, Easton, Maryland 21601
Aquatic bioassay laboratory, Baton Rouge, Louisiana
P. Hohler/ CH-4314 Zeiningen, Switzerland
Houghton Springs Fish Farm, Dorset, UK
Cultures maintained at Wildlife International Ltd. Laboratories
Aquatic Bio Systems, Inc., Fort Collins, Colorado
Smithers Viscient, 790 Main Street, Wareham, Massachusetts 02571- 1037 USA
Smithers Viscient, 790 Main Street, Wareham, Massachusetts 02571 USA
Springborn laboratories
Syngenta Ltd. Jealott’s Hill International Research Centre Bracknell, Berkshire, RG42 6EY United Kingdom
Wildlife International Ltd., Maryland, USA
Wildlife International Ltd., Easton, USA
Smithers Viscient, 790 Main Street, Wareham, Massachusetts 02571 USA
Brixham Environmental Laboratory, AstraZeneca UK Limited, Brixham, UK
Springborn Laboratories Inc., Massachusetts 02571, USA
Smithers Viscient, 790 Main Street, Wareham, MA 02571-1037, USA
Wildlife International Ltd, Easton, MD, USA
Wildlife International A Division of EAG Inc. 8598 Commerce Drive Easton, MD 21601 USA
Smithers Viscient, 790 Main Street, Wareham, MA 02571-1037 USA
Ciba-Geigy Corporation, Post Office Box 18300, Greensboro, NC 27419, USA
Chesapeake Cultures, Hayes, Virginia
Smithers Viscient, 790 Main Street, Wareham, Massachusetts 02571-1037 USA
University of Sheffield, UK
Blue Frog Scientific Limited, Scott House, South St. Andrew Street, Edinburgh, EH2 2AZ, UK
MBL Aquaculture, Sarasota, Florida
Bayer AG (Pflanzenschutz Umweltforschung, Institut für Oekobiologie, D- 5090 Leverkusen)
Pflanzenphysiologisches Institut University, Nikolausberger Weg 180, D-3400 Göttingen, Germany
Envigo Research Limited Shardlow Business Park, Shardlow, Derbyshire, DE72 2GD, UK
Smithers Viscient, 790 Main Street, Wareham, Massachusetts 02571- 1037 USA
Wildlife International Ltd., Easton, Maryland, USA
David Francis, W.J. Mead Apiarist Supplies, Fowlmere, Cambridgshire
RCC AG, Itingen, Switzerland
Blades Biological Ltd, United Kingdom
ECT Oekotoxikologie GmbH, Germany
BioChem agrar, Labor für biologische und chemische, Analytik GmbH, Kupferstraße 6, 04827 Gerichshain, Germany
RCC Umweltchemie AG, P.O. Box, CH-4452 Itingen/BL, Switzerland
RCC Ltd, Environmental Chemistry & Pharmanalytics Division, CH-4452 Itingen, Switzerland
BioChem agrar Labor für biologische und chemische, Analytik GmbH, Kupferstraße 6 04827 Gerichshain, Germany
BioChem agrar, Labor für biologische und chemische, Analytik GmbH, Kupferstraße 6, 04827 Gerichshain, Germany
Pan-Agricultural Labs, Inc. 32380 Avenue 10 Madera, CA 93638 USA
Syngenta AG. Basel. Switzerland
Deutsche Sammlung von Mikroorganismen und Zellkulturen GmbH, Inhoffenstraße 7 B, 38124 Braunschweig, Germany
CIBA-GEIGY Ltd., Product Safety, Ecotoxicology, CH-4002 Basel, Switzerland
Springborn Smithers Laboratories 790 Main Street Wareham, MA 02571-1037
Syngenta crop protection AG, Research Biological science, Disease control, Stein
Syngenta Biosciences Pvt. Ltd., Ilhas Goa, India
Syngenta Technology & Projects, Huddersfield, United Kingdom
Stillmeadow. Inc.. 12852 Park One Drive. Sugar Land. TX 77478. USA
Texas Animal Specialties. Humble. TX
Nichols Rabbitry Inc. ; Lumberton. TX
Charles River Laboratories.. Wilmington. MA
Charles River Laboratories Edinburgh Ltd.. Elphinstone Research Centre. Tranent. East Lothian. EH33 2NE
Charles River Laboratories, Edinburgh, United Kingdom
tier3 solutions GmbH
tier3 solutions GmbH, Kolberger Str. 61-63, 51381 Leverkusen, Germany
Bayer CropScience AG
Syngenta, Jealott’s Hill International Research Centre, UK
Brixham Environmental Laboratory, Brixham, Devon, TQ5 8BA, UK
MITOX Consultants Science Park 408, 1098XH Amsterdam, The Netherlands
Eurofins Agrosciences Services EcoChem GmbH, Eutinger Str. 24, 75233 Niefern-Öschelbronn, Germany
Mambo-Tox Ltd., 2 Venture Road, Chilworth Science Park, Southampton SO16 7NP, United Kingdom
Biochem agrar GmbH, Gerichshain, Germany
“W. Neudorff GmbH KG”, An der Mühle 3, D- 31860 Emmertal
BioChem Agrar, Kupferstraβe 6, 04827 Gerichshain, Germany
Bayer CropScience AG, Monheim
BioChem agrar, Labor für biologische und chemische Analytik GmbH, Kupferstraße 6, 04827 Gerichshain, Germany
RIFCON GmbH, Hirschberg, Germany
Dr K Thomae GMBH, Chemisch-pharmazeutische Fabrik, D-7950 Biberach, Riss
Centre International de Toxicologie (C.I.T), Miserey, 27005 Evreux, France
Centre d’Elevage Lebeau, 78950 Gambais, France
CIBA-GEIGY Limited, Toxicology Services, Short-term Toxicology, 4332 Stein, Switzerland
Ciba-Geigy Ltd., CH-4002, Basel, Switzerland
Osage Catfish, Box 222, Missouri, USA
Mambo-Tox Ltd., 2 Venture Road, University Science Park, Southampton, SO16 7NP
Biologische Bundesanstalt (BBA), Berlin-Dahlem
“Bayer CropScience AG” Monheim
Zeneca Agrochemicals, Jealott’s Hill, United Kingdom
Eurofins Agroscience Services Chem GmbH, Hamburg, Germany
Harlan Cytotest Cell Research GmbH (Harlan CCR), Germany
Smithers Viscient (ESG) Ltd, Harrogate, UK
Covance Laboratories Limited, Harrogate, UK
Central Toxicology Laboratory, Alderley Park, Macclesfield, Cheshire, UK
Biological Services Section, Alderley Park, Macclesfield, Cheshire, UK
Charles River
Harlan Cytotest Cell Research GmBH, Rossdorf, Germany
Syngenta Crop Protection, Inc., Greensboro, NC 27419, USA
Cambridge Environmental Assessments, Battlegate Road, Boxworth, Cambridgeshire
Central Toxicology Laboratory, Syngenta
Harlan Laboratories Ltd. Zelgliweg,445 Itingen/Switzerland
Tecsolve UK Ltd., Glendale Park, North Ascot, Berkshire
Harlan Laboratories Ltd, Zelgliweg 1, 4452 Itingen, Switzerland
Harlan Laboratories
Katz Biotech AG, Baruth, Germany
Mambo-Tox Ltd., 2 Venture Road, Chilworth Science Park, Southampton, SO16 7NP
BioChem agrar, 04827 Gerichshain, Germany
W. Neudorff, 31860 Emmerthal, Germany
W. Neudorff GmbH KG, An der Mühle 3, 31860 Emmerthal, Germany
BioChem agrar Labor für biologische und chemische Analytik GmbH, Kupferstraße 6 04827 Gerichshain, Germany
“Biologische Bundesanstalt (BBA)”, Berlin-Dahlem
BioChem agrar, Labor für biologische und chemische Analytik GmbH, Kupferstraβe 6, 04827 Gerichshain, Germany
Syngenta Crop Protection, Münchwilen, Switzerland
Ciba-Geigy Ltd., Basle, Switzerland
Ciba-Geigy Corporation , Greensboro, NC, USA
Ciba-Geigy Corp., Greensboro, NC, USA
Nauchi, Shiraimachi, Inba-Gun, Chiba, Japan
Animal Metabolism, Ciba-Geigy Ltd., Basle, Switzerland
Hazleton Wisconsin, Inc. Madison, Wisconsin USA
CiToxLAB Hungary Ltd, Szabadsagpuszta, Hungary
Hazleton Wisconsin, Inc. Madison, Wis- consin USA
Stillmeadow Inc., Sugar Land TX, USA
Ciba-Geigy Corporation, Summit, NJ, USA
Ciba-Geigy Corp., Environmental Health Center, Farmington, CT, USA
Ciba-Geigy Limited, Pharmaceutical Division, 4002 Basel / Switzerland
Ciba-Geigy Limited, Experimental Pathology, 4002 Basel/ Switzerland
Ciba-Geigy Limited, Experimental Pathol- ogy, 4002 Basel / Switzerland
Hazleton Wisconsin, Madison, WI, USA
Ciba-Geigy Toxicology Services, ShortTerm Toxicology, 4332 Stein/ Switzerland
Ciba-Geigy Limited, Experimental Pathology, 4002 Basel / Switzerland
Hazleton Biotechnologies Company, Kensington, Maryland, USA
Ciba-Geigy Limited, Genetic Toxicology, 4002 Basel / Switzerland
Hazleton Washington, Inc., Vienna, Virginia 22182, USA
Ciba-Geigy Limited, 4002 Basel / Switzerland
Hazleton Raltech, Inc., a Subsidiary of Hazleton Laboratories America, Inc., Madison, Wisconsin, USA
Experimental Pathology Laboratories, Research Triangle Park
Toxicology/Cell Biology, Novartis Crop Protection Inc., Basel, Switzerland
Toxigenics, Inc., Decatur, IL 62526, USA
Argus Research Laboratories, Inc., Perkasie, PA, USA
Argus Research Laboratories Inc., Horsham, Pennsylvania 19044, USA
Ciba-Geigy Ltd.,Stein, Switzerland
Ciba-Geigy Ltd., Genetic Toxicology, Basle, Switzerland
Novartis Crop Protection AG, Stein, CH
Safepharm Laboratories Ltd., Shadlow, United Kingdom
Sandoz Agro Ltd., Department of Toxicology CH-4132 Muttenz, Switzerland
Hazleton Washington, Inc. Vienna, Virginia, USA
CXR Biosciences. Laboratory
Ciba-Geigy Corp., Greensboro NC, USA
Ciba-Geigy Ltd., Basel, CH
Novartis Agro S.A., Aigues-Vives, F
Ciba-Geigy SA, Rueil-Malmaison, F
Novartis Agro S.A., Aigues-Vives, France
Osage Catfisheries Inc., Osage Beach, Missouri 65065, USA
Aquatic Biosystems Corvalis
EPA, Corvalis, OR
Ward’s Natural Science, ON
Chilliwack Hatchery
Sun Valley Trout Farm, Abbotsford BC
Chilliwack Hatchery, BC
P. Hohler, CH-4314 Zeiningen, Switzerland
University of Sheffield , UK
Wildlife International Ltd., Maryland, US
Ciba-Geigy Ltd., Basle, CH
Stillmeadow Inc., Sugar Land, United States
Hazleton Wisconsin, Inc
ToxigeneticsINc. Decatur, IL, US
EG&G Bionomics
Biospheric Inc., Rockville, USA
Bionomics Aquatic Tox. Lab., Wareham, USA
Springborn Laboratories Inc.
Syngenta – Jealott’s Hill International, Bracknell, Berkshire, United Kingdom
Wildlife International Ltd., Easton MD, USA
Springborn Smithers Laboratories, Wareham, USA
Springborn Life Sciences Inc
Eg&G Bionomics (Fl), Pensacola, USA
Harlan Laboratories Ltd., Itingen, Switzerland
Solvias AG, Basel, Switzerland
T.R. Wilbury Laboratories Inc., Massachusetts, USA
Ciba-Geigy Ltd., Basle, CH
Stillmeadow Inc., Sugar Land, TX, USA
Syngenta Crop Protection, Munchwilen, Switzerland
RCC - Biological Research Laboratories, Füllinsdorf, Switzerland
Covance Laboratories, Harrogate, United Kingdom
Battelle UK Ltd, Chelmsford, Essex, UK
Zeneca Agrochemicals, Jealott’s Hill Research Station, Bracknell, Berkshire, UK
Xenobiotic Laboratories, Inc., Plainsboro, USA
Fraunhofer Institute, Schmallenberg, Germany
PTRL West, Hercules CA, USA
Eurofins Agroscience Services GmbH, Niefern-Öschel., Germany
ICI Agrochemicals, Bracknell, Berkshire, United Kingdom
Chemex International plc, Cambridge, United Kingdom
BASF, Limburgerhof, Germany
RIFCON, Leichlingen, Germany
Eurofins - GAB, Niefern Öschelbronn, Germany
River Thames, Maidenhead, Berkshire, UK
Beach N o . 24, Hayling Island, Hampshire, UK
Jealott’s Hill International Research Centre, Bracknell, Berkshire, RG42 6EY, UK
Zeneca Agrochemical s, Jealott’s Hill, United Kingdom
Zeneca Agrochemicals, Jealott’s Hill, United Kingdom
Jealott’s Hill Research Station. Syngenta Crop protection AG
Bayer CropScience, Monheim, Germany
Huntingdon Life Sciences Ltd., Huntingdon, United Kingdom
Eurofins Agroscience Services EcoChem GmbH, N- Osch., Germany
Eurofins Agroscience Services EcoChem GmbH, NOsch., Germany
Tier3 solutions GmbH, Germany
Syngenta Crop Protection AG
Jealott’s Hill Research Centre. Syngenta Crop protection AG
RCC Umweltchemie GmbH & Co KG
@@ -1,10 +0,0 @@
Long-term
Brown liquid
Brown solid
Hand-held
Manual-Hand held
Manual-Hand held
Weight:
Sprague
Weight and length
Aeration: Gentle
@@ -0,0 +1,2 @@
guideline
unpublished
@@ -1,4 +1,3 @@
Fundamentals of Applied Toxicology
in vitro
in-vitro
published paper
in vitro
in-vitro
@@ -1,87 +0,0 @@
7th ed.; The Iowa State University Press: Ames, IA
Analytical chemistry
Animal Reproduction
Animal Reproduction Science
Apidologie
Aquatic toxicology
Archives of Environmental Contamination and Toxicology
ATLA
Atmospheric. Environment
Australasian Journal of Ecotoxicology
Background lesions of laboratory Animals, A Color Atlas
Biometrics
Biometrika
Birth Defects Res. B. Dev. Reprod. Toxicol.
Crit Rev Toxicol
Current approaches in the statistical analysis of ecotoxicity data: guidance to application
Curr. Med. Chem.
Dongbei Nongye Daxue Xuebao
Ecotoxicology and Environmental Safety
Environ and Molecular Mutagenesis
Environ. Health Perspec.
Environ Health Perspect
Environ Health Perspect.
Environ Health Perspect. 1
Environmental Health
Environmental Health Perspectives
Environmental & Molecular Mutagenesis
Environmental Pollution
Environmental Science
Environmental Science and Technology.
Environmental Science & Technology
Environmental Toxicology and Chemistry
Environ Mol Muta- gen
Environ. Sci. Technol.
Environ Toxicol.
Env. Mol. Mutagen
Erna¨hrung
Essays in Honor of Harold Hotelling
Fish Sci
Food Cosmet. Toxicol.
Fundamentals of Applied Toxicology
Fundamentals of Applied Toxicology1988
High-Throughput Screening Methods in Toxicity Testing
High-Throughput Screening Methods in Toxicity Testing. Hoboken, NJ: John Wiley & Sons
http://www.iobc-wprs.org
Irish Journal of Agricultural and Food Research
J Econom
J Endocrinol
J. Invest. Derm
Journal of agricultural and food chemistry
Journal of Applied Entomology
Journal of Environmental Science and Health
Journal of Experimental Biology and Ecology
Journal of Hazardous Materials
Journal of the American College of Toxicology
Journal of Toxicology and
J Steroid Biochem Mol Biol
Marine Pollution Bulletin
Middle Atlantic Reproduction and Teratology Association
Mol. Cell. Endocrinol.
Mol Mutagen
Mutagenesis
Mutat Res
Nonparametric Statistics for the Behavioral Sciences
Principles and Procedures of Statistics
Principles and Procedures of Statistics, A Biometrical Approach
Proc Natl Acad Sci USA
Progress in
Psychopharmacologia
Reproductive Toxicology
RNA
Science of the Total Environment
Stain Technol
Statistical Methods
Statistical Methods,
Steinberg P,
Teratology
The American Statistician
Toxicol Chem
Toxicol in Vitro
Toxicological and Environmental Chemistry
Toxicological Sciences
Toxicologic Pathology
Toxicology and Applied Pharmacology
Toxicol Sci
Toxicol Sci.
Toxicol Sci. 1
@@ -1,12 +1,9 @@
acute toxicity
acute-toxicity
dermal penetration
eco toxicity
eco-toxicity
Env. Mol. Mutagen,
in vivo
in-vivo
ld
dermal penetration
oral toxicity
oral-toxicity
Processes & Impacts
acute toxicity
acute-toxicity
eco toxicity
eco-toxicity
@@ -1,48 +0,0 @@
Bartlett’s Test
Bonferroni-Holm Adjustment
Cochran-Armitage test
Cochran-Armitage Trend Step-Down Test
Dunnett's Multiple Comparison test
Dunnett’s test
Dunnett's "t" test
Dunnett test
Fisher Count-AllTM c
Fisher Scientific Company
Fisher’s Exact
Fisher's exact test
Fisher’s Exact Test
Ham's F10
Heinz body
Heinz body determinations
Jonckheere-Terpstra test (step-down)
Jonckheere Trend Test
Kay and Calandra system
Klimisch code
Kolmogorov-Smirnov's test
Krebs cycle
Litchfield and Wilcoxon method
Magnusson and Kligman
Magnusson & Kligman
MAGNUSSON & KLINGMAN METHOD
Mann-Whitney's test
Maximisation test, Buehler
Method of Magnusson and Kligman
methodology validated at Smithers Viscient.
n Williams' medium E a
Patterson-Kelly twinshell blender
Shapiro-Wilks and Levene’s tests
Shapiro-Wilks’ Test
Steel’s test
Tamhane-Dunnett
Tamhane-Dunnett test
the method Davies
Vogel-Bonner Medium E
Welch’s test
Welch t-test
Whitney test
Wilcoxon and Ansari-Bradley statistics
Williams multiple sequential t-test
Williams' test
Williams t-test
William test
Winkler titration technique,
@@ -1,66 +1,48 @@
Vulpes vulpes
african clawed frog
agalychnis callidryas
albino rat
american bullfrog tadpole
american toad
amphibian
amphibians
American bullfrog tadpole
american toad
anad platyrhynchos
anas platyrhynchos
Anas platyrhynchos
anuran
anurans
apodemus
apodemus flavicollis
apodemus syl vaticus
apodemus sylvaticus
arvicola terrestris
a. sylvaticus
avian
bank vole
bird
birds
bluegill
bluegill sunfish
bobwhite
bobwhite quail
Bovine
brachydanio rerio
brown hare
bufo americanus
bullfrog
Bufo americanus
brachydanio rerio
canary
carassius carassius
carp
catesbeiana
catfish
cattle
cattles
channel catfish
Chinook
chicken
chinese hamster
chinese hamsters
chinook
coho salmon
Colinus virginianus
colinus virginianus
columba palumbus
columbidae
common carp
common vole
Common carp
coturnix japonica
Coturnix japonica
cow
cows
crocidura russula
crucian carp
Crucian carp
cyprinodon variegatus
cyprinus carpio
(Danio rerio
dog
dogs
duck
ducks
Environmental & Molecular Mutagenesis
european brown hare
european rabbit
fathead minnow
fish
fishes
@@ -74,123 +56,175 @@ galaxias truttaceus
gasterosteus aculeatus
goat
goats
greater white-toothed shrew
guinea
guinea pig
guinea pigs
guinea-pigs
guppy
Guppy
hamster
hamsters
hen
hens
house mouse
hyla versicolor
Hyla versicolor
ictalurus melas
ictalurus punctatus
japanese quail
japonica
kisutch
lagomorph
lebistes reticulatus
leiostomus xanthurus
leisostomus xanthurus
lepomis macrochirus
lepus europaeus
limnocharis
limnodynastes
limnodynastes tasmaniensis
livestock
livestocks
mallard
mallard duck
mammal
mammalian
mammals
marten
martes
Mammalian
mice
microtus
microtus agrestis
microtus arvalis
microtus subterraneus
midwestern anurans
minnow
minnows
monkey
mouse
mus musculus
myodes glareolus
northern bobwhite
o. cuniculus
o. mykiss
oncorhynchus
oncorhynchus mykiss
oncorhynchus tshawytscha
oryctolagus cuniculus
Oncorhynchus mykiss
Oncorhynchus
O. mykiss
oryzias melastigma
oryzias melastigma larvae
o. tshawytscha
p. promelas
pagrus major
palumbus
pig
pigeon
pigeons
pigs
pimephales promela
pimephales promelas
Pseudacris triseriata
poecilia reticulata
poultry
p. promelas
pseudacris
pseudacris triseriata
quail
rabbit
rabbits
rainbow trout
Rana limnocharis
rana
rana catesbeiana
rana limnocharis
limnocharis
rana pipiens
rat
rats
r. catesbeiana
reptile
reptiles
ricefish
ruminant
ruminants
salmo gairdneri
salmon
serinus canaria
sheepshead minnow
sheepshead minnows
shrew
shrews
sorex araneus
spea multiplicata
Salmo gairdneri
salmon
spotted march frog
tadpoles
Taeniopygia guttata
terrestrial vertrebrates
toad
treefrog
triseriata
toad
terrestrial vertrebrates
Limnodynastes tasmaniensis
trout
tshawytscha
vole
voles
vulpes vulpes
white rabbits
white-toothed shrew
Vulpes vulpes
wistar
wistar rats
wood mice
wood mouse
wood pigeon
xenopus laevis
xenpous leavis
yellow-necked mouse
Zebra Finch
zebra fish
zebrafish
Salmo gairdneri
minnow
minnows
Pimephales promela
Cyprinodon variegatus
limnodynastes
Rana catesbeiana
R. catesbeiana
coho salmon
Oncorhynchus tshawytscha
O. tshawytscha
tshawytscha
catesbeiana
kisutch
Pseudacris triseriata
Pseudacris
triseriata
Wood pigeon
Columba palumbus
palumbus
Columbidae
shrew
shrews
bank vole
common vole
vole
voles
lagomorph
Wood mouse
Apodemus sylvaticus
A. sylvaticus
Apodemus flavicollis
Apodemus
mus musculus
Microtus arvalis
Microtus agrestis
Microtus
Arvicola terrestris
Sorex araneus
Myodes glareolus
yellow-necked mouse
house mouse
Oryctolagus cuniculus
marten
martes
white-toothed shrew
greater white-toothed shrew
Lepus europaeus
brown hare
European brown hare
European rabbit
O. cuniculus
Crocidura russula
Chinese Hamster
Rat
Rats
Dog
Chinese hamsters
Chinese hamster
Mouse
Guinea pig
Wistar rats
Rabbit
mammalian
Japanese quail
Microtus subterraneus
Lepomis macrochirus
P. promelas
Cyprinus carpio
Fish
Ictalurus punctatus
Carassius carassius
Lepomis macrochirus
Poecilia reticulata
Lebistes reticulatus
Lepomis macrochirus
Leiostomus xanthurus
Pimephales promelas
Lepomis macrochirus
Albino rat
Hen
Goat
Livestock
Guinea Pigs
Hamster
wood mouse
Rabbits
Mice
Rainbow trout
Canary
Serinus canaria
Guinea Pig
Cow
Pigs
Poultry
Guinea-pigs
White rabbits
Birds
Wood mice
Mallard
@@ -5,269 +5,74 @@ import com.iqser.red.service.redaction.v1.server.redaction.model.Section
global Section section
// --------------------------------------- CBI rules -------------------------------------------------------------------
rule "1: Redacted because Section contains Vertebrate"
when
Section(matchesType("vertebrate"))
then
section.redact("CBI_author", 1, "Vertebrate found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 1, "Vertebrate found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
when
eval(section.contains("vertebrate")==true);
then
section.redact("name", 1, "Redacted because Section contains Vertebrate");
section.redact("address", 1, "Redacted because Section contains Vertebrate");
end
rule "2: Not Redacted because Section contains no Vertebrate"
when
Section(!matchesType("vertebrate"))
then
section.redactNot("CBI_author", 2, "No Vertebrate found");
section.redactNot("CBI_address", 2, "No Vertebrate found");
end
when
eval(section.contains("vertebrate")==false);
then
section.redactNot("name", 2, "Not Redacted because Section contains no Vertebrate");
section.redactNot("address", 2, "Not Redacted because Section contains no Vertebrate");
end
rule "3: Do not redact Names and Addresses if no redaction Indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("no_redaction_indicator"))
then
section.redactNot("CBI_author", 3, "Vertebrate and No Redaction Indicator found");
section.redactNot("CBI_address", 3, "Vertebrate and No Redaction Indicator found");
end
when
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true);
then
section.redactNot("name", 3, "Vertebrate was found, but also a no redaction indicator");
section.redactNot("address", 3, "Vertebrate was found, but also a no redaction indicator");
end
rule "4: Do not redact Names and Addresses if no redaction Indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("published_information"))
then
section.redactNot("CBI_author", 4, "Vertebrate and Published Information found");
section.redactNot("CBI_address", 4, "Vertebrate and Published Information found");
end
rule "4: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
when
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true && section.contains("redaction_indicator")==true);
then
section.redact("name", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
section.redact("address", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
end
rule "5: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("no_redaction_indicator"), matchesType("redaction_indicator"))
then
section.redact("CBI_author", 5, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 5, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "5: Do not redact in guideline sections"
when
eval(section.headlineContainsWord("guideline") || section.headlineContainsWord("Guidance"));
then
section.redactNot("name", 5, "Section is a guideline section.");
section.redactNot("address", 5, "Section is a guideline section.");
end
rule "6: Not redacted because Vertebrate Study = N"
when
Section(rowEquals("Vertebrate study Y/N", "N") || rowEquals("Vertebrate study Y/N", "No"))
then
section.redactNotCell("Author(s)", 6, "CBI_author", true, "Not redacted because row is not a vertebrate study");
section.redactNot("CBI_author", 6, "Not redacted because row is not a vertebrate study");
section.redactNot("CBI_address", 6, "Not redacted because row is not a vertebrate study");
section.highlightCell("Vertebrate study Y/N", 6, "hint_only");
end
rule "6: Redact contact information, if applicant is found"
when
eval(section.getText().toLowerCase().contains("applicant") == true);
then
section.redactLineAfter("Name:", "address", 6, "contact information was found");
section.redactBetween("Address:", "Contact", "address", 6, "contact information was found");
section.redactLineAfter("Contact point:", "address", 6, "contact information was found");
section.redactLineAfter("Phone:", "address", 6, "contact information was found");
section.redactLineAfter("Fax:", "address", 6, "contact information was found");
section.redactLineAfter("E-mail:", "address", 6, "contact information was found");
section.redactLineAfter("Contact:", "address", 6, "contact information was found");
section.redactLineAfter("Telephone number:", "address", 6, "contact information was found");
end
rule "7: Redact if must redact entry is found"
when
Section(matchesType("must_redact"))
then
section.redact("CBI_author", 7, "must_redact entry was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 7, "must_redact entry was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "8: Redact Authors and Addresses in Reference Table if it is a Vertebrate study"
when
Section(rowEquals("Vertebrate study Y/N", "Y") || rowEquals("Vertebrate study Y/N", "Yes"))
then
section.redactCell("Author(s)", 8, "CBI_author", true, "Redacted because row is a vertebrate study", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 8, "Redacted because row is a vertebrate study", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.highlightCell("Vertebrate study Y/N", 8, "must_redact");
end
rule "9: Redact sponsor company"
when
Section(searchText.toLowerCase().contains("batches produced at"))
then
section.redactIfPrecededBy("batches produced at", "CBI_sponsor", 9, "Redacted because it represents a sponsor company", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.addHintAnnotation("batches produced at", "must_redact");
end
rule "10: Redact determination of residues"
when
Section((
searchText.toLowerCase.contains("determination of residues") ||
searchText.toLowerCase.contains("determination of total residues")
) && (
searchText.toLowerCase.contains("livestock") ||
searchText.toLowerCase.contains("live stock") ||
searchText.toLowerCase.contains("tissue") ||
searchText.toLowerCase.contains("tissues") ||
searchText.toLowerCase.contains("liver") ||
searchText.toLowerCase.contains("muscle") ||
searchText.toLowerCase.contains("bovine") ||
searchText.toLowerCase.contains("ruminant") ||
searchText.toLowerCase.contains("ruminants")
))
then
section.redact("CBI_author", 10, "Determination of residues was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 10, "Determination of residues was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.addHintAnnotation("determination of residues", "must_redact");
section.addHintAnnotation("livestock", "must_redact");
section.addHintAnnotation("live stock", "must_redact");
section.addHintAnnotation("tissue", "must_redact");
section.addHintAnnotation("tissues", "must_redact");
section.addHintAnnotation("liver", "must_redact");
section.addHintAnnotation("muscle", "must_redact");
section.addHintAnnotation("bovine", "must_redact");
section.addHintAnnotation("ruminant", "must_redact");
section.addHintAnnotation("ruminants", "must_redact");
end
rule "11: Redact if CTL/* or BL/* was found"
when
Section(searchText.contains("CTL/") || searchText.contains("BL/"))
then
section.redact("CBI_author", 11, "Laboraty for vertebrate studies found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 11, "Laboraty for vertebrate studies found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.addHintAnnotation("CTL", "must_redact");
section.addHintAnnotation("BL", "must_redact");
end
rule "12: Add recommendation for et al. author"
when
Section(searchText.contains("et al."))
then
section.addRecommendationByRegEx("\\b([A-ZÄÖÜ][^\\s\\.,]+( [A-ZÄÖÜ]\\.?)?( [A-ZÄÖÜ]\\.?)?) et al\\.?", false, 1, "CBI_author");
end
// --------------------------------------- PII rules -------------------------------------------------------------------
rule "13: Redacted PII Personal Identification Information"
when
Section(matchesType("PII"))
then
section.redact("PII", 13, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "14: Redact Emails by RegEx"
when
Section(searchText.contains("@"))
then
section.redactByRegEx("\\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,4}\\b", true, 0, "PII", 14, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "15: Redact contact information"
when
Section(text.contains("Contact point:")
|| text.contains("Phone:")
|| text.contains("Fax:")
|| text.contains("Tel.:")
|| text.contains("Tel:")
|| text.contains("E-mail:")
|| text.contains("Email:")
|| text.contains("e-mail:")
|| text.contains("E-mail address:")
|| text.contains("Alternative contact:")
|| text.contains("Telephone number:")
|| text.contains("Telephone No:")
|| text.contains("Fax number:")
|| text.contains("Telephone:")
|| text.contains("European contact:"))
then
section.redactLineAfter("Contact point:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "16: Redact contact information if applicant is found"
when
Section(headlineContainsWord("applicant") || text.contains("Applicant") || headlineContainsWord("Primary contact") || headlineContainsWord("Alternative contact") || text.contains("Telephone number:"))
then
section.redactLineAfter("Contact point:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "17: Redact contact information if Producer is found"
when
Section(text.toLowerCase().contains("producer of the plant protection") || text.toLowerCase().contains("producer of the active substance") || text.contains("Manufacturer of the active substance") || text.contains("Manufacturer:") || text.contains("Producer or producers of the active substance"))
then
section.redactLineAfter("Contact:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "18: Redact AUTHOR(S)"
when
Section(searchText.contains("AUTHOR(S):"))
then
section.redactLinesBetween("AUTHOR(S):", "COMPLETION DATE:", "PII", 18, true, "AUTHOR(S) was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "19: Redact PERFORMING LABORATORY"
when
Section(searchText.contains("PERFORMING LABORATORY:"))
then
section.redactBetween("PERFORMING LABORATORY:", "LABORATORY PROJECT ID:", "PII", 19, true, "PERFORMING LABORATORY was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "20: Redact On behalf of Sequani Ltd.:"
when
Section(searchText.contains("On behalf of Sequani Ltd.: Name Title"))
then
section.redactBetween("On behalf of Sequani Ltd.: Name Title", "On behalf of", "PII", 20, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "21: Redact On behalf of Syngenta Ltd.:"
when
Section(searchText.contains("On behalf of Syngenta Ltd.: Name Title"))
then
section.redactBetween("On behalf of Syngenta Ltd.: Name Title", "Study dates", "PII", 21, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "7: Redact contact information, if 'Producer of the plant protection product' is found"
when
eval(section.getText().contains("Producer of the plant protection product"));
then
section.redactLineAfter("Name:", "address", 7, "Producer of the plant protection product was found");
section.redactBetween("Address:", "Contact", "address", 7, "Producer of the plant protection product was found");
section.redactBetween("Contact:", "Phone", "address", 7, "Producer of the plant protection product was found");
section.redactLineAfter("Phone:", "address", 7, "Producer of the plant protection product was found");
section.redactLineAfter("Fax:", "address", 7, "Producer of the plant protection product was found");
section.redactLineAfter("E-mail:", "address", 7, "Producer of the plant protection product was found");
end

Some files were not shown because too many files have changed in this diff Show More