Compare commits

...
7 Commits
Author SHA1 Message Date
lmaldacker 9ceba4abf3 Adjust imports 2021-02-09 10:36:58 +01:00
lmaldacker ebca1fc750 Adjust pom 2021-02-09 10:21:34 +01:00
lmaldacker 37125f14fd Add logs 2021-02-08 16:22:26 +01:00
Dominique Eiflaender 6d9ed080ce Pull request #120: RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line
Merge in RED/redaction-service from RED-1039 to master

* commit '00b0cb160342f1857ac0e523f994918057d5fc6b':
  RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line
2021-02-08 14:50:02 +01:00
Dominique Eifländer 00b0cb1603 RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line 2021-02-08 14:09:11 +01:00
Dominique Eiflaender 5ecf21290c Pull request #119: RED-1046: Ignore dictionary rank for words that are explicitly set in the rules
Merge in RED/redaction-service from RED-1046 to master

* commit '8965e7654867ff666c54c841319376c5899e7326':
  RED-1046: Ignore dictionary rank for words that are explicitly set in the rules
2021-02-05 15:08:15 +01:00
Dominique Eifländer 8965e76548 RED-1046: Ignore dictionary rank for words that are explicitly set in the rules 2021-02-05 14:55:16 +01:00
13 changed files with 96 additions and 16 deletions
@@ -10,6 +10,9 @@
</parent>
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<slf4j.version>1.7.30</slf4j.version>
</properties>
<dependencies>
<dependency>
@@ -68,6 +71,11 @@
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
</dependency>
<dependency>
<groupId>org.slf4j</groupId>
<artifactId>slf4j-api</artifactId>
<version>${slf4j.version}</version>
</dependency>
<!-- spring -->
<dependency>
<groupId>org.springframework.cloud</groupId>
@@ -51,6 +51,8 @@ public class RedactionController implements RedactionResource {
@Override
public AnalyzeResult analyze(@RequestBody AnalyzeRequest analyzeRequest) {
log.info("Starting redaction analysis...");
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(analyzeRequest.getDocument()))) {
pdDocument.setAllSecurityToBeRemoved(true);
@@ -32,6 +32,7 @@ public class SearchableText {
}
@SuppressWarnings("checkstyle:ModifiedControlVariable")
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive,
List<TextPositionSequence> sequencesSubList) {
@@ -66,9 +67,12 @@ public class SearchableText {
for (int j = 0; j < searchSpace.get(i).length(); j++) {
if (i > 0 && j == 0 && searchSpace.get(i).charAt(0, caseInsensitive) == ' ' && searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i).charAt(j - 1, caseInsensitive) == ' ') {
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
.charAt(searchSpace.get(i - 1)
.length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i)
.charAt(j - 1, caseInsensitive) == ' ') {
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions()
.isEmpty()) {
crossSequenceParts.add(partMatch);
}
continue;
@@ -80,8 +84,8 @@ public class SearchableText {
counter++;
}
if (searchSpace.get(i)
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace.get(i)
if (searchSpace.get(i).charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace
.get(i)
.charAt(j, caseInsensitive) == '-') {
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(searchSpace.get(i)
@@ -100,14 +104,15 @@ public class SearchableText {
if (counter == searchString.length()) {
crossSequenceParts.add(partMatch);
if (i == searchSpace.size() - 1 && j == searchSpace.get(i).length() - 1 || j != searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i)
if (i == searchSpace.size() - 1 && j == searchSpace.get(i)
.length() - 1 || j != searchSpace.get(i).length() - 1 && isSeparator(searchSpace.get(i)
.charAt(j + 1, caseInsensitive)) || j == searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i + 1)
.charAt(0, caseInsensitive)) || j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
.charAt(0, caseInsensitive)) || j == searchSpace.get(i)
.length() - 1 && searchSpace.get(i)
.charAt(j, caseInsensitive) != ' ' && searchSpace.get(i + 1)
.charAt(0, caseInsensitive) != ' ') {
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts));
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts, normalizedSearchString));
}
counter = 0;
@@ -130,15 +135,21 @@ public class SearchableText {
}
return finalMatches;
}
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts,
String searchString) {
List<EntityPositionSequence> result = new ArrayList<>();
String asString = buildString(crossSequenceParts);
if (!asString.equalsIgnoreCase(searchString)) {
return result;
}
String plainId = IdBuilder.buildId(crossSequenceParts);
String id = plainId;
List<EntityPositionSequence> result = new ArrayList<>();
int currentPage = -1;
int idDiffentPageSuffix = 1;
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
@@ -173,6 +184,12 @@ public class SearchableText {
@Override
public String toString() {
return buildString(sequences);
}
public String buildString(List<TextPositionSequence> sequences) {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
@@ -156,7 +156,7 @@ public class Section {
public void addHintAnnotation(String value, String asType) {
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
}
@@ -30,7 +30,9 @@ import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.model.SectionRectangle;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class AnnotationService {
@@ -39,7 +41,7 @@ public class AnnotationService {
public void annotate(PDDocument document, RedactionLog redactionLog, SectionGrid sectionGrid) throws IOException {
log.info("Annotating document.");
Map<Integer, List<RedactionLogEntry>> redactionLogPerPage = convertRedactionLog(redactionLog);
for (int page = 1; page <= document.getNumberOfPages(); page++) {
@@ -56,6 +58,7 @@ public class AnnotationService {
addAnnotations(logEntries, pdPage, page);
}
}
log.info("Finished document annotation.");
}
@@ -50,6 +50,8 @@ public class EntityRedactionService {
public void processDocument(Document classifiedDoc, String ruleSetId, ManualRedactions manualRedactions) {
log.info("Processing document.");
dictionaryService.updateDictionary(ruleSetId);
KieContainer container = droolsExecutionService.updateRules(ruleSetId);
long rulesVersion = droolsExecutionService.getRulesVersion(ruleSetId);
@@ -93,6 +95,9 @@ public class EntityRedactionService {
classifiedDoc.setDictionaryVersion(dictionary.getVersion());
classifiedDoc.setRulesVersion(rulesVersion);
log.info("Finished document processing.");
}
@@ -34,7 +34,9 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class RedactionLogCreatorService {
@@ -47,6 +49,8 @@ public class RedactionLogCreatorService {
public void createRedactionLog(Document classifiedDoc, int numberOfPages, ManualRedactions manualRedactions,
String ruleSetId) {
log.info("Creating redaction log.");
Set<Integer> manualRedactionPages = getManualRedactionPages(manualRedactions);
for (int page = 1; page <= numberOfPages; page++) {
@@ -65,6 +69,8 @@ public class RedactionLogCreatorService {
addImageEntries(classifiedDoc, page, ruleSetId);
}
}
log.info("Finished redaction log creation.");
}
@@ -194,7 +200,7 @@ public class RedactionLogCreatorService {
startIndex = i;
}
}
if (startIndex != textPositions.size() - 1) {
if (startIndex != textPositions.size()) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, textPositions.size()), page).getRectangle());
}
}
@@ -121,4 +121,10 @@ public class EntitySearchUtils {
}
entities.add(found);
}
public void addEntitiesIgnoreRank(Set<Entity> entities, Set<Entity> found){
// HashSet keeps old value but we want the new.
entities.removeAll(found);
entities.addAll(found);
}
}
@@ -39,6 +39,8 @@ public class PdfSegmentationService {
public Document parseDocument(PDDocument pdDocument) throws IOException {
log.info("Parsing document.");
Document document = new Document();
List<Page> pages = new ArrayList<>();
@@ -91,6 +93,8 @@ public class PdfSegmentationService {
sectionsBuilderService.buildSections(document);
log.info("Finished document parsing.");
return document;
}
@@ -19,11 +19,16 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractT
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
public class SectionsBuilderService {
public void buildSections(Document document) {
log.info("Building sections.");
List<AbstractTextContainer> chunkWords = new ArrayList<>();
List<Paragraph> chunkBlockList = new ArrayList<>();
List<Header> headers = new ArrayList<>();
@@ -87,6 +92,9 @@ public class SectionsBuilderService {
document.setParagraphs(chunkBlockList);
document.setHeaders(headers);
document.setFooters(footers);
log.info("Finished section building.");
}
@@ -15,11 +15,16 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.CleanRuli
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
public class RulingCleaningService {
public CleanRulings getCleanRulings(List<Ruling> rulings, float minCharWidth, float maxCharHeight) {
log.info("Getting clean rulings.");
if (!rulings.isEmpty()) {
snapPoints(rulings, minCharWidth, maxCharHeight);
}
@@ -23,11 +23,16 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
public class TableExtractionService {
public void extractTables(CleanRulings cleanRulings, Page page) {
log.info("Extracting tables.");
List<Cell> cells = findCells(cleanRulings.getHorizontal(), cleanRulings.getVertical());
List<TextBlock> toBeRemoved = new ArrayList<>();
@@ -78,6 +83,9 @@ public class TableExtractionService {
}
page.getTextBlocks().removeAll(toBeRemoved);
log.info("Finished table extraction.");
}
@@ -418,6 +418,7 @@ public class RedactionIntegrationTest {
}
private List<File> getPathsRecursively(File path) {
List<File> result = new ArrayList<>();
@@ -439,9 +440,16 @@ public class RedactionIntegrationTest {
@Test
public void redactionTest() throws IOException {
// 49 Cyprodinil - EU AIR3 - MCA Section 8 Supplement - Ecotoxicological studies on the active substance.pdf
// 182 Fludioxonil - EU AIR3 - MCA Section 8 Supplement - Ecotoxicological studies on the active substance.pdf
// 38 A14325E - EU AIR3 - MCP Section 10 - Ecotoxicological studies on the plant protection product.pdf
// 91 Trinexapac-ethyl_RAR_01_Volume_1_2018-02-23.pdf
// 95 Trinexapac-ethyl_RAR_08_Volume_3CA_B-6_2018-01-10.pdf
System.out.println("redactionTest");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Cyprodinil/49 Cyprodinil - EU AIR3 - MCA Section 8 Supplement - Ecotoxicological studies on the active substance.pdf");
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)