Compare commits

...
Author SHA1 Message Date
Dominique Eifländer 22caa232c0 First steps for incremental analysis 2021-03-05 15:14:34 +01:00
Dominique Eiflaender dd4f1b34e1 Pull request #125: RED-1090: Return text with areas and sectionNumber
Merge in RED/redaction-service from RED-1090 to master

* commit '0dd47111c5294d9a9a0a10e174c2832436983205':
  RED-1090: Return text with areas and sectionNumber
2021-02-25 15:10:58 +01:00
Dominique Eifländer 0dd47111c5 RED-1090: Return text with areas and sectionNumber 2021-02-25 14:55:53 +01:00
Dominique Eiflaender 7d507c8fda Pull request #124: Adjusted rules to new requirements, add possibility to addRedaction for rules
Merge in RED/redaction-service from NewRequirementsRules to master

* commit 'b6a13d1ff8a55f1b5ef528af5ea726b5e17b4950':
  Adjusted rules to new requirements, add possibility to addRedaction for rules
2021-02-18 10:48:26 +01:00
Dominique Eifländer b6a13d1ff8 Adjusted rules to new requirements, add possibility to addRedaction for rules 2021-02-18 10:33:38 +01:00
Dominique Eiflaender 06e418c34e Pull request #123: RED-1070: Fixed not finding annotation on not classified textblocks
Merge in RED/redaction-service from RED-1070 to master

* commit 'de1dea7ac323dc005baeb0660cc492ab84a88164':
  RED-1070: Fixed not finding annotation on not classified textblocks
2021-02-18 09:18:59 +01:00
Dominique Eifländer de1dea7ac3 RED-1070: Fixed not finding annotation on not classified textblocks 2021-02-18 08:58:51 +01:00
Dominique Eiflaender 53509f072e Pull request #122: RED-1055: Enabled to force redact ignore annotations
Merge in RED/redaction-service from RED-1055 to master

* commit 'e0fba8d38ce9858235ca84182f4587d472890ef1':
  RED-1055: Enabled to force redact ignore annotations
2021-02-12 10:55:04 +01:00
Dominique Eifländer e0fba8d38c RED-1055: Enabled to force redact ignore annotations 2021-02-12 10:41:32 +01:00
Lena  Maldacker 8989191314 Pull request #121: Add hint annotation by reg-ex and expand to hint annotation by reg-ex
Merge in RED/redaction-service from dismissGen to master

* commit '0f263c69b8f2eb7201f062cc263c8d06b7a710a1':
  Add hint annotation by reg-ex and expand to hint annotation by reg-ex
2021-02-11 17:22:55 +01:00
lmaldacker 0f263c69b8 Add hint annotation by reg-ex and expand to hint annotation by reg-ex 2021-02-11 16:53:07 +01:00
Kresnadi Budisantoso ea1c2898bc Remove duplicate build tasks in bamboo specs 2021-02-11 14:11:22 +01:00
Dominique Eiflaender 6d9ed080ce Pull request #120: RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line
Merge in RED/redaction-service from RED-1039 to master

* commit '00b0cb160342f1857ac0e523f994918057d5fc6b':
  RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line
2021-02-08 14:50:02 +01:00
Dominique Eifländer 00b0cb1603 RED-1039: Fixed finding textpositions, RED-1042: Fixed get rectangles per line 2021-02-08 14:09:11 +01:00
Dominique Eiflaender 5ecf21290c Pull request #119: RED-1046: Ignore dictionary rank for words that are explicitly set in the rules
Merge in RED/redaction-service from RED-1046 to master

* commit '8965e7654867ff666c54c841319376c5899e7326':
  RED-1046: Ignore dictionary rank for words that are explicitly set in the rules
2021-02-05 15:08:15 +01:00
Dominique Eifländer 8965e76548 RED-1046: Ignore dictionary rank for words that are explicitly set in the rules 2021-02-05 14:55:16 +01:00
Dominique Eiflaender ba8c4dc240 Pull request #118: RED-1045: Enabled to redact in headers and footers
Merge in RED/redaction-service from RED-1045 to master

* commit '577db37b11dfd0243d22986a6d224700f0b49693':
  RED-1045: Enabled to redact in headers and footers
2021-02-05 13:16:55 +01:00
Dominique Eifländer 577db37b11 RED-1045: Enabled to redact in headers and footers 2021-02-05 13:05:12 +01:00
Dominique Eiflaender 154e09b843 Pull request #117: Fixed several table extraction problems
Merge in RED/redaction-service from tableExtractionImprovement to master

* commit 'a101b98a400b2a35635f2c5e0e894ac849288c3b':
  Fixed several table extraction problems
2021-02-04 15:21:12 +01:00
Dominique Eifländer a101b98a40 Fixed several table extraction problems 2021-02-04 15:08:48 +01:00
Dominique Eiflaender 76ecfdabd3 Pull request #116: Fixed table extraction problems
Merge in RED/redaction-service from TableExtractionProbemFix to master

* commit 'fc2ac03691871a8a1f3c8a29756e3a60e63e16bf':
  Fixed table extraction problems
2021-02-03 14:49:01 +01:00
Dominique Eifländer fc2ac03691 Fixed table extraction problems 2021-02-03 14:34:29 +01:00
Dominique Eiflaender 7898f6a30f Pull request #115: Expand CBI Authors with firstname initials
Merge in RED/redaction-service from ExpandAuthors to master

* commit '35f3582d0850053c1dc38ef0b149daa47bd9cc56':
  Expand CBI Authors with firstname initials
2021-02-02 12:35:07 +01:00
Dominique Eifländer 35f3582d08 Expand CBI Authors with firstname initials 2021-02-02 12:20:40 +01:00
Dominique Eiflaender b4699caaae Pull request #114: Fixed position problem for rotated images
Merge in RED/redaction-service from rotatedImages to master

* commit 'd8e444280b5a025e19785c70c9fcbb11480b5042':
  Fixed position problem for rotated images
2021-02-01 16:17:19 +01:00
Dominique Eifländer d8e444280b Fixed position problem for rotated images 2021-02-01 16:04:45 +01:00
Dominique Eiflaender ec4db538a7 Pull request #113: Ignore too small images
Merge in RED/redaction-service from IgnoreTooSmall to master

* commit '39ca191b9cb5dec5fbb01e8319cbbe361c7cd85e':
  Ignore too small images
2021-02-01 15:46:57 +01:00
Dominique Eifländer 39ca191b9c Ignore too small images 2021-02-01 15:14:46 +01:00
Dominique Eiflaender 0cec88f1b4 Pull request #112: Annotate images
Merge in RED/redaction-service from annotateImages to master

* commit 'acddfafa5b60d0120a48bdf47fe218bf59e359d2':
  Annotate images
2021-02-01 14:12:28 +01:00
Dominique Eifländer acddfafa5b Annotate images 2021-02-01 13:52:18 +01:00
Dominique Eiflaender 6b68890fcf Pull request #111: RED-1019: Fixed returning numberOfPages
Merge in RED/redaction-service from RED-1019 to master

* commit '1ed99412595b560d92e542d949e52d19d58beb2c':
  RED-1019: Fixed returning numberOfPages
2021-01-29 13:32:05 +01:00
Dominique Eifländer 1ed9941259 RED-1019: Fixed returning numberOfPages 2021-01-29 13:17:25 +01:00
Timo Bejan 1e7d535a33 Pull request #110: fixed issue affecting service on restarts
Merge in RED/redaction-service from bugfix-annotate to master

* commit '1091d6a886243387f23da32d0c678b66b0c21f26':
  fixed issue affecting service on restarts
2021-01-28 10:17:02 +01:00
Timo 1091d6a886 fixed issue affecting service on restarts 2021-01-28 11:05:01 +02:00
Dominique Eiflaender d8a1e43991 Pull request #109: RED-1010: Splittet redaction endpoint to analyis and annotation endpoints
Merge in RED/redaction-service from RED-1010 to master

* commit 'd739a4f2f5e95d25fc68cda35f1a5cc79b4ded5f':
  RED-1010: Splittet redaction endpoint to analyis and annotation endpoints
2021-01-27 15:32:24 +01:00
Dominique Eifländer d739a4f2f5 RED-1010: Splittet redaction endpoint to analyis and annotation endpoints 2021-01-27 15:19:38 +01:00
Dominique Eiflaender 40a9b39688 Pull request #108: RED-1004: Fixed position problems when mediabox is bigger than cropbox
Merge in RED/redaction-service from RED-1004 to master

* commit '62b960f2ead02b6f29537447077822c171f65bf8':
  RED-1004: Fixed position problems when mediabox is bigger than cropbox
2021-01-27 11:47:00 +01:00
Dominique Eifländer 62b960f2ea RED-1004: Fixed position problems when mediabox is bigger than cropbox 2021-01-27 11:34:17 +01:00
Dominique Eiflaender 0d9e318d25 Pull request #107: Always check dictionary rank when overriding annotations
Merge in RED/redaction-service from checkOverrideRank to master

* commit 'a76095c5d62e1d5f39013d77c95328dee9f7fbea':
  Always check dictionary rank when overriding annotations
2021-01-26 12:16:18 +01:00
Dominique Eifländer a76095c5d6 Always check dictionary rank when overriding annotations 2021-01-26 12:04:26 +01:00
Dominique Eiflaender 8d1e574dde Pull request #106: Added possibility to redact all Author tables
Merge in RED/redaction-service from AllAuthorTables to master

* commit '43a3d76f1c0ac9e3803cc1e228508b48686e893d':
  Added possibility to redact all Author tables
2021-01-25 16:11:05 +01:00
Dominique Eifländer 43a3d76f1c Added possibility to redact all Author tables 2021-01-25 15:59:08 +01:00
Dominique Eiflaender 4ed41b1a7e Pull request #105: Fixed recognize rules and exceptions for redactAndRecommend, fixed wrong order of rules(published information)
Merge in RED/redaction-service from redactAndRecommendFix2 to master

* commit '7dfaf604e7fb0d09e66a1688b1657006a3e92156':
  Fixed recognize rules and exceptions for redactAndRecommend, fixed wrong order of rules(published information)
2021-01-25 11:39:17 +01:00
Dominique Eifländer 7dfaf604e7 Fixed recognize rules and exceptions for redactAndRecommend, fixed wrong order of rules(published information) 2021-01-25 11:20:35 +01:00
Dominique Eiflaender 70dea61e24 Pull request #104: Recoginze rules and exceptions for redactAndRecommend, Do not redact publish information authors in new rulesset
Merge in RED/redaction-service from RedactAndRecommendFix to master

* commit '531eeebae1b5811d4e03086f5a4060450f86513a':
  Recoginze rules and exceptions for redactAndRecommend, Do not redact publish information authors in new rulesset
2021-01-25 10:21:21 +01:00
Dominique Eifländer 531eeebae1 Recoginze rules and exceptions for redactAndRecommend, Do not redact publish information authors in new rulesset 2021-01-25 09:46:09 +01:00
Dominique Eiflaender 375d84b236 Pull request #103: RED-934: Added rule to redact purity and et al Authors will be redacted where they are found
Merge in RED/redaction-service from RED-934 to master

* commit '33795527fdc4fd82f6ad26ec5720a539620b1293':
  RED-934: Added rule to redact purity and et al Authors will be redacted where they are found
2021-01-20 13:30:16 +01:00
Dominique Eifländer 33795527fd RED-934: Added rule to redact purity and et al Authors will be redacted where they are found 2021-01-20 13:18:37 +01:00
Dominique Eiflaender 91772dfb0d Pull request #102: RED-937: Handle rules version per RuleSetId
Merge in RED/redaction-service from RED-937 to master

* commit '89f642ba900ea60aa08f75da75ff09fd93eb2fc2':
  RED-937: Handle rules version per RuleSetId
2021-01-13 13:04:25 +01:00
Dominique Eifländer 89f642ba90 RED-937: Handle rules version per RuleSetId 2021-01-13 12:52:59 +01:00
Dominique Eiflaender 2dd8dd73c8 Pull request #101: Clean recommendation values starting with :
Merge in RED/redaction-service from cleanRecommendation to master

* commit '9104db6fa4f186b4758a2ccc8265f6ca004fca19':
  Clean recommendation values starting with :
2021-01-12 13:52:48 +01:00
Dominique Eifländer 9104db6fa4 Clean recommendation values starting with : 2021-01-12 13:41:19 +01:00
Dominique Eiflaender 59778682a2 Pull request #100: Recommend CBI_address in Test Organism/Animals sections
Merge in RED/redaction-service from recommendCBIaddress to master

* commit 'f3b0eb291107966d4ae2490f8d096ef2eabd8452':
  Recommend CBI_address in Test Organism/Animals sections
2021-01-12 11:42:52 +01:00
Dominique Eifländer f3b0eb2911 Recommend CBI_address in Test Organism/Animals sections 2021-01-12 11:30:40 +01:00
Dominique Eiflaender 4aef72552c Pull request #99: Revert idRemoval fix, removed duplicate code
Merge in RED/redaction-service from duplicateCodeRemove to master

* commit 'd10c0a7900ffd85c254656db377d5dfc1505c5ab':
  Revert idRemoval fix, removed duplicate code
2021-01-08 14:08:18 +01:00
56 changed files with 2750 additions and 561 deletions
@@ -72,14 +72,6 @@ public class PlanSpec {
.jobs(new Job("Default Job",
new BambooKey("JOB1"))
.tasks(
new ScriptTask()
.description("Clean")
.inlineBody("#!/bin/bash\n" +
"set -e\n" +
"rm -rf ./*"),
new VcsCheckoutTask()
.description("Checkout Default Repository")
.checkoutItems(new CheckoutItem().defaultRepository()),
new ScriptTask()
.description("Clean")
.inlineBody("#!/bin/bash\n" +
+1 -1
View File
@@ -5,7 +5,7 @@
<parent>
<artifactId>platform-dependency</artifactId>
<groupId>com.iqser.red</groupId>
<version>1.0.2</version>
<version>1.0.3</version>
</parent>
<modelVersion>4.0.0</modelVersion>
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnalyzeRequest {
private byte[] document;
private String ruleSetId;
private ManualRedactions manualRedactions;
}
@@ -0,0 +1,19 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnalyzeResult {
private int numberOfPages;
private RedactionLog redactionLog;
private SectionGrid sectionGrid;
private Text text;
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnnotateRequest {
private byte[] document;
private RedactionLog redactionLog;
private SectionGrid sectionGrid;
}
@@ -0,0 +1,15 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class AnnotateResponse {
private byte[] document;
}
@@ -0,0 +1,19 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualForceRedact {
private String id;
private String user;
private Status status;
private String legalBasis;
}
@@ -1,5 +1,5 @@
package com.iqser.red.service.redaction.v1.model;
public enum ManualRedactionType {
ADD, REMOVE
ADD, REMOVE, FORCE_REDACT
}
@@ -20,6 +20,9 @@ public class ManualRedactions {
@Builder.Default
private Set<IdRemoval> idsToRemove = new HashSet<>();
@Builder.Default
private Set<ManualForceRedact> forceRedacts = new HashSet<>();
@Builder.Default
private Set<ManualRedactionEntry> entriesToAdd = new HashSet<>();
@@ -0,0 +1,15 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class ReanalyzeResult {
private RedactionLog redactionLog;
}
@@ -37,4 +37,10 @@ public class RedactionLogEntry {
private String textBefore;
private String textAfter;
@Builder.Default
private List<Comment> comments = new ArrayList<>();
private int startOffset;
private int endOffset;
}
@@ -13,6 +13,5 @@ public class RedactionRequest {
private byte[] document;
private String ruleSetId;
private boolean flatRedaction;
private ManualRedactions manualRedactions;
}
@@ -0,0 +1,22 @@
package com.iqser.red.service.redaction.v1.model;
import java.time.OffsetDateTime;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class RenalyzeRequest {
private byte[] document;
private String ruleSetId;
private ManualRedactions manualRedactions;
private Text text;
private RedactionLog redactionLog;
private OffsetDateTime lastProcessed;
}
@@ -0,0 +1,33 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
import lombok.NonNull;
import lombok.RequiredArgsConstructor;
@Data
@RequiredArgsConstructor
@NoArgsConstructor
@AllArgsConstructor
public class SectionArea {
@NonNull
private Point topLeft;
@NonNull
private float width;
@NonNull
private float height;
@NonNull
private int page;
private String header;
public boolean contains(Rectangle other) {
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft().getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft().getY() + other.getHeight();
}
}
@@ -0,0 +1,25 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@NoArgsConstructor
@AllArgsConstructor
public class SectionText {
private int sectionNumber;
private String text;
private boolean isTable;
private String headline;
private List<SectionArea> sectionAreas = new ArrayList<>();
}
@@ -0,0 +1,17 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
@AllArgsConstructor
public class Text {
private List<SectionText> sectionTexts = new ArrayList<>();
}
@@ -1,7 +1,14 @@
package com.iqser.red.service.redaction.v1.resources;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.AnnotateRequest;
import com.iqser.red.service.redaction.v1.model.AnnotateResponse;
import com.iqser.red.service.redaction.v1.model.ReanalyzeResult;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.RenalyzeRequest;
import org.springframework.http.MediaType;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
@@ -14,8 +21,14 @@ public interface RedactionResource {
String RULE_SET_PARAMETER_NAME = "ruleSetId";
String RULE_SET_PATH_VARIABLE = "/{" + RULE_SET_PARAMETER_NAME + "}";
@PostMapping(value = "/redact", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult redact(@RequestBody RedactionRequest redactionRequest);
@PostMapping(value = "/analyze", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
AnalyzeResult analyze(@RequestBody AnalyzeRequest analyzeRequest);
@PostMapping(value = "/reanalyze", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
ReanalyzeResult reanalyze(@RequestBody RenalyzeRequest renalyzeRequest);
@PostMapping(value = "/annotate", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest);
@PostMapping(value = "/debug/classifications", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
RedactionResult classify(@RequestBody RedactionRequest redactionRequest);
@@ -8,6 +8,7 @@ import java.util.Map;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.model.SectionText;
import lombok.Data;
import lombok.NoArgsConstructor;
@@ -18,6 +19,9 @@ public class Document {
private List<Page> pages = new ArrayList<>();
private List<Paragraph> paragraphs = new ArrayList<>();
private List<Header> headers = new ArrayList<>();
private List<Footer> footers = new ArrayList<>();
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
private Map<Integer, List<Entity>> entities = new HashMap<>();
private FloatFrequencyCounter textHeightCounter = new FloatFrequencyCounter();
private FloatFrequencyCounter fontSizeCounter= new FloatFrequencyCounter();
@@ -29,4 +33,6 @@ public class Document {
private SectionGrid sectionGrid = new SectionGrid();
private long dictionaryVersion;
private long rulesVersion;
private List<SectionText> sectionText = new ArrayList<>();
}
@@ -0,0 +1,24 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class Footer {
private List<TextBlock> textBlocks;
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -0,0 +1,24 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class Header {
private List<TextBlock> textBlocks;
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.awt.geom.Rectangle2D;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
@@ -16,6 +17,8 @@ public class Page {
@NonNull
private List<AbstractTextContainer> textBlocks;
private List<Rectangle2D> imageBounds;
private Rectangle bodyTextFrame;
private boolean landscape;
@@ -0,0 +1,24 @@
package com.iqser.red.service.redaction.v1.server.classification.model;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class UnclassifiedText {
private List<TextBlock> textBlocks;
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> searchableText.addAll(block.getSequences()));
return searchableText;
}
}
@@ -111,6 +111,8 @@ public class ClassificationService {
textBlock.setClassification("TextBlock Italic");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock)) {
textBlock.setClassification("TextBlock Unknown");
} else {
textBlock.setClassification("Other");
}
}
@@ -1,23 +1,35 @@
package com.iqser.red.service.redaction.v1.server.controller;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.AnnotateRequest;
import com.iqser.red.service.redaction.v1.model.AnnotateResponse;
import com.iqser.red.service.redaction.v1.model.ReanalyzeResult;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.RenalyzeRequest;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.model.Text;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.AnnotationService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.EntityRedactionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.ReanalyzeService;
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogCreatorService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import com.iqser.red.service.redaction.v1.server.visualization.service.AnnotationHighlightService;
import com.iqser.red.service.redaction.v1.server.visualization.service.PdfVisualisationService;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.RequestBody;
@@ -35,34 +47,37 @@ public class RedactionController implements RedactionResource {
private final PdfVisualisationService pdfVisualisationService;
private final PdfSegmentationService pdfSegmentationService;
private final AnnotationHighlightService annotationHighlightService;
private final RedactionLogCreatorService redactionLogCreatorService;
private final EntityRedactionService entityRedactionService;
private final DroolsExecutionService droolsExecutionService;
private final DictionaryService dictionaryService;
private final AnnotationService annotationService;
private final ReanalyzeService reanalyzeService;
@Override
public RedactionResult redact(@RequestBody RedactionRequest redactionRequest) {
public AnalyzeResult analyze(@RequestBody AnalyzeRequest analyzeRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(analyzeRequest.getDocument()))) {
pdDocument.setAllSecurityToBeRemoved(true);
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
log.info("Document structure analysis successful, starting redaction analysis...");
entityRedactionService.processDocument(classifiedDoc, redactionRequest.getRuleSetId(), redactionRequest.getManualRedactions());
annotationHighlightService.highlight(pdDocument, classifiedDoc, redactionRequest.isFlatRedaction(), redactionRequest
.getManualRedactions(), redactionRequest.getRuleSetId());
entityRedactionService.processDocument(classifiedDoc, analyzeRequest.getRuleSetId(), analyzeRequest.getManualRedactions());
redactionLogCreatorService.createRedactionLog(classifiedDoc, pdDocument.getNumberOfPages(), analyzeRequest.getManualRedactions(), analyzeRequest
.getRuleSetId());
log.info("Redaction analysis successful...");
return convert(pdDocument,
classifiedDoc.getPages().size(),
classifiedDoc.getRedactionLogEntities(),
classifiedDoc.getSectionGrid(),
classifiedDoc.getDictionaryVersion(),
classifiedDoc.getRulesVersion(),
redactionRequest.getRuleSetId());
return AnalyzeResult.builder()
.sectionGrid(classifiedDoc.getSectionGrid())
.redactionLog(new RedactionLog(classifiedDoc.getRedactionLogEntities(), classifiedDoc.getDictionaryVersion(), classifiedDoc
.getRulesVersion(), analyzeRequest.getRuleSetId()))
.numberOfPages(classifiedDoc.getPages().size())
.text(new Text(classifiedDoc.getSectionText()))
.build();
} catch (Exception e) {
throw new RedactionException(e);
@@ -71,6 +86,31 @@ public class RedactionController implements RedactionResource {
}
public ReanalyzeResult reanalyze(@RequestBody RenalyzeRequest renalyzeRequest) {
return reanalyzeService.reanalyze(renalyzeRequest);
}
public AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest) {
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(annotateRequest.getDocument()))) {
pdDocument.setAllSecurityToBeRemoved(true);
dictionaryService.updateDictionary(annotateRequest.getRedactionLog().getRuleSetId());
annotationService.annotate(pdDocument, annotateRequest.getRedactionLog(), annotateRequest.getSectionGrid());
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
pdDocument.save(byteArrayOutputStream);
return AnnotateResponse.builder().document(byteArrayOutputStream.toByteArray()).build();
}
} catch (Exception e) {
throw new RedactionException(e);
}
}
@Override
public RedactionResult classify(@RequestBody RedactionRequest pdfSegmentationRequest) {
@@ -136,11 +176,14 @@ public class RedactionController implements RedactionResource {
@Override
public void updateRules(@PathVariable(RULE_SET_PARAMETER_NAME) String ruleSetId) {
droolsExecutionService.updateRules(ruleSetId);
}
@Override
public void testRules(@RequestBody String rules) {
droolsExecutionService.testRules(rules);
}
@@ -152,8 +195,8 @@ public class RedactionController implements RedactionResource {
private RedactionResult convert(PDDocument document, int numberOfPages,
List<RedactionLogEntry> redactionLogEntities,
SectionGrid sectionGrid, long dictionaryVersion, long rulesVersion, String ruleSetId) throws IOException {
List<RedactionLogEntry> redactionLogEntities, SectionGrid sectionGrid,
long dictionaryVersion, long rulesVersion, String ruleSetId) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
@@ -0,0 +1,83 @@
package com.iqser.red.service.redaction.v1.server.parsing;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import org.apache.pdfbox.text.PDFTextStripperByArea;
import org.apache.pdfbox.text.TextPosition;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.Getter;
import lombok.Setter;
public class PDFAreaTextStripper extends PDFTextStripperByArea {
@Getter
private List<TextPositionSequence> textPositionSequences = new ArrayList<>();
@Setter
private int pageNumber;
public PDFAreaTextStripper() throws IOException {
}
@Override
public void writeString(String text, List<TextPosition> textPositions) throws IOException {
int startIndex = 0;
for (int i = 0; i <= textPositions.size() - 1; i++) {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0"))) {
startIndex++;
continue;
}
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0")) && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i + 1;
}
}
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
.getUnicode()
.equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
.getUnicode()
.equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
super.writeString(text);
}
public void clearPositions(){
textPositionSequences = new ArrayList<>();
}
}
@@ -1,6 +1,8 @@
package com.iqser.red.service.redaction.v1.server.parsing;
import java.awt.geom.AffineTransform;
import java.awt.geom.Point2D;
import java.awt.geom.Rectangle2D;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
@@ -28,10 +30,15 @@ import org.apache.pdfbox.contentstream.operator.state.SetLineWidth;
import org.apache.pdfbox.contentstream.operator.state.SetRenderingIntent;
import org.apache.pdfbox.contentstream.operator.text.SetFontAndSize;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.cos.COSNumber;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.text.PDFTextStripper;
import org.apache.pdfbox.text.TextPosition;
import org.apache.pdfbox.util.Matrix;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
@@ -43,8 +50,17 @@ import lombok.extern.slf4j.Slf4j;
@Slf4j
public class PDFLinesTextStripper extends PDFTextStripper {
@Setter
protected PDPage pdpage;
@Getter
private int maxCharWidths;
private int minCharWidth;
@Getter
private int maxCharWidth;
@Getter
private int minCharHeight;
@Getter
private int maxCharHeight;
@@ -57,13 +73,18 @@ public class PDFLinesTextStripper extends PDFTextStripper {
private final List<Ruling> graphicsPath = new ArrayList<>();
@Getter
private List<Rectangle2D> imageBounds = new ArrayList<>();
private float path_x;
private float path_y;
@Setter
private int pageNumber;
public PDFLinesTextStripper() throws IOException {
super();
this.addOperator(new SetStrokingColorSpace());
this.addOperator(new SetNonStrokingColorSpace());
@@ -87,9 +108,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
this.addOperator(new SetLineWidth());
}
@Override
protected void processOperator(Operator operator, List<COSBase> arguments)
throws IOException {
protected void processOperator(Operator operator, List<COSBase> arguments) throws IOException {
String operation = operator.getName();
@@ -110,9 +131,11 @@ public class PDFLinesTextStripper extends PDFTextStripper {
// The direction of vertical lines must always be from bottom to top for the table extraction algorithm.
if (pos.getY() > path_y) {
graphicsPath.add(new Ruling(new Point2D.Float(path_x, path_y), new Point2D.Float((float) pos.getX(), (float) pos.getY())));
graphicsPath.add(new Ruling(new Point2D.Float(path_x, path_y), new Point2D.Float((float) pos.getX(), (float) pos
.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float(path_x, (float) pos.getY()), new Point2D.Float((float) pos.getX(), path_y)));
graphicsPath.add(new Ruling(new Point2D.Float(path_x, (float) pos.getY()), new Point2D.Float((float) pos
.getX(), path_y)));
}
path_x = (float) pos.getX();
@@ -133,19 +156,25 @@ public class PDFLinesTextStripper extends PDFTextStripper {
Point2D p2 = transformPosition(x + width, y + height);
// Horizontal lines
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p2.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p2.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p2
.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p2
.getX(), (float) p2.getY())));
// Vertical lines, direction must always be from bottom to top for the table extraction algorithm.
if (p2.getY() > p1.getY()) {
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p1.getY()), new Point2D.Float((float) p2.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p1.getY()), new Point2D.Float((float) p2
.getX(), (float) p2.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p2.getY()), new Point2D.Float((float) p2.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p2.getX(), (float) p2.getY()), new Point2D.Float((float) p2
.getX(), (float) p1.getY())));
}
if (p2.getY() > p1.getY()) {
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p1.getX(), (float) p2.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p1.getY()), new Point2D.Float((float) p1
.getX(), (float) p2.getY())));
} else {
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p1.getX(), (float) p1.getY())));
graphicsPath.add(new Ruling(new Point2D.Float((float) p1.getX(), (float) p2.getY()), new Point2D.Float((float) p1
.getX(), (float) p1.getY())));
}
}
break;
@@ -168,12 +197,59 @@ public class PDFLinesTextStripper extends PDFTextStripper {
case OperatorName.ENDPATH:
graphicsPath.clear();
break;
case OperatorName.DRAW_OBJECT:
processImageOperation(arguments);
break;
}
super.processOperator(operator, arguments);
}
protected void processImageOperation(List<COSBase> arguments) {
try {
COSName objectName = (COSName) arguments.get(0);
PDXObject xobject = getResources().getXObject(objectName);
if (xobject instanceof PDImageXObject) {
PDImageXObject pdfImage = (PDImageXObject) xobject;
Rectangle2D imageBounds = calculateImagePosition(pdfImage);
Rectangle2D rect = new Rectangle2D.Float((float) imageBounds.getX(), (float) imageBounds.getY(), (float) imageBounds
.getWidth(), (float) imageBounds.getHeight());
if (rect.getHeight() > 2 && rect.getWidth() > 2) {
this.imageBounds.add(rect);
}
}
} catch (Exception e) {
log.warn("Problem during image extraction: {}", e.getMessage());
}
}
private Rectangle2D calculateImagePosition(PDImageXObject pdfImage) throws IOException {
Matrix ctm = getGraphicsState().getCurrentTransformationMatrix();
Rectangle2D imageBounds = pdfImage.getImage().getRaster().getBounds();
AffineTransform imageTransform = new AffineTransform(ctm.createAffineTransform());
imageTransform.scale(1.0 / pdfImage.getWidth(), -1.0 / pdfImage.getHeight());
imageTransform.translate(0, -pdfImage.getHeight());
AffineTransform pageTransform = new AffineTransform();
pageTransform.concatenate(imageTransform);
return pageTransform.createTransformedShape(imageBounds).getBounds2D();
}
private float floatValue(COSBase value) {
if (value instanceof COSNumber) {
return ((COSNumber) value).floatValue();
} else {
@@ -181,38 +257,56 @@ public class PDFLinesTextStripper extends PDFTextStripper {
}
}
private Point2D.Float transformPosition(float x, float y) {
return super.transformedPoint(x, y);
}
private void addVisibleRulings(List<Ruling> path, boolean stroke) throws IOException {
try {
if (stroke && !getGraphicsState().getStrokingColor().isPattern() && getGraphicsState().getStrokingColor().toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor().isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
if (stroke && !getGraphicsState().getStrokingColor().isPattern() && getGraphicsState().getStrokingColor()
.toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor()
.isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
rulings.addAll(path);
}
} catch (UnsupportedOperationException e) {
log.error("UnsupportedOperationException: " + getGraphicsState().getStrokingColor().getColorSpace().getName() + " or " + getGraphicsState().getNonStrokingColor().getColorSpace().getName() + " does not support toRGB");
log.error("UnsupportedOperationException: " + getGraphicsState().getStrokingColor()
.getColorSpace()
.getName() + " or " + getGraphicsState().getNonStrokingColor()
.getColorSpace()
.getName() + " does not support toRGB");
}
}
@Override
public void writeString(String text, List<TextPosition> textPositions) throws IOException {
int startIndex = 0;
for (int i = 0; i <= textPositions.size() - 1; i++) {
int charWidth = (int) textPositions.get(i).getWidthDirAdj();
if (charWidth < minCharWidth) {
minCharWidth = charWidth;
}
if (charWidth > maxCharWidth) {
maxCharWidth = charWidth;
}
int charHeight = (int) textPositions.get(i).getHeightDir();
if(charHeight > maxCharHeight){
if (charHeight < minCharHeight) {
minCharHeight = charHeight;
}
if (charWidth > maxCharHeight) {
maxCharHeight = charHeight;
}
int charWidth = (int) textPositions.get(i).getWidthDirAdj();
if(charWidth > maxCharWidths){
maxCharWidths = charWidth;
}
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0"))) {
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0"))) {
startIndex++;
continue;
}
@@ -220,15 +314,21 @@ public class PDFLinesTextStripper extends PDFTextStripper {
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i;
}
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0")) && i <= textPositions.size() - 2) {
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
.getUnicode()
.equals("\u00A0")) && i <= textPositions.size() - 2) {
List<TextPosition> sublist = textPositions.subList(startIndex, i);
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
.getUnicode()
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
startIndex = i + 1;
@@ -236,21 +336,29 @@ public class PDFLinesTextStripper extends PDFTextStripper {
}
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
.getUnicode()
.equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
sublist = sublist.subList(0, sublist.size() - 1);
}
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
.getUnicode()
.equals("\u00A0")))) {
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
}
super.writeString(text);
}
@Override
public String getText(PDDocument doc) throws IOException {
maxCharWidths = 0;
maxCharWidths = 0;
minCharWidth = Integer.MAX_VALUE;
maxCharWidth = 0;
minCharHeight = Integer.MAX_VALUE;
maxCharHeight = 0;
textPositionSequences.clear();
imageBounds = new ArrayList<>();
rulings.clear();
graphicsPath.clear();
path_x = 0.0f;
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server.parsing.model;
import java.awt.geom.Rectangle2D;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
@@ -13,10 +14,11 @@ public class ParsedElements {
private List<TextPositionSequence> sequences;
private List<Ruling> rulings;
private List<Rectangle2D> imageBounds;
private boolean landscape;
private boolean rotated;
private float minCharWidth;
private float maxCharWidth;
private float maxCharHeight;
}
@@ -28,6 +28,14 @@ public class Dictionary {
}
public int getDictionaryRank(String type){
if(!localAccessMap.containsKey(type)){
return 0;
}
return localAccessMap.get(type).getRank();
}
public boolean isRecommendation(String type) {
DictionaryModel model = localAccessMap.get(type);
@@ -0,0 +1,15 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Data;
@Data
@AllArgsConstructor
public class DictionaryIncrement {
private Set<String> values;
private long dictionaryVersion;
}
@@ -39,7 +39,7 @@ public class Entity {
private String textAfter;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber, String legalBasis, boolean isDictionaryEntry, String textBefore, String textAfter) {
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber, String legalBasis, boolean isDictionaryEntry, String textBefore, String textAfter, Integer start, Integer end) {
this.word = word;
this.type = type;
@@ -53,6 +53,8 @@ public class Entity {
this.isDictionaryEntry = isDictionaryEntry;
this.textBefore = textBefore;
this.textAfter = textAfter;
this.start = start;
this.end = end;
}
@@ -0,0 +1,34 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@NoArgsConstructor
public class ReanalysisSection {
private int sectionNumber;
private String headline;
private List<TextBlock> textBlocks;
private Map<String, CellValue> tabularData = new HashMap<>();
private List<Integer> cellStarts;
public SearchableText getSearchableText() {
SearchableText searchableText = new SearchableText();
textBlocks.forEach(block -> {
if (block instanceof TextBlock) {
searchableText.addAll(block.getSequences());
}
});
return searchableText;
}
}
@@ -32,6 +32,7 @@ public class SearchableText {
}
@SuppressWarnings("checkstyle:ModifiedControlVariable")
public List<EntityPositionSequence> getSequences(String searchString, boolean caseInsensitive,
List<TextPositionSequence> sequencesSubList) {
@@ -66,9 +67,12 @@ public class SearchableText {
for (int j = 0; j < searchSpace.get(i).length(); j++) {
if (i > 0 && j == 0 && searchSpace.get(i).charAt(0, caseInsensitive) == ' ' && searchSpace.get(i - 1)
.charAt(searchSpace.get(i - 1).length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i).charAt(j - 1, caseInsensitive) == ' ') {
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions().isEmpty()) {
.charAt(searchSpace.get(i - 1)
.length() - 1, caseInsensitive) == ' ' || j > 0 && searchSpace.get(i)
.charAt(j, caseInsensitive) == ' ' && searchSpace.get(i)
.charAt(j - 1, caseInsensitive) == ' ') {
if (j == searchSpace.get(i).length() - 1 && counter != 0 && !partMatch.getTextPositions()
.isEmpty()) {
crossSequenceParts.add(partMatch);
}
continue;
@@ -80,8 +84,8 @@ public class SearchableText {
counter++;
}
if (searchSpace.get(i)
.charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace.get(i)
if (searchSpace.get(i).charAt(j, caseInsensitive) == searchChars[counter] || counter != 0 && searchSpace
.get(i)
.charAt(j, caseInsensitive) == '-') {
if (counter != 0 || i == 0 && j == 0 || j != 0 && isSeparator(searchSpace.get(i)
@@ -100,14 +104,15 @@ public class SearchableText {
if (counter == searchString.length()) {
crossSequenceParts.add(partMatch);
if (i == searchSpace.size() - 1 && j == searchSpace.get(i).length() - 1 || j != searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i)
if (i == searchSpace.size() - 1 && j == searchSpace.get(i)
.length() - 1 || j != searchSpace.get(i).length() - 1 && isSeparator(searchSpace.get(i)
.charAt(j + 1, caseInsensitive)) || j == searchSpace.get(i)
.length() - 1 && isSeparator(searchSpace.get(i + 1)
.charAt(0, caseInsensitive)) || j == searchSpace.get(i).length() - 1 && searchSpace.get(i)
.charAt(0, caseInsensitive)) || j == searchSpace.get(i)
.length() - 1 && searchSpace.get(i)
.charAt(j, caseInsensitive) != ' ' && searchSpace.get(i + 1)
.charAt(0, caseInsensitive) != ' ') {
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts));
finalMatches.addAll(buildEntityPositionSequence(crossSequenceParts, normalizedSearchString));
}
counter = 0;
@@ -130,15 +135,21 @@ public class SearchableText {
}
return finalMatches;
}
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts,
String searchString) {
List<EntityPositionSequence> result = new ArrayList<>();
String asString = buildString(crossSequenceParts);
if (!asString.equalsIgnoreCase(searchString)) {
return result;
}
String plainId = IdBuilder.buildId(crossSequenceParts);
String id = plainId;
List<EntityPositionSequence> result = new ArrayList<>();
int currentPage = -1;
int idDiffentPageSuffix = 1;
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
@@ -173,6 +184,12 @@ public class SearchableText {
@Override
public String toString() {
return buildString(sequences);
}
public String buildString(List<TextPositionSequence> sequences) {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
@@ -62,6 +62,13 @@ public class Section {
}
public boolean hasTableHeader(String headerName) {
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
return tabularData != null && tabularData.containsKey(cleanHeaderName);
}
public boolean matchesType(String type) {
return entities.stream().anyMatch(entity -> entity.getType().equals(type));
@@ -74,6 +81,34 @@ public class Section {
}
public void expandByRegEx(String type, String pattern, boolean patternCaseInsensitive, int group) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Set<Entity> expanded = new HashSet<>();
for (Entity entity : entities) {
if (!entity.getType().equals(type) || entity.getTextAfter() == null) {
continue;
}
Matcher matcher = compiledPattern.matcher(entity.getTextAfter());
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
expanded.addAll(findEntities(entity.getWord() + match, type, false, entity.isRedaction(), entity.getMatchedRule(), entity
.getRedactionReason(), entity.getLegalBasis()));
}
}
}
EntitySearchUtils.addEntitiesWithHigherRank(entities, expanded, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
}
public void redact(String type, int ruleNumber, String reason, String legalBasis) {
boolean hasRecommendationDictionary = dictionaryTypes.contains(RECOMMENDATION_PREFIX + type);
@@ -105,6 +140,48 @@ public class Section {
}
public void expandToHintAnnotationByRegEx(String type, String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Set<Entity> expanded = new HashSet<>();
for (Entity entity : entities) {
if (!entity.getType().equals(type) || entity.getTextAfter() == null) {
continue;
}
Matcher matcher = compiledPattern.matcher(entity.getTextAfter());
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
expanded.addAll(findEntities(entity.getWord() + match, asType, false, false, 0, null, null));
}
}
}
EntitySearchUtils.addEntitiesWithHigherRank(entities, expanded, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
}
public void addHintAnnotationByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, false, 0, null, null);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
}
}
}
public void redactIfPrecededBy(String prefix, String type, int ruleNumber, String reason, String legalBasis) {
entities.forEach(entity -> {
@@ -121,7 +198,14 @@ public class Section {
public void addHintAnnotation(String value, String asType) {
Set<Entity> found = findEntities(value.trim(), asType, true, false, 0, null, null);
addNewerToEntities(found);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
}
public void addRedaction(String value, String asType, int ruleNumber, String reason, String legalBasis) {
Set<Entity> found = findEntities(value.trim(), asType, true, true, ruleNumber, reason, legalBasis);
EntitySearchUtils.addEntitiesIgnoreRank(entities, found);
}
@@ -134,14 +218,44 @@ public class Section {
for (String value : values) {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
}
}
}
}
}
public void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis) {
public void recommendLineAfter(String start, String asType) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
if (values != null) {
for (String value : values) {
String trimmedValue = value.trim();
String cleanValue;
if (trimmedValue.startsWith(":")) {
cleanValue = trimmedValue.substring(1).trim();
} else {
cleanValue = trimmedValue;
}
if (StringUtils.isNotBlank(cleanValue) && cleanValue.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>())
.add(cleanValue);
}
}
}
}
public void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber,
String reason, String legalBasis) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
@@ -150,13 +264,14 @@ public class Section {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match)) {
Set<Entity> found = findEntities(match.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
}
}
}
public void addRecommendationByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(text);
@@ -164,8 +279,24 @@ public class Section {
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>())
.add(match);
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>()).add(match);
}
}
}
public void redactAndRecommendByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType,
int ruleNumber, String reason, String legalBasis) {
Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive);
Matcher matcher = compiledPattern.matcher(searchText);
while (matcher.find()) {
String match = matcher.group(group);
if (StringUtils.isNotBlank(match) && match.length() >= 3) {
localDictionaryAdds.computeIfAbsent(RECOMMENDATION_PREFIX + asType, (x) -> new HashSet<>()).add(match);
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(match);
}
}
}
@@ -181,7 +312,7 @@ public class Section {
if (StringUtils.isNotBlank(value)) {
Set<Entity> found = findEntities(value.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(value.trim());
@@ -192,8 +323,7 @@ public class Section {
}
public void redactLinesBetween(String start, String stop, String asType, int ruleNumber,
boolean redactEverywhere,
public void redactLinesBetween(String start, String stop, String asType, int ruleNumber, boolean redactEverywhere,
String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(text, start, stop);
@@ -210,7 +340,7 @@ public class Section {
}
Set<Entity> found = findEntities(line.trim(), asType, false, true, ruleNumber, reason, legalBasis);
addNewerToEntities(found);
EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary);
if (redactEverywhere && !isLocal()) {
localDictionaryAdds.computeIfAbsent(asType, (x) -> new HashSet<>()).add(line.trim());
@@ -228,8 +358,7 @@ public class Section {
}
public void redactCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations, String
reason,
public void redactCell(String cellHeader, int ruleNumber, String type, boolean addAsRecommendations, String reason,
String legalBasis) {
annotateCell(cellHeader, ruleNumber, type, true, addAsRecommendations, reason, legalBasis);
@@ -290,7 +419,7 @@ public class Section {
singleEntitySet.add(entity);
EntitySearchUtils.clearAndFindPositions(singleEntitySet, searchableText, dictionary);
addNewerToEntities(entity);
EntitySearchUtils.addEntitiesWithHigherRank(entities, entity, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(entities);
@@ -313,19 +442,6 @@ public class Section {
}
}
private void addNewerToEntities(Set<Entity> found) {
// HashSet keeps the older value, but we want the new only.
entities.removeAll(found);
entities.addAll(found);
}
private void addNewerToEntities(Entity found) {
// HashSet keeps the older value, but we want the new only.
entities.remove(found);
entities.add(found);
}
}
@@ -0,0 +1,268 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.awt.Color;
import java.io.IOException;
import java.util.ArrayList;
import java.util.GregorianCalendar;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.stream.Collectors;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.graphics.color.PDColor;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationText;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.CellRectangle;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.SectionGrid;
import com.iqser.red.service.redaction.v1.model.SectionRectangle;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class AnnotationService {
private final DictionaryService dictionaryService;
public void annotate(PDDocument document, RedactionLog redactionLog, SectionGrid sectionGrid) throws IOException {
Map<Integer, List<RedactionLogEntry>> redactionLogPerPage = convertRedactionLog(redactionLog);
for (int page = 1; page <= document.getNumberOfPages(); page++) {
PDPage pdPage = document.getPage(page - 1);
List<SectionRectangle> sectionRectangles = sectionGrid.getRectanglesPerPage().get(page);
if (sectionRectangles != null && !sectionRectangles.isEmpty()) {
drawSectionGrid(document, pdPage, sectionRectangles);
}
List<RedactionLogEntry> logEntries = redactionLogPerPage.get(page);
if (logEntries != null && !logEntries.isEmpty()) {
addAnnotations(logEntries, pdPage, page);
}
}
}
private void addAnnotations(List<RedactionLogEntry> logEntries, PDPage pdPage, int page) throws IOException {
List<PDAnnotation> annotations = pdPage.getAnnotations();
for (RedactionLogEntry entry : logEntries) {
annotations.addAll(createAnnotation(entry, page, pdPage.getMediaBox(), pdPage.getCropBox()));
}
}
private List<PDAnnotation> createAnnotation(RedactionLogEntry redactionLogEntry, int page, PDRectangle mediaBox,
PDRectangle cropBox) {
List<PDAnnotation> annotations = new ArrayList<>();
List<Rectangle> rectangles = redactionLogEntry.getPositions()
.stream()
.filter(pos -> pos.getPage() == page)
.collect(Collectors.toList());
if (rectangles.isEmpty()) {
return annotations;
}
PDAnnotationTextMarkup annotation = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
annotation.constructAppearances();
PDRectangle pdRectangle = toPDRectangle(rectangles, mediaBox, cropBox);
annotation.setRectangle(pdRectangle);
annotation.setQuadPoints(toQuadPoints(rectangles, mediaBox, cropBox));
if (!redactionLogEntry.isHint()) {
annotation.setContents(createAnnotationContent(redactionLogEntry));
}
annotation.setTitlePopup(redactionLogEntry.getId());
annotation.setAnnotationName(redactionLogEntry.getId());
annotation.setColor(new PDColor(redactionLogEntry.getColor(), PDDeviceRGB.INSTANCE));
annotations.add(annotation);
if (redactionLogEntry.getComments() != null) {
for (Comment comment : redactionLogEntry.getComments()) {
PDAnnotationText txtAnnot = new PDAnnotationText();
txtAnnot.setAnnotationName(comment.getId());
txtAnnot.setInReplyTo(annotation); // Reference to highlight annotation
txtAnnot.setName(PDAnnotationText.NAME_COMMENT);
txtAnnot.setCreationDate(GregorianCalendar.from(comment.getDate().toZonedDateTime()));
txtAnnot.setTitlePopup(comment.getUser());
txtAnnot.setContents(comment.getText());
txtAnnot.setRectangle(pdRectangle);
annotations.add(txtAnnot);
}
}
return annotations;
}
private String createAnnotationContent(RedactionLogEntry redactionLogEntry) {
if (redactionLogEntry.isManual()) {
return "\nManual Redaction\n\nIn Section : \"" + redactionLogEntry.getSection() + "\"";
}
return "\nRule " + redactionLogEntry.getMatchedRule() + " matched\n\n" + redactionLogEntry.getReason() + "\n\nLegal basis:" + redactionLogEntry
.getLegalBasis() + "\n\nIn section: \"" + redactionLogEntry.getSection() + "\"";
}
private PDRectangle toPDRectangle(List<Rectangle> rectangles, PDRectangle mediaBox, PDRectangle cropBox) {
float lowerLeftX = Float.MAX_VALUE;
float upperRightX = 0;
float lowerLeftY = 0;
float upperRightY = Float.MAX_VALUE;
for (Rectangle rectangle : rectangles) {
if (rectangle.getTopLeft().getX() < lowerLeftX) {
lowerLeftX = rectangle.getTopLeft().getX();
}
if (rectangle.getTopLeft().getX() + rectangle.getWidth() > upperRightX) {
upperRightX = rectangle.getTopLeft().getX() + rectangle.getWidth();
}
if (rectangle.getTopLeft().getY() + rectangle.getHeight() > lowerLeftY) {
lowerLeftY = rectangle.getTopLeft().getY() + rectangle.getHeight();
}
if (rectangle.getTopLeft().getY() < upperRightY) {
upperRightY = rectangle.getTopLeft().getY();
}
}
var x1 = lowerLeftX + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
var y1 = lowerLeftY + (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
var x2 = upperRightX + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
var y2 = upperRightY - (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(x1);
annotationPosition.setLowerLeftY(y1);
annotationPosition.setUpperRightX(x2);
annotationPosition.setUpperRightY(y2);
return annotationPosition;
}
private float[] toQuadPoints(List<Rectangle> rectangles, PDRectangle mediaBox, PDRectangle cropBox) {
float[] quadPoints = new float[rectangles.size() * 8];
int i = 0;
for (Rectangle rectangle : rectangles) {
float[] quadPoint = toQuadPoint(rectangle, mediaBox, cropBox);
for (int j = 0; j <= 7; j++) {
quadPoints[i + j] = quadPoint[j];
}
i += 8;
}
return quadPoints;
}
private float[] toQuadPoint(Rectangle rectangle, PDRectangle mediaBox, PDRectangle cropBox) {
var x1 = rectangle.getTopLeft().getX() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
var y1 = rectangle.getTopLeft().getY() + (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
var x2 = rectangle.getTopLeft()
.getX() + rectangle.getWidth() + cropBox.getLowerLeftX() - mediaBox.getLowerLeftY();
var y2 = rectangle.getTopLeft()
.getY() + rectangle.getHeight() - (mediaBox.getLowerLeftY() - cropBox.getLowerLeftY());
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
return new float[]{x1, y1, x2, y2, x1, y2 - rectangle.getHeight(), x2, y1 - rectangle.getHeight()};
}
private void drawSectionGrid(PDDocument document, PDPage pdPage,
List<SectionRectangle> sectionRectangles) throws IOException {
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
for (SectionRectangle sectionRectangle : sectionRectangles) {
drawSectionRectangle(contentStream, sectionRectangle);
drawSectionPartNumberText(contentStream, sectionRectangle);
drawTableCells(contentStream, sectionRectangle);
}
contentStream.close();
}
private void drawSectionRectangle(PDPageContentStream contentStream,
SectionRectangle sectionRectangle) throws IOException {
contentStream.setStrokingColor(Color.LIGHT_GRAY);
contentStream.setLineWidth(0.5f);
contentStream.addRect(sectionRectangle.getTopLeft().getX(), sectionRectangle.getTopLeft()
.getY(), sectionRectangle.getWidth(), sectionRectangle.getHeight());
contentStream.stroke();
}
private void drawSectionPartNumberText(PDPageContentStream contentStream,
SectionRectangle sectionRectangle) throws IOException {
contentStream.beginText();
contentStream.setNonStrokingColor(Color.DARK_GRAY);
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
if (sectionRectangle.getTableCells() == null) {
contentStream.newLineAtOffset(sectionRectangle.getTopLeft().getX(), sectionRectangle.getTopLeft()
.getY() + sectionRectangle.getHeight());
} else {
contentStream.newLineAtOffset(sectionRectangle.getTopLeft().getX(), sectionRectangle.getTopLeft().getY());
}
contentStream.showText(sectionRectangle.getPart() + "/" + sectionRectangle.getNumberOfParts());
contentStream.endText();
}
private void drawTableCells(PDPageContentStream contentStream,
SectionRectangle sectionRectangle) throws IOException {
if (sectionRectangle.getTableCells() != null) {
for (CellRectangle cell : sectionRectangle.getTableCells()) {
contentStream.setLineWidth(0.5f);
contentStream.setStrokingColor(Color.CYAN);
contentStream.addRect(cell.getTopLeft().getX(), cell.getTopLeft()
.getY(), cell.getWidth(), cell.getHeight());
contentStream.stroke();
}
}
}
private Map<Integer, List<RedactionLogEntry>> convertRedactionLog(RedactionLog redactionLog) {
Map<Integer, List<RedactionLogEntry>> redactionLogPerPage = new HashMap<>();
if (redactionLog == null) {
return redactionLogPerPage;
}
for (RedactionLogEntry entry : redactionLog.getRedactionLogEntry()) {
int page = 0;
for (Rectangle position : entry.getPositions()) {
if (position.getPage() != page) {
redactionLogPerPage.computeIfAbsent(position.getPage(), x -> new ArrayList<>()).add(entry);
page = position.getPage();
}
}
}
return redactionLogPerPage;
}
}
@@ -5,6 +5,7 @@ import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryIncrement;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryRepresentation;
import feign.FeignException;
@@ -48,6 +49,14 @@ public class DictionaryService {
}
public DictionaryIncrement getDictionaryIncrements(String ruleSetId, long fromVersion){
updateDictionary(ruleSetId);
return new DictionaryIncrement(Set.of("Properties"), 1L);
}
private void updateDictionaryEntry(String ruleSetId, long version) {
try {
@@ -1,11 +1,11 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.HashMap;
import java.util.Map;
import org.apache.commons.lang3.StringUtils;
import org.kie.api.KieServices;
import org.kie.api.builder.KieBuilder;
@@ -15,11 +15,12 @@ import org.kie.api.runtime.KieContainer;
import org.kie.api.runtime.KieSession;
import org.springframework.stereotype.Service;
import java.io.ByteArrayInputStream;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.HashMap;
import java.util.Map;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
@@ -29,7 +30,11 @@ public class DroolsExecutionService {
private Map<String, KieContainer> kieContainers = new HashMap<>();
private Map<String, Long> rulesVersionPerRuleSetId = new HashMap<>();
public KieContainer getKieContainer(String ruleSetId) {
KieContainer container = kieContainers.get(ruleSetId);
if (container == null) {
return createOrUpdateKieContainer(ruleSetId);
@@ -39,9 +44,6 @@ public class DroolsExecutionService {
}
@Getter
private long rulesVersion = -1;
public Section executeRules(KieContainer kieContainer, Section section) {
KieSession kieSession = kieContainer.newKieSession();
@@ -54,17 +56,24 @@ public class DroolsExecutionService {
}
public KieContainer updateRules(String ruleSetId) {
long version = rulesClient.getVersion(ruleSetId);
if (version > rulesVersion) {
rulesVersion = version;
Long rulesVersion = rulesVersionPerRuleSetId.get(ruleSetId);
if (rulesVersion == null) {
rulesVersion = -1L;
}
if (version > rulesVersion.longValue()) {
rulesVersionPerRuleSetId.put(ruleSetId, version);
return createOrUpdateKieContainer(ruleSetId);
}
return getKieContainer(ruleSetId);
}
private KieContainer createOrUpdateKieContainer(String ruleSetId) {
try {
@@ -92,7 +101,9 @@ public class DroolsExecutionService {
}
private KieModule getKieModule(String ruleSetId, String rules, KieServices kieServices) {
KieFileSystem kieFileSystem = kieServices.newKieFileSystem();
InputStream input = new ByteArrayInputStream(rules.getBytes(StandardCharsets.UTF_8));
kieFileSystem.write("src/main/resources/drools/rules" + ruleSetId + ".drl", kieServices.getResources()
@@ -104,10 +115,22 @@ public class DroolsExecutionService {
public void testRules(String rules) {
KieServices kieServices = KieServices.Factory.get();
KieModule kieModule = getKieModule("test-rules", rules, kieServices);
var container = kieServices.newKieContainer(kieModule.getReleaseId());
container.newKieSession();
container.dispose();
}
public long getRulesVersion(String ruleSetId) {
Long rulesVersion = rulesVersionPerRuleSetId.get(ruleSetId);
if (rulesVersion == null) {
return -1;
}
return rulesVersion.longValue();
}
}
@@ -17,10 +17,15 @@ import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Footer;
import com.iqser.red.service.redaction.v1.server.classification.model.Header;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.model.UnclassifiedText;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryModel;
@@ -29,6 +34,7 @@ import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionS
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.redaction.model.SectionSearchableTextPair;
import com.iqser.red.service.redaction.v1.model.SectionText;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@@ -50,7 +56,7 @@ public class EntityRedactionService {
dictionaryService.updateDictionary(ruleSetId);
KieContainer container = droolsExecutionService.updateRules(ruleSetId);
long rulesVersion = droolsExecutionService.getRulesVersion();
long rulesVersion = droolsExecutionService.getRulesVersion(ruleSetId);
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(ruleSetId);
@@ -67,10 +73,7 @@ public class EntityRedactionService {
});
Set<Entity> foundByLocal = findEntities(classifiedDoc, container, manualRedactions, dictionary, true, hintsPerSectionNumber);
// HashSet keeps the older value, but we want the new only.
documentEntities.removeAll(foundByLocal);
documentEntities.addAll(foundByLocal);
EntitySearchUtils.addEntitiesWithHigherRank(documentEntities, foundByLocal, dictionary);
EntitySearchUtils.removeEntitiesContainedInLarger(documentEntities);
}
@@ -86,7 +89,7 @@ public class EntityRedactionService {
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber(), entity
.getLegalBasis(), entity.isDictionaryEntry(), entity.getTextBefore(), entity.getTextAfter()));
.getLegalBasis(), entity.isDictionaryEntry(), entity.getTextBefore(), entity.getTextAfter(), entity.getStart(), entity.getEnd()));
}
}
@@ -110,13 +113,29 @@ public class EntityRedactionService {
List<Table> tables = paragraph.getTables();
for (Table table : tables) {
if (table.getColCount() == 2) {
sectionSearchableTextPairs.addAll(processTableAsOneText(table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionSearchableTextPairs.addAll(processTableAsOneText(classifiedDoc, table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
} else {
sectionSearchableTextPairs.addAll(processTablePerRow(table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionSearchableTextPairs.addAll(processTablePerRow(classifiedDoc, table, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
}
sectionNumber.incrementAndGet();
}
sectionSearchableTextPairs.add(processText(paragraph, manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionSearchableTextPairs.add(processText(classifiedDoc, paragraph.getSearchableText(), paragraph.getTextBlocks(), paragraph
.getHeadline(), manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionNumber.incrementAndGet();
}
for (Header header : classifiedDoc.getHeaders()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, header.getSearchableText(), header.getTextBlocks(), "Header", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionNumber.incrementAndGet();
}
for (Footer footer : classifiedDoc.getFooters()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, footer.getSearchableText(), footer.getTextBlocks(), "Footer", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionNumber.incrementAndGet();
}
for (UnclassifiedText unclassifiedText : classifiedDoc.getUnclassifiedTexts()) {
sectionSearchableTextPairs.add(processText(classifiedDoc, unclassifiedText.getSearchableText(), unclassifiedText.getTextBlocks(), "", manualRedactions, sectionNumber, dictionary, local, hintsPerSectionNumber));
sectionNumber.incrementAndGet();
}
@@ -153,32 +172,44 @@ public class EntityRedactionService {
}
private List<SectionSearchableTextPair> processTablePerRow(Table table, ManualRedactions manualRedactions,
private List<SectionSearchableTextPair> processTablePerRow(Document classifiedDoc, Table table, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
Map<String, CellValue> tabularData = new HashMap<>();
int start = 0;
List<Integer> cellStarts = new ArrayList<>();
SectionText sectionText = new SectionText();
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
SectionArea sectionArea = new SectionArea(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight(), cell.getTextBlocks().get(0).getSequences().get(0).getPage());
sectionText.getSectionAreas().add(sectionArea);
addSectionToManualRedactions(cell.getTextBlocks(), manualRedactions, table.getHeadline(), sectionNumber.intValue());
int cellStart = start;
cell.getHeaderCells().forEach(headerCell -> {
StringBuilder headerBuilder = new StringBuilder();
headerCell.getTextBlocks().forEach(textBlock -> headerBuilder.append(textBlock.getText()));
String headerName = headerBuilder.toString()
.replaceAll("\n", "")
.replaceAll(" ", "")
.replaceAll("-", "");
tabularData.put(headerName, new CellValue(cell.getTextBlocks(), cellStart));
});
if (!cell.isHeaderCell()) {
cell.getHeaderCells().forEach(headerCell -> {
StringBuilder headerBuilder = new StringBuilder();
headerCell.getTextBlocks().forEach(textBlock -> headerBuilder.append(textBlock.getText()));
String headerName = headerBuilder.toString()
.replaceAll("\n", "")
.replaceAll(" ", "")
.replaceAll("-", "");
sectionArea.setHeader(headerName);
tabularData.put(headerName, new CellValue(cell.getTextBlocks(), cellStart));
});
}
for (TextBlock textBlock : cell.getTextBlocks()) {
// TODO avoid cell overlap merging.
@@ -186,6 +217,7 @@ public class EntityRedactionService {
}
cellStarts.add(cellStart);
start = start + cell.toString().trim().length() + 1;
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, searchableRow, dictionary, cellStarts);
@@ -205,24 +237,41 @@ public class EntityRedactionService {
.dictionary(dictionary)
.build(), searchableRow));
if(!local) {
sectionText.setText(searchableRow.toString());
sectionText.setHeadline(table.getHeadline());
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(true);
classifiedDoc.getSectionText().add(sectionText);
}
sectionNumber.incrementAndGet();
}
return sectionSearchableTextPairs;
}
private List<SectionSearchableTextPair> processTableAsOneText(Table table, ManualRedactions manualRedactions,
private List<SectionSearchableTextPair> processTableAsOneText(Document classifiedDoc, Table table, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary,
boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
SearchableText entireTableText = new SearchableText();
SectionText sectionText = new SectionText();
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
if(!local) {
SectionArea sectionArea = new SectionArea(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight(), cell.getTextBlocks().get(0).getSequences().get(0).getPage());
sectionText.getSectionAreas().add(sectionArea);
}
for (TextBlock textBlock : cell.getTextBlocks()) {
entireTableText.addAll(textBlock.getSequences());
}
@@ -230,6 +279,7 @@ public class EntityRedactionService {
}
}
Set<Entity> rowEntities = findEntities(entireTableText, table.getHeadline(), sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(rowEntities, entireTableText, dictionary);
@@ -247,18 +297,40 @@ public class EntityRedactionService {
.dictionary(dictionary)
.build(), entireTableText));
if(!local) {
sectionText.setText(entireTableText.toString());
sectionText.setHeadline(table.getHeadline());
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(true);
classifiedDoc.getSectionText().add(sectionText);
}
return sectionSearchableTextPairs;
}
private SectionSearchableTextPair processText(Paragraph paragraph, ManualRedactions manualRedactions,
private SectionSearchableTextPair processText(Document classifiedDoc, SearchableText searchableText, List<TextBlock> paragraphTextBlocks,
String headline, ManualRedactions manualRedactions,
AtomicInteger sectionNumber, Dictionary dictionary, boolean local,
Map<Integer, Set<Entity>> hintsPerSectionNumber) {
SearchableText searchableText = paragraph.getSearchableText();
addSectionToManualRedactions(paragraph.getTextBlocks(), manualRedactions, paragraph.getHeadline(), sectionNumber
.intValue());
Set<Entity> entities = findEntities(searchableText, paragraph.getHeadline(), sectionNumber.intValue(), dictionary, local);
if(!local) {
SectionText sectionText = new SectionText();
for (TextBlock paragraphTextBlock : paragraphTextBlocks) {
SectionArea sectionArea = new SectionArea(new Point(paragraphTextBlock.getMinX(), paragraphTextBlock.getMinY()), paragraphTextBlock
.getWidth(), paragraphTextBlock.getHeight(), paragraphTextBlock.getPage());
sectionText.getSectionAreas().add(sectionArea);
}
sectionText.setText(searchableText.toString());
sectionText.setHeadline(headline);
sectionText.setSectionNumber(sectionNumber.intValue());
sectionText.setTable(false);
classifiedDoc.getSectionText().add(sectionText);
}
addSectionToManualRedactions(paragraphTextBlocks, manualRedactions, headline, sectionNumber.intValue());
Set<Entity> entities = findEntities(searchableText, headline, sectionNumber.intValue(), dictionary, local);
surroundingWordsService.addSurroundingText(entities, searchableText, dictionary);
return new SectionSearchableTextPair(Section.builder()
@@ -269,7 +341,7 @@ public class EntityRedactionService {
.collect(Collectors.toSet()) : entities)
.text(searchableText.getAsStringWithLinebreaks())
.searchText(searchableText.toString())
.headline(paragraph.getHeadline())
.headline(headline)
.sectionNumber(sectionNumber.intValue())
.searchableText(searchableText)
.dictionary(dictionary)
@@ -277,7 +349,7 @@ public class EntityRedactionService {
}
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber,
public Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber,
Dictionary dictionary, boolean local) {
Set<Entity> found = new HashSet<>();
@@ -0,0 +1,281 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.awt.geom.Rectangle2D;
import java.io.ByteArrayInputStream;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.Iterator;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import java.util.stream.Stream;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.kie.api.runtime.KieContainer;
import org.springframework.stereotype.Service;
import org.springframework.web.bind.annotation.RequestBody;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualForceRedact;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.ReanalyzeResult;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RenalyzeRequest;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.model.SectionText;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.parsing.PDFAreaTextStripper;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Dictionary;
import com.iqser.red.service.redaction.v1.server.redaction.model.DictionaryIncrement;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.ReanalysisSection;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import com.iqser.red.service.redaction.v1.server.redaction.model.SectionSearchableTextPair;
import com.iqser.red.service.redaction.v1.server.redaction.utils.EntitySearchUtils;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class ReanalyzeService {
private final DictionaryService dictionaryService;
private final DroolsExecutionService droolsExecutionService;
private final SurroundingWordsService surroundingWordsService;
private final EntityRedactionService entityRedactionService;
private final RedactionLogCreatorService redactionLogCreatorService;
public ReanalyzeResult reanalyze(@RequestBody RenalyzeRequest renalyzeRequest) {
DictionaryIncrement dictionaryIncrement = dictionaryService.getDictionaryIncrements(renalyzeRequest.getRuleSetId(), renalyzeRequest
.getRedactionLog()
.getDictionaryVersion());
Set<String> manualForceAndRemoveIds = getForceAndRemoveIds(renalyzeRequest.getManualRedactions());
Map<String, List<Comment>> comments = null;
Set<ManualRedactionEntry> manualAdds = null;
if (renalyzeRequest.getManualRedactions() != null) {
// TODO comments will be removed from redactionLog, so we ignore this first.
comments = renalyzeRequest.getManualRedactions().getComments();
manualAdds = renalyzeRequest.getManualRedactions().getEntriesToAdd();
}
Set<Integer> sectionsToReanaylse = new HashSet<>();
for (RedactionLogEntry entry : renalyzeRequest.getRedactionLog().getRedactionLogEntry()) {
if (entry.isManual() || manualForceAndRemoveIds.contains(entry.getId())) {
sectionsToReanaylse.add(entry.getSectionNumber());
}
}
for (SectionText sectionText : renalyzeRequest.getText().getSectionTexts()) {
Set<Entity> entities = EntitySearchUtils.find(sectionText.getText(), dictionaryIncrement.getValues(), "find", sectionText
.getHeadline(), sectionText.getSectionNumber(), false);
if (!entities.isEmpty()) {
sectionsToReanaylse.add(sectionText.getSectionNumber());
}
if (manualAdds != null) {
for (SectionArea sectionArea : sectionText.getSectionAreas()) {
for (ManualRedactionEntry manualAdd : manualAdds) {
for (Rectangle manualPosition : manualAdd.getPositions()) {
if (sectionArea.contains(manualPosition)) {
manualAdd.setSection(sectionText.getHeadline());
manualAdd.setSectionNumber(sectionText.getSectionNumber());
}
}
}
}
}
}
if (sectionsToReanaylse.isEmpty() && (manualAdds == null || manualAdds.isEmpty())) {
renalyzeRequest.getRedactionLog().setDictionaryVersion(dictionaryIncrement.getDictionaryVersion());
return ReanalyzeResult.builder().redactionLog(renalyzeRequest.getRedactionLog()).build();
}
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(renalyzeRequest.getDocument()))) {
List<ReanalysisSection> reanalysisSections = new ArrayList<>();
for (SectionText sectionText : renalyzeRequest.getText().getSectionTexts()) {
if (!sectionsToReanaylse.contains(sectionText.getSectionNumber())) {
continue;
}
ReanalysisSection reanalysisSection = new ReanalysisSection();
reanalysisSection.setHeadline(sectionText.getHeadline());
reanalysisSection.setSectionNumber(sectionText.getSectionNumber());
List<TextBlock> textBlocks = new ArrayList<>();
Map<Integer, List<SectionArea>> sectionAreasPerPage = new HashMap<>();
for (SectionArea sectionArea : sectionText.getSectionAreas()) {
sectionAreasPerPage.computeIfAbsent(sectionArea.getPage(), (x) -> new ArrayList<>())
.add(sectionArea);
}
Map<String, CellValue> tabularData = new HashMap<>();
List<Integer> cellStarts = new ArrayList<>();
for (Integer page : sectionAreasPerPage.keySet()) {
List<SectionArea> areasOnPage = sectionAreasPerPage.get(page);
PDPage pdPage = pdDocument.getPage(page - 1);
PDRectangle cropBox = pdPage.getCropBox();
PDFAreaTextStripper textStripper = new PDFAreaTextStripper();
textStripper.setPageNumber(page);
int cellStart = 0;
for (SectionArea sectionArea : areasOnPage) {
Rectangle2D rect = null;
if (pdPage.getRotation() == 90) {
rect = new Rectangle2D.Float(sectionArea.getTopLeft().getY(), sectionArea.getTopLeft()
.getX(), sectionArea.getHeight(), sectionArea.getWidth() + 0.001f);
} else {
rect = new Rectangle2D.Float(sectionArea.getTopLeft().getX(), -sectionArea.getTopLeft()
.getY() + cropBox.getUpperRightY() - sectionArea.getHeight(), sectionArea.getWidth(), sectionArea
.getHeight() + 0.001f);
}
textStripper.addRegion(String.valueOf(1), rect);
textStripper.extractRegions(pdPage);
textStripper.getTextForRegion(String.valueOf(1));
List<TextPositionSequence> positions = textStripper.getTextPositionSequences();
TextBlock textBlock = new TextBlock(sectionArea.getTopLeft().getX(), sectionArea.getTopLeft()
.getX() + sectionArea.getWidth(), sectionArea.getTopLeft()
.getY(), sectionArea.getTopLeft().getY() + sectionArea.getHeight(), positions, 0);
if (sectionText.isTable()) {
Cell cell = new Cell();
cell.addTextBlock(textBlock);
tabularData.put(sectionArea.getHeader(), new CellValue(cell.getTextBlocks(), cellStart));
cellStarts.add(cellStart);
cellStart = cellStart + cell.toString().trim().length() + 1;
}
textBlocks.add(textBlock);
textStripper.clearPositions();
}
}
reanalysisSection.setTextBlocks(textBlocks);
reanalysisSection.setTabularData(tabularData);
reanalysisSections.add(reanalysisSection);
if (sectionText.isTable()) {
reanalysisSection.setCellStarts(cellStarts);
}
}
//--
KieContainer kieContainer = droolsExecutionService.updateRules(renalyzeRequest.getRuleSetId());
Dictionary dictionary = dictionaryService.getDeepCopyDictionary(renalyzeRequest.getRuleSetId());
List<SectionSearchableTextPair> sectionSearchableTextPairs = new ArrayList<>();
for (ReanalysisSection reanalysisSection : reanalysisSections) {
Set<Entity> entities = entityRedactionService.findEntities(reanalysisSection.getSearchableText(), reanalysisSection
.getHeadline(), reanalysisSection.getSectionNumber(), dictionary, false);
if (reanalysisSection.getCellStarts() != null) {
surroundingWordsService.addSurroundingText(entities, reanalysisSection.getSearchableText(), dictionary, reanalysisSection
.getCellStarts());
} else {
surroundingWordsService.addSurroundingText(entities, reanalysisSection.getSearchableText(), dictionary);
}
sectionSearchableTextPairs.add(new SectionSearchableTextPair(Section.builder()
.isLocal(false)
.dictionaryTypes(dictionary.getTypes())
.entities(entities)
.text(reanalysisSection.getSearchableText().getAsStringWithLinebreaks())
.searchText(reanalysisSection.getSearchableText().toString())
.headline(reanalysisSection.getHeadline())
.sectionNumber(reanalysisSection.getSectionNumber())
.tabularData(reanalysisSection.getTabularData())
.searchableText(reanalysisSection.getSearchableText())
.dictionary(dictionary)
.build(), reanalysisSection.getSearchableText()));
}
Set<Entity> entities = new HashSet<>();
sectionSearchableTextPairs.forEach(sectionSearchableTextPair -> {
Section analysedRowSection = droolsExecutionService.executeRules(kieContainer, sectionSearchableTextPair
.getSection());
entities.addAll(analysedRowSection.getEntities());
});
Map<Integer, List<Entity>> entitiesPerPage = new HashMap<>();
for (Entity entity : entities) {
Map<Integer, List<EntityPositionSequence>> sequenceOnPage = new HashMap<>();
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
sequenceOnPage.computeIfAbsent(entityPositionSequence.getPageNumber(), (x) -> new ArrayList<>())
.add(entityPositionSequence);
}
for (Map.Entry<Integer, List<EntityPositionSequence>> entry : sequenceOnPage.entrySet()) {
entitiesPerPage.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber(), entity
.getLegalBasis(), entity.isDictionaryEntry(), entity.getTextBefore(), entity.getTextAfter(), entity
.getStart(), entity.getEnd()));
}
}
List<RedactionLogEntry> newRedactionLogEntries = new ArrayList<>();
for (int page = 1; page <= pdDocument.getNumberOfPages(); page++) {
if (entitiesPerPage.get(page) != null) {
newRedactionLogEntries.addAll(redactionLogCreatorService.addEntries(entitiesPerPage, renalyzeRequest
.getManualRedactions(), page, renalyzeRequest.getRuleSetId()));
}
newRedactionLogEntries.addAll(redactionLogCreatorService.addManualAddEntries(manualAdds, comments, page, renalyzeRequest
.getRuleSetId()));
}
Iterator<RedactionLogEntry> itty = renalyzeRequest.getRedactionLog().getRedactionLogEntry().iterator();
while (itty.hasNext()) {
RedactionLogEntry entry = itty.next();
if (sectionsToReanaylse.contains(entry.getSectionNumber()) || entry.getSectionNumber() == 0) {
itty.remove();
}
}
renalyzeRequest.getRedactionLog().getRedactionLogEntry().addAll(newRedactionLogEntries);
renalyzeRequest.getRedactionLog().setDictionaryVersion(dictionaryIncrement.getDictionaryVersion());
return ReanalyzeResult.builder().redactionLog(renalyzeRequest.getRedactionLog()).build();
} catch (Exception e) {
throw new RedactionException(e);
}
}
private Set<String> getForceAndRemoveIds(ManualRedactions manualRedactions) {
if (manualRedactions == null) {
return new HashSet<>();
}
return Stream.concat(manualRedactions.getIdsToRemove()
.stream()
.map(IdRemoval::getId), manualRedactions.getForceRedacts().stream().map(ManualForceRedact::getId))
.collect(Collectors.toSet());
}
}
@@ -1,8 +1,21 @@
package com.iqser.red.service.redaction.v1.server.visualization.service;
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.awt.geom.Rectangle2D;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.text.TextPosition;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.CellRectangle;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualForceRedact;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactionType;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
@@ -17,59 +30,62 @@ import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.graphics.color.PDColor;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationText;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.apache.pdfbox.text.TextPosition;
import org.springframework.stereotype.Service;
import java.awt.Color;
import java.io.IOException;
import java.util.ArrayList;
import java.util.GregorianCalendar;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import lombok.RequiredArgsConstructor;
@Service
@RequiredArgsConstructor
public class AnnotationHighlightService {
public class RedactionLogCreatorService {
private static final String IMAGE = "image";
private final DictionaryService dictionaryService;
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions, String ruleSetId) throws IOException {
public void createRedactionLog(Document classifiedDoc, int numberOfPages, ManualRedactions manualRedactions,
String ruleSetId) {
Set<Integer> manualRedactionPages = getManualRedactionPages(manualRedactions);
for (int page = 1; page <= document.getNumberOfPages(); page++) {
for (int page = 1; page <= numberOfPages; page++) {
PDPage pdPage = document.getPage(page - 1);
drawSectionFrames(document, classifiedDoc, flatRedaction, pdPage, page);
addSectionGrid(classifiedDoc, page);
if (classifiedDoc.getEntities().get(page) != null) {
addAnnotations(pdPage, classifiedDoc, flatRedaction, manualRedactions, page, ruleSetId);
classifiedDoc.getRedactionLogEntities().addAll(addEntries(classifiedDoc.getEntities(), manualRedactions, page, ruleSetId));
}
if (manualRedactionPages.contains(page)) {
addManualAnnotations(pdPage, classifiedDoc, manualRedactions, page, ruleSetId);
classifiedDoc.getRedactionLogEntities().addAll(addManualAddEntries(manualRedactions.getEntriesToAdd(), manualRedactions.getComments(), page, ruleSetId));
}
if (!classifiedDoc.getPages().get(page - 1).getImageBounds().isEmpty()) {
addImageEntries(classifiedDoc, page, ruleSetId);
}
}
}
private void addImageEntries(Document classifiedDoc, int pageNumber, String ruleSetId) {
for (Rectangle2D imageBounds : classifiedDoc.getPages().get(pageNumber - 1).getImageBounds()) {
RedactionLogEntry redactionLogEntry = RedactionLogEntry.builder()
.id(IdBuilder.buildId(imageBounds, pageNumber))
.color(getColor(IMAGE, ruleSetId))
.type(IMAGE)
.redacted(false)
.isHint(true)
.manual(false)
.isDictionaryEntry(false)
.isRecommendation(false)
.positions(List.of(new Rectangle(new Point((float) imageBounds.getX(), (float) imageBounds.getY()), (float) imageBounds
.getWidth(), (float) imageBounds.getHeight(), pageNumber)))
.build();
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
@@ -91,22 +107,16 @@ public class AnnotationHighlightService {
}
private void addAnnotations(PDPage pdPage, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions, int page, String ruleSetId) throws IOException {
public List<RedactionLogEntry> addEntries(Map<Integer, List<Entity>> entities, ManualRedactions manualRedactions, int page, String ruleSetId) {
List<PDAnnotation> annotations = pdPage.getAnnotations();
List<RedactionLogEntry> redactionLogEntities = new ArrayList<>();
// Duplicates can exist due table extraction colums over multiple rows.
Set<String> processedIds = new HashSet<>();
entityLoop:
for (Entity entity : classifiedDoc.getEntities().get(page)) {
for (Entity entity : entities.get(page)) {
if (flatRedaction && !isRedactionType(entity, ruleSetId)) {
continue;
}
boolean requestedToRemove = false;
List<Comment> comments = null;
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
@@ -130,10 +140,11 @@ public class AnnotationHighlightService {
redactionLogEntry.setRedacted(false);
redactionLogEntry.setStatus(Status.APPROVED);
manualOverrideReason = entity.getRedactionReason() + ", removed by manual override";
redactionLogEntry.setColor(getColor(entity, ruleSetId, false));
} else if (manualRemoval.getStatus().equals(Status.REQUESTED)) {
requestedToRemove = true;
manualOverrideReason = entity.getRedactionReason() + ", requested to remove";
redactionLogEntry.setStatus(Status.REQUESTED);
redactionLogEntry.setColor(getColor(entity, ruleSetId, true));
} else {
redactionLogEntry.setStatus(Status.DECLINED);
}
@@ -144,7 +155,34 @@ public class AnnotationHighlightService {
redactionLogEntry.setManualRedactionType(ManualRedactionType.REMOVE);
}
}
}
if (manualRedactions != null && !manualRedactions.getForceRedacts().isEmpty()) {
for (ManualForceRedact manualForceRedact : manualRedactions.getForceRedacts()) {
if (manualForceRedact.getId().equals(entityPositionSequence.getId())) {
String manualOverrideReason = null;
if (manualForceRedact.getStatus().equals(Status.APPROVED)) {
entity.setRedaction(true);
redactionLogEntry.setRedacted(true);
redactionLogEntry.setStatus(Status.APPROVED);
redactionLogEntry.setColor(getColor(entity, ruleSetId, false));
manualOverrideReason = entity.getRedactionReason() + ", forced by manual override";
redactionLogEntry.setLegalBasis(manualForceRedact.getLegalBasis());
} else if (manualForceRedact.getStatus().equals(Status.REQUESTED)) {
manualOverrideReason = entity.getRedactionReason() + ", requested to force redact";
redactionLogEntry.setStatus(Status.REQUESTED);
redactionLogEntry.setColor(getColor(entity, ruleSetId, true));
redactionLogEntry.setLegalBasis(manualForceRedact.getLegalBasis());
} else {
redactionLogEntry.setStatus(Status.DECLINED);
}
entity.setRedactionReason(manualOverrideReason != null ? manualOverrideReason : entity.getRedactionReason());
redactionLogEntry.setReason(manualOverrideReason);
redactionLogEntry.setManual(true);
redactionLogEntry.setManualRedactionType(ManualRedactionType.FORCE_REDACT);
}
}
}
if (CollectionUtils.isNotEmpty(entityPositionSequence.getSequences())) {
@@ -157,21 +195,21 @@ public class AnnotationHighlightService {
comments = manualRedactions.getComments().get(entityPositionSequence.getId());
}
redactionLogEntry.setComments(comments);
redactionLogEntry.getPositions().addAll(rectanglesPerLine);
annotations.addAll(createAnnotation(rectanglesPerLine, entityPositionSequence.getId(), createAnnotationContent(entity), getColor(entity, ruleSetId, requestedToRemove), comments, !isHint(entity, ruleSetId)));
}
redactionLogEntry.setId(entityPositionSequence.getId());
// FIXME ids should never be null. Figure out why this happens.
if (redactionLogEntry.getId() != null) {
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
redactionLogEntities.add(redactionLogEntry);
}
}
}
return redactionLogEntities;
}
@@ -191,7 +229,7 @@ public class AnnotationHighlightService {
startIndex = i;
}
}
if (startIndex != textPositions.size() - 1) {
if (startIndex != textPositions.size()) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, textPositions.size()), page).getRectangle());
}
}
@@ -200,16 +238,16 @@ public class AnnotationHighlightService {
}
private void addManualAnnotations(PDPage pdPage, Document classifiedDoc, ManualRedactions manualRedactions,
int page, String ruleSetId) throws IOException {
public List<RedactionLogEntry> addManualAddEntries(Set<ManualRedactionEntry> manualAdds, Map<String, List<Comment>> comments, int page,
String ruleSetId) {
if (manualRedactions == null) {
return;
List<RedactionLogEntry> redactionLogEntities = new ArrayList<>();
if (manualAdds == null) {
return redactionLogEntities;
}
List<PDAnnotation> annotations = pdPage.getAnnotations();
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
for (ManualRedactionEntry manualRedactionEntry : manualAdds) {
String id = manualRedactionEntry.getId();
@@ -223,12 +261,13 @@ public class AnnotationHighlightService {
}
}
redactionLogEntry.setComments(comments.get(id));
if (!rectanglesOnPage.isEmpty() && !approvedAndShouldBeInDictionary(manualRedactionEntry)) {
annotations.addAll(createAnnotation(rectanglesOnPage, id, createAnnotationContent(manualRedactionEntry), getColorForManualAdd(manualRedactionEntry
.getType(), ruleSetId, manualRedactionEntry.getStatus()), manualRedactions.getComments().get(id), true));
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
redactionLogEntities.add(redactionLogEntry);
}
}
return redactionLogEntities;
}
@@ -238,11 +277,12 @@ public class AnnotationHighlightService {
}
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry, String id, String ruleSetId) {
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry, String id,
String ruleSetId) {
return RedactionLogEntry.builder()
.id(id)
.color(getColor(manualRedactionEntry.getType(), ruleSetId))
.color(getColorForManualAdd(manualRedactionEntry.getType(), ruleSetId, manualRedactionEntry.getStatus()))
.reason(manualRedactionEntry.getReason())
.legalBasis(manualRedactionEntry.getLegalBasis())
.value(manualRedactionEntry.getValue())
@@ -276,126 +316,12 @@ public class AnnotationHighlightService {
.isDictionaryEntry(entity.isDictionaryEntry())
.textAfter(entity.getTextAfter())
.textBefore(entity.getTextBefore())
.startOffset(entity.getStart())
.endOffset(entity.getEnd())
.build();
}
private List<PDAnnotation> createAnnotation(List<Rectangle> rectangles, String id, String content, float[] color,
List<Comment> comments, boolean popup) {
List<PDAnnotation> annotations = new ArrayList<>();
PDAnnotationTextMarkup annotation = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
annotation.constructAppearances();
PDRectangle pdRectangle = toPDRectangle(rectangles);
annotation.setRectangle(pdRectangle);
annotation.setQuadPoints(toQuadPoints(rectangles));
if (popup) {
annotation.setContents(content);
}
annotation.setTitlePopup(id);
annotation.setAnnotationName(id);
annotation.setColor(new PDColor(color, PDDeviceRGB.INSTANCE));
annotations.add(annotation);
if (comments != null) {
for (Comment comment : comments) {
PDAnnotationText txtAnnot = new PDAnnotationText();
txtAnnot.setAnnotationName(comment.getId());
txtAnnot.setInReplyTo(annotation); // Reference to highlight annotation
txtAnnot.setName(PDAnnotationText.NAME_COMMENT);
txtAnnot.setCreationDate(GregorianCalendar.from(comment.getDate().toZonedDateTime()));
txtAnnot.setTitlePopup(comment.getUser());
txtAnnot.setContents(comment.getText());
txtAnnot.setRectangle(pdRectangle);
annotations.add(txtAnnot);
}
}
return annotations;
}
private String createAnnotationContent(Entity entity) {
return "\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\nLegal basis:" + entity
.getLegalBasis() + "\n\nIn section: \"" + entity.getHeadline() + "\"";
}
private String createAnnotationContent(ManualRedactionEntry entry) {
return "\nManual Redaction\n\nIn Section : \"" + entry.getSection() + "\"";
}
private PDRectangle toPDRectangle(List<Rectangle> rectangles) {
float lowerLeftX = Float.MAX_VALUE;
float upperRightX = 0;
float lowerLeftY = 0;
float upperRightY = Float.MAX_VALUE;
for (Rectangle rectangle : rectangles) {
if (rectangle.getTopLeft().getX() < lowerLeftX) {
lowerLeftX = rectangle.getTopLeft().getX();
}
if (rectangle.getTopLeft().getX() + rectangle.getWidth() > upperRightX) {
upperRightX = rectangle.getTopLeft().getX() + rectangle.getWidth();
}
if (rectangle.getTopLeft().getY() + rectangle.getHeight() > lowerLeftY) {
lowerLeftY = rectangle.getTopLeft().getY() + rectangle.getHeight();
}
if (rectangle.getTopLeft().getY() < upperRightY) {
upperRightY = rectangle.getTopLeft().getY();
}
}
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(lowerLeftX);
annotationPosition.setLowerLeftY(lowerLeftY);
annotationPosition.setUpperRightX(upperRightX);
annotationPosition.setUpperRightY(upperRightY);
return annotationPosition;
}
private float[] toQuadPoints(List<Rectangle> rectangles) {
float[] quadPoints = new float[rectangles.size() * 8];
int i = 0;
for (Rectangle rectangle : rectangles) {
float[] quadPoint = toQuadPoint(rectangle);
for (int j = 0; j <= 7; j++) {
quadPoints[i + j] = quadPoint[j];
}
i += 8;
}
return quadPoints;
}
private float[] toQuadPoint(Rectangle rectangle) {
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
return new float[]{rectangle.getTopLeft().getX(), rectangle.getTopLeft().getY(), rectangle.getTopLeft()
.getX() + rectangle.getWidth(), rectangle.getTopLeft().getY(), rectangle.getTopLeft().getX(), rectangle.getTopLeft()
.getY() + rectangle.getHeight(), rectangle.getTopLeft()
.getX() + rectangle.getWidth(), rectangle.getTopLeft().getY() + rectangle.getHeight()};
}
private boolean isRedactionType(Entity entity, String ruleSetId) {
if (!entity.isRedaction()) {
return false;
}
return !isHint(entity, ruleSetId);
}
private float[] getColor(Entity entity, String ruleSetId, boolean requestedToRemove) {
if (requestedToRemove) {
@@ -420,27 +346,25 @@ public class AnnotationHighlightService {
private float[] getColor(String type, String ruleSetId) {
return dictionaryService.getColor(type, ruleSetId);
}
private boolean isHint(Entity entity, String ruleSetId) {
return dictionaryService.isHint(entity.getType(), ruleSetId);
}
private boolean isRecommendation(Entity entity, String ruleSetId) {
return dictionaryService.isRecommendation(entity.getType(), ruleSetId);
}
private void drawSectionFrames(PDDocument document, Document classifiedDoc, boolean flatRedaction, PDPage pdPage,
int page) throws IOException {
private void addSectionGrid(Document classifiedDoc, int page) {
if (flatRedaction) {
return;
}
PDPageContentStream contentStream = new PDPageContentStream(document, pdPage, PDPageContentStream.AppendMode.APPEND, true);
for (Paragraph paragraph : classifiedDoc.getParagraphs()) {
for (int i = 0; i <= paragraph.getPageBlocks().size() - 1; i++) {
@@ -450,10 +374,8 @@ public class AnnotationHighlightService {
if (textBlock.getPage() != page) {
continue;
}
if (textBlock instanceof TextBlock) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
visualizeTextBlock((TextBlock) textBlock, contentStream);
if (textBlock instanceof TextBlock) {
classifiedDoc.getSectionGrid()
.getRectanglesPerPage()
@@ -463,8 +385,15 @@ public class AnnotationHighlightService {
} else if (textBlock instanceof Table) {
textBlock.setClassification((i + 1) + "/" + paragraph.getPageBlocks().size());
List<CellRectangle> cellRectangles = visualizeTable((Table) textBlock, contentStream);
List<CellRectangle> cellRectangles = new ArrayList<>();
for (List<Cell> row : ((Table) textBlock).getRows()) {
for (Cell cell : row) {
if (cell != null) {
cellRectangles.add(new CellRectangle(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight()));
}
}
}
classifiedDoc.getSectionGrid()
.getRectanglesPerPage()
@@ -475,56 +404,6 @@ public class AnnotationHighlightService {
}
}
}
contentStream.close();
}
private void visualizeTextBlock(TextBlock textBlock, PDPageContentStream contentStream) throws IOException {
contentStream.setStrokingColor(Color.LIGHT_GRAY);
contentStream.setLineWidth(0.5f);
contentStream.addRect(textBlock.getMinX(), textBlock.getMinY(), textBlock.getWidth(), textBlock.getHeight());
contentStream.stroke();
if (textBlock.getClassification() != null) {
contentStream.beginText();
contentStream.setNonStrokingColor(Color.DARK_GRAY);
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
contentStream.newLineAtOffset(textBlock.getMinX(), textBlock.getMaxY());
contentStream.showText(textBlock.getClassification());
contentStream.endText();
}
}
private List<CellRectangle> visualizeTable(Table table, PDPageContentStream contentStream) throws IOException {
List<CellRectangle> cellRectangles = new ArrayList<>();
for (List<Cell> row : table.getRows()) {
for (Cell cell : row) {
if (cell != null) {
contentStream.setLineWidth(0.5f);
contentStream.setStrokingColor(Color.CYAN);
contentStream.addRect((float) cell.getX(), (float) cell.getY(), (float) cell.getWidth(), (float) cell
.getHeight());
contentStream.stroke();
cellRectangles.add(new CellRectangle(new Point((float) cell.getX(), (float) cell.getY()), (float) cell
.getWidth(), (float) cell.getHeight()));
}
}
}
if (table.getClassification() != null) {
contentStream.beginText();
contentStream.setNonStrokingColor(Color.DARK_GRAY);
contentStream.setFont(PDType1Font.TIMES_ROMAN, 8f);
contentStream.newLineAtOffset(table.getMinX(), table.getMinY());
contentStream.showText(table.getClassification());
contentStream.endText();
}
return cellRectangles;
}
}
@@ -104,4 +104,27 @@ public class EntitySearchUtils {
}
entities.removeAll(wordsToRemove);
}
public void addEntitiesWithHigherRank(Set<Entity> entities, Set<Entity> found, Dictionary dictionary) {
found.forEach(f -> addEntitiesWithHigherRank(entities, f, dictionary));
}
public void addEntitiesWithHigherRank(Set<Entity> entities, Entity found, Dictionary dictionary) {
if(entities.contains(found)){
Entity existing = entities.stream().filter(entity -> entity.equals(found)).findFirst().get();
if (dictionary.getDictionaryRank(existing.getType()) <= dictionary.getDictionaryRank(found.getType())){
entities.remove(found);
}
}
entities.add(found);
}
public void addEntitiesIgnoreRank(Set<Entity> entities, Set<Entity> found){
// HashSet keeps old value but we want the new.
entities.removeAll(found);
entities.addAll(found);
}
}
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import java.awt.geom.Rectangle2D;
import java.nio.charset.StandardCharsets;
import java.util.List;
@@ -23,4 +24,16 @@ public class IdBuilder {
return hashFunction.hashString(sb.toString(), StandardCharsets.UTF_8).toString();
}
public String buildId(Rectangle2D rectangle2D, int page){
StringBuilder sb = new StringBuilder();
sb.append("x").append(rectangle2D.getX()).append("y").append(rectangle2D.getY()).append("h").append(rectangle2D.getHeight()).append("w").append(rectangle2D.getWidth()).append("p").append(page);
return hashFunction.hashString(sb.toString(), StandardCharsets.UTF_8).toString();
}
}
@@ -48,6 +48,7 @@ public class PdfSegmentationService {
stripper.setPageNumber(pageNumber);
stripper.setStartPage(pageNumber);
stripper.setEndPage(pageNumber);
stripper.setPdpage(pdPage);
stripper.getText(pdDocument);
PDRectangle pdr = pdPage.getMediaBox();
@@ -56,18 +57,17 @@ public class PdfSegmentationService {
int rotation = pdPage.getRotation();
boolean isRotated = rotation != 0 && rotation != 360;
ParsedElements parsedElements = ParsedElements.builder()
.rulings(stripper.getRulings())
.sequences(stripper.getTextPositionSequences())
.maxCharWidth(stripper.getMaxCharWidths())
.maxCharHeight(stripper.getMaxCharWidths())
.imageBounds(stripper.getImageBounds())
.minCharWidth(stripper.getMinCharWidth())
.maxCharWidth(stripper.getMaxCharWidth())
.landscape(isLandscape)
.rotated(isRotated)
.build();
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), parsedElements
.getMaxCharWidth(), parsedElements.getMaxCharHeight());
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), stripper.getMinCharWidth(), stripper.getMaxCharHeight());
Page page = blockificationService.blockify(parsedElements.getSequences(), cleanRulings.getHorizontal(), cleanRulings
.getVertical());
@@ -81,8 +81,10 @@ public class PdfSegmentationService {
page.setPageNumber(pageNumber);
increaseDocumentStatistics(page, document);
page.setImageBounds(parsedElements.getImageBounds());
pages.add(page);
}
document.setPages(pages);
classificationService.classifyDocument(document);
@@ -90,11 +92,9 @@ public class PdfSegmentationService {
sectionsBuilderService.buildSections(document);
return document;
}
private void increaseDocumentStatistics(Page page, Document document) {
if (!page.isLandscape()) {
@@ -10,9 +10,12 @@ import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Footer;
import com.iqser.red.service.redaction.v1.server.classification.model.Header;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.classification.model.UnclassifiedText;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
@@ -24,25 +27,46 @@ public class SectionsBuilderService {
List<AbstractTextContainer> chunkWords = new ArrayList<>();
List<Paragraph> chunkBlockList = new ArrayList<>();
List<Header> headers = new ArrayList<>();
List<Footer> footers = new ArrayList<>();
List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
AbstractTextContainer prev = null;
String lastHeadline = "";
Table previousTable = null;
for (Page page : document.getPages()) {
List<TextBlock> header = new ArrayList<>();
List<TextBlock> footer = new ArrayList<>();
List<TextBlock> unclassifiedText = new ArrayList<>();
for (AbstractTextContainer current : page.getTextBlocks()) {
if (current.getClassification() == null || current.getClassification()
.equals("Header") || current.getClassification().equals("Footer")) {
if (current.getClassification() == null) {
continue;
}
current.setPage(page.getPageNumber());
if (prev != null && current.getClassification().startsWith("H ") && !prev.getClassification().startsWith("H ") || !document.isHeadlines()) {
if (current.getClassification().equals("Header")) {
header.add((TextBlock) current);
continue;
}
if (current.getClassification().equals("Footer")) {
footer.add((TextBlock) current);
continue;
}
if (current.getClassification().equals("Other")) {
unclassifiedText.add((TextBlock) current);
continue;
}
if (prev != null && current.getClassification().startsWith("H ") && !prev.getClassification()
.startsWith("H ") || !document.isHeadlines()) {
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
if(document.isHeadlines()) {
if (document.isHeadlines()) {
lastHeadline = current.getText();
}
chunkBlockList.add(chunkBlock);
@@ -60,6 +84,16 @@ public class SectionsBuilderService {
chunkWords.add(current);
prev = current;
}
if(!header.isEmpty()) {
headers.add(new Header(header));
}
if(!footer.isEmpty()) {
footers.add(new Footer(footer));
}
if(!unclassifiedText.isEmpty()) {
unclassifiedTexts.add(new UnclassifiedText(unclassifiedText));
}
}
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
@@ -67,6 +101,9 @@ public class SectionsBuilderService {
chunkBlockList.add(chunkBlock);
document.setParagraphs(chunkBlockList);
document.setHeaders(headers);
document.setFooters(footers);
document.setUnclassifiedTexts(unclassifiedTexts);
}
@@ -175,9 +212,9 @@ public class SectionsBuilderService {
private boolean hasInvalidHeaderInformation(Table table) {
return table.getRows().stream()
.flatMap(row -> row.stream()
.filter(cell -> CollectionUtils.isNotEmpty(cell.getHeaderCells())))
return table.getRows()
.stream()
.flatMap(row -> row.stream().filter(cell -> CollectionUtils.isNotEmpty(cell.getHeaderCells())))
.findAny()
.isEmpty();
@@ -188,6 +225,9 @@ public class SectionsBuilderService {
for (int i = table.getRowCount() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = table.getRows().get(i);
if (row.size() == 1) {
continue;
}
boolean allNonHeader = true;
for (Cell cell : row) {
if (cell.isHeaderCell()) {
@@ -11,10 +11,12 @@ import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizati
import lombok.Data;
import lombok.EqualsAndHashCode;
import lombok.NoArgsConstructor;
@SuppressWarnings("serial")
@Data
@EqualsAndHashCode(callSuper = true)
@NoArgsConstructor
public class Cell extends Rectangle {
private List<TextBlock> textBlocks = new ArrayList<>();
@@ -59,6 +59,16 @@ public class Table extends AbstractTextContainer {
if (rows == null) {
rows = computeRows();
// Ignore rows that does not contain any cells and values.
List<List<Cell>> rowsToRemove = new ArrayList<>();
for (List<Cell> row: rows){
if (row.size() == 1 && row.get(0).getTextBlocks().isEmpty()){
rowsToRemove.add(row);
}
}
rows.removeAll(rowsToRemove);
computeHeaders();
}
@@ -100,6 +110,10 @@ public class Table extends AbstractTextContainer {
// we move from left to right and top to bottom
for (int rowIndex = 0; rowIndex < rows.size(); rowIndex++) {
List<Cell> rowCells = rows.get(rowIndex);
if(rowCells.size() == 1){
continue;
}
for (int colIndex = 0; colIndex < rowCells.size(); colIndex++) {
Cell cell = rowCells.get(colIndex);
List<Cell> cellsToTheLeft = rowCells.subList(0, colIndex);
@@ -18,9 +18,10 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@Service
public class RulingCleaningService {
public CleanRulings getCleanRulings(List<Ruling> rulings, float maxCharWidth, float maxCharHeight){
public CleanRulings getCleanRulings(List<Ruling> rulings, float minCharWidth, float maxCharHeight) {
if (!rulings.isEmpty()) {
snapPoints(rulings, maxCharWidth , maxCharHeight);
snapPoints(rulings, minCharWidth, maxCharHeight);
}
List<Ruling> vrs = new ArrayList<>();
@@ -39,13 +40,10 @@ public class RulingCleaningService {
}
List<Ruling> horizontalRulingLines = collapseOrientedRulings(hrs);
return CleanRulings
.builder()
.vertical(verticalRulingLines)
.horizontal(horizontalRulingLines)
.build();
return CleanRulings.builder().vertical(verticalRulingLines).horizontal(horizontalRulingLines).build();
}
public void snapPoints(List<? extends Line2D.Float> rulings, float xThreshold, float yThreshold) {
// collect points and keep a Line -> p1,p2 map
@@ -122,12 +120,14 @@ public class RulingCleaningService {
private List<Ruling> collapseOrientedRulings(List<Ruling> lines) {
int COLINEAR_OR_PARALLEL_PIXEL_EXPAND_AMOUNT = 1;
return collapseOrientedRulings(lines, COLINEAR_OR_PARALLEL_PIXEL_EXPAND_AMOUNT);
}
private List<Ruling> collapseOrientedRulings(List<Ruling> lines, int expandAmount) {
ArrayList<Ruling> rv = new ArrayList<>();
lines.sort((a, b) -> {
final float diff = a.getPosition() - b.getPosition();
@@ -141,25 +141,24 @@ public class RulingCleaningService {
final float lastStart = last.getStart();
final float lastEnd = last.getEnd();
final boolean lastFlipped = lastStart > lastEnd;
final boolean lastFlipped = lastStart > lastEnd;
final boolean nextFlipped = next_line.getStart() > next_line.getEnd();
boolean differentDirections = nextFlipped != lastFlipped;
float nextS = differentDirections ? next_line.getEnd() : next_line.getStart();
float nextS = differentDirections ? next_line.getEnd() : next_line.getStart();
float nextE = differentDirections ? next_line.getStart() : next_line.getEnd();
final float newStart = lastFlipped ? Math.max(nextS, lastStart) : Math.min(nextS, lastStart);
final float newEnd = lastFlipped ? Math.min(nextE, lastEnd) : Math.max(nextE, lastEnd);
final float newEnd = lastFlipped ? Math.min(nextE, lastEnd) : Math.max(nextE, lastEnd);
last.setStartEnd(newStart, newEnd);
assert !last.oblique();
}
else if (next_line.length() == 0) {
} else if (next_line.length() == 0) {
continue;
}
else {
} else {
rv.add(next_line);
}
}
return rv;
}
}
@@ -1,27 +1,43 @@
package com.iqser.red.service.redaction.v1.server;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.RANDOM_PORT;
import java.awt.Color;
import java.awt.geom.AffineTransform;
import java.awt.geom.Rectangle2D;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.UUID;
import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.text.PDFTextStripperByArea;
import org.apache.pdfbox.util.Matrix;
import org.junit.Before;
import org.junit.Ignore;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
@@ -37,28 +53,44 @@ import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.UUID;
import java.util.stream.Collectors;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.RANDOM_PORT;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.AnalyzeRequest;
import com.iqser.red.service.redaction.v1.model.AnalyzeResult;
import com.iqser.red.service.redaction.v1.model.AnnotateRequest;
import com.iqser.red.service.redaction.v1.model.AnnotateResponse;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualForceRedact;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.ReanalyzeResult;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.RenalyzeRequest;
import com.iqser.red.service.redaction.v1.model.SectionArea;
import com.iqser.red.service.redaction.v1.model.SectionText;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.parsing.PDFAreaTextStripper;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.ReanalysisSection;
import com.iqser.red.service.redaction.v1.server.redaction.utils.ResourceLoader;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
@RunWith(SpringRunner.class)
@SpringBootTest(webEnvironment = RANDOM_PORT)
@@ -75,6 +107,8 @@ public class RedactionIntegrationTest {
private static final String MUST_REDACT = "must_redact";
private static final String PUBLISHED_INFORMATION = "published_information";
private static final String TEST_METHOD = "test_method";
private static final String PURITY = "purity";
private static final String IMAGE = "image";
private static final String RECOMMENDATION_AUTHOR = "recommendation_CBI_author";
private static final String RECOMMENDATION_ADDRESS = "recommendation_CBI_address";
@@ -86,6 +120,9 @@ public class RedactionIntegrationTest {
@Autowired
private RedactionController redactionController;
@Autowired
private ObjectMapper objectMapper;
@MockBean
private RulesClient rulesClient;
@@ -97,6 +134,7 @@ public class RedactionIntegrationTest {
private final Map<String, Boolean> hintTypeMap = new HashMap<>();
private final Map<String, Boolean> caseInSensitiveMap = new HashMap<>();
private final Map<String, Boolean> recommendationTypeMap = new HashMap<>();
private final Map<String, Integer> rankTypeMap = new HashMap<>();
private final Colors colors = new Colors();
private final static String TEST_RULESET_ID = "123";
@@ -132,7 +170,9 @@ public class RedactionIntegrationTest {
loadDictionaryForTest();
loadTypeForTest();
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(0L);
when(dictionaryClient.getAllTypes(TEST_RULESET_ID)).thenReturn(TypeResponse.builder().types(getTypeResponse()).build());
when(dictionaryClient.getAllTypes(TEST_RULESET_ID)).thenReturn(TypeResponse.builder()
.types(getTypeResponse())
.build());
when(dictionaryClient.getDictionaryForType(VERTEBRATE, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(VERTEBRATE));
when(dictionaryClient.getDictionaryForType(ADDRESS, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(ADDRESS));
when(dictionaryClient.getDictionaryForType(AUTHOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(AUTHOR));
@@ -147,6 +187,8 @@ public class RedactionIntegrationTest {
when(dictionaryClient.getDictionaryForType(RECOMMENDATION_AUTHOR, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(RECOMMENDATION_AUTHOR));
when(dictionaryClient.getDictionaryForType(RECOMMENDATION_ADDRESS, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(RECOMMENDATION_ADDRESS));
when(dictionaryClient.getDictionaryForType(FALSE_POSITIVE, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(FALSE_POSITIVE));
when(dictionaryClient.getDictionaryForType(PURITY, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(PURITY));
when(dictionaryClient.getDictionaryForType(IMAGE, TEST_RULESET_ID)).thenReturn(getDictionaryResponse(IMAGE));
when(dictionaryClient.getColors(TEST_RULESET_ID)).thenReturn(colors);
}
@@ -223,6 +265,16 @@ public class RedactionIntegrationTest {
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(PURITY, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/purity.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(IMAGE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/image.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
}
@@ -248,6 +300,8 @@ public class RedactionIntegrationTest {
typeColorMap.put(RECOMMENDATION_AUTHOR, "#8df06c");
typeColorMap.put(RECOMMENDATION_ADDRESS, "#8df06c");
typeColorMap.put(FALSE_POSITIVE, "#ffffff");
typeColorMap.put(PURITY, "#ffe187");
typeColorMap.put(IMAGE, "#fcc5fb");
hintTypeMap.put(VERTEBRATE, true);
hintTypeMap.put(ADDRESS, false);
@@ -263,6 +317,8 @@ public class RedactionIntegrationTest {
hintTypeMap.put(RECOMMENDATION_AUTHOR, false);
hintTypeMap.put(RECOMMENDATION_ADDRESS, false);
hintTypeMap.put(FALSE_POSITIVE, true);
hintTypeMap.put(PURITY, false);
hintTypeMap.put(IMAGE, true);
caseInSensitiveMap.put(VERTEBRATE, true);
caseInSensitiveMap.put(ADDRESS, false);
@@ -278,6 +334,8 @@ public class RedactionIntegrationTest {
caseInSensitiveMap.put(RECOMMENDATION_AUTHOR, false);
caseInSensitiveMap.put(RECOMMENDATION_ADDRESS, false);
caseInSensitiveMap.put(FALSE_POSITIVE, false);
caseInSensitiveMap.put(PURITY, false);
caseInSensitiveMap.put(IMAGE, true);
recommendationTypeMap.put(VERTEBRATE, false);
recommendationTypeMap.put(ADDRESS, false);
@@ -293,6 +351,25 @@ public class RedactionIntegrationTest {
recommendationTypeMap.put(RECOMMENDATION_AUTHOR, true);
recommendationTypeMap.put(RECOMMENDATION_ADDRESS, true);
recommendationTypeMap.put(FALSE_POSITIVE, false);
recommendationTypeMap.put(PURITY, false);
recommendationTypeMap.put(IMAGE, false);
rankTypeMap.put(FALSE_POSITIVE, 160);
rankTypeMap.put(PURITY, 155);
rankTypeMap.put(PII, 150);
rankTypeMap.put(ADDRESS, 140);
rankTypeMap.put(AUTHOR, 130);
rankTypeMap.put(SPONSOR, 120);
rankTypeMap.put(VERTEBRATE, 110);
rankTypeMap.put(MUST_REDACT, 100);
rankTypeMap.put(REDACTION_INDICATOR, 90);
rankTypeMap.put(NO_REDACTION_INDICATOR, 80);
rankTypeMap.put(PUBLISHED_INFORMATION, 70);
rankTypeMap.put(TEST_METHOD, 60);
rankTypeMap.put(HINT_ONLY, 50);
rankTypeMap.put(RECOMMENDATION_AUTHOR, 40);
rankTypeMap.put(RECOMMENDATION_ADDRESS, 30);
rankTypeMap.put(IMAGE, 30);
colors.setDefaultColor("#acfc00");
colors.setNotRedacted("#cccccc");
@@ -312,6 +389,7 @@ public class RedactionIntegrationTest {
.isHint(hintTypeMap.get(typeColor.getKey()))
.isCaseInsensitive(caseInSensitiveMap.get(typeColor.getKey()))
.isRecommendation(recommendationTypeMap.get(typeColor.getKey()))
.rank(rankTypeMap.get(typeColor.getKey()))
.build())
.collect(Collectors.toList());
@@ -326,6 +404,7 @@ public class RedactionIntegrationTest {
.isHint(hintTypeMap.get(type))
.isCaseInsensitive(caseInSensitiveMap.get(type))
.isRecommendation(recommendationTypeMap.get(type))
.rank(rankTypeMap.get(type))
.build();
}
@@ -343,12 +422,12 @@ public class RedactionIntegrationTest {
input.addAll(getPathsRecursively(file));
}
for (File path : input) {
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(new FileInputStream(path)))
.build();
System.out.println("Redacting file : " + path.getName());
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
Map<String, List<RedactionLogEntry>> duplicates = new HashMap<>();
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
@@ -358,6 +437,23 @@ public class RedactionIntegrationTest {
duplicates.entrySet().forEach(entry -> {
assertThat(entry.getValue().size()).isEqualTo(1);
});
dictionary.get(AUTHOR).add("Drinking water");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(1L);
long rstart = System.currentTimeMillis();
ReanalyzeResult reanalyzeResult = redactionController.reanalyze(RenalyzeRequest.builder()
.redactionLog(result.getRedactionLog())
.document(IOUtils.toByteArray(new FileInputStream(path)))
.manualRedactions(null)
.text(result.getText())
.ruleSetId(TEST_RULESET_ID)
.build());
long rend = System.currentTimeMillis();
System.out.println("reanalysis analysis duration: " + (rend - rstart));
}
long end = System.currentTimeMillis();
@@ -390,32 +486,194 @@ public class RedactionIntegrationTest {
System.out.println("redactionTest");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Applicant Producer Table.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
// result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
// if(!entry.isHint()){
// System.out.println(entry.getPositions().get(0).getPage() +":"+ entry.getTextBefore() +"--->"+ entry.getValue() + "--->" + entry.getTextAfter());
// }
// });
long end = System.currentTimeMillis();
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
System.out.println("first analysis duration: " + (end - start));
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Test.json")) {
fileOutputStream.write(objectMapper.writeValueAsBytes(result.getText()));
}
int correctFound = 0;
loop:
for (RedactionLogEntry redactionLogEntry : result.getRedactionLog().getRedactionLogEntry()) {
for (SectionText sectionText : result.getText().getSectionTexts()) {
if (redactionLogEntry.getType().equals("image")) {
correctFound++;
continue loop;
}
if (redactionLogEntry.getSectionNumber() == sectionText.getSectionNumber()) {
String value = sectionText.getText()
.substring(redactionLogEntry.getStartOffset(), redactionLogEntry.getEndOffset());
if (redactionLogEntry.getValue().equalsIgnoreCase(value)) {
correctFound++;
} else {
throw new RuntimeException("WTF");
}
}
}
}
assertThat(correctFound).isEqualTo(result.getRedactionLog().getRedactionLogEntry().size());
dictionary.get(AUTHOR).add("Properties");
when(dictionaryClient.getVersion(TEST_RULESET_ID)).thenReturn(1L);
start = System.currentTimeMillis();
ReanalyzeResult reanalyzeResult = redactionController.reanalyze(RenalyzeRequest.builder()
.redactionLog(result.getRedactionLog())
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.manualRedactions(null)
.text(result.getText())
.ruleSetId(TEST_RULESET_ID)
.build());
end = System.currentTimeMillis();
System.out.println("reanalysis analysis duration: " + (end - start));
AnnotateResponse annotateResponse = redactionController.annotate(AnnotateRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.redactionLog(reanalyzeResult.getRedactionLog())
.sectionGrid(result.getSectionGrid())
.build());
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated.pdf")) {
fileOutputStream.write(annotateResponse.getDocument());
}
}
@Test
@Ignore
public void fillRecanTest() throws IOException {
System.out.println("redactionTest");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/S5.pdf");
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
AnalyzeResult result = redactionController.analyze(request);
AnnotateResponse annotateResponse = redactionController.annotate(AnnotateRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.redactionLog(result.getRedactionLog())
.sectionGrid(result.getSectionGrid())
.build());
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated.pdf")) {
fileOutputStream.write(annotateResponse.getDocument());
}
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Test.json")) {
fileOutputStream.write(objectMapper.writeValueAsBytes(result.getText()));
}
int correctFound = 0;
loop:
for (RedactionLogEntry redactionLogEntry : result.getRedactionLog().getRedactionLogEntry()) {
for (SectionText sectionText : result.getText().getSectionTexts()) {
if (redactionLogEntry.getType().equals("image")) {
correctFound++;
continue loop;
}
if (redactionLogEntry.getSectionNumber() == sectionText.getSectionNumber()) {
String value = sectionText.getText()
.substring(redactionLogEntry.getStartOffset(), redactionLogEntry.getEndOffset());
if (redactionLogEntry.getValue().equalsIgnoreCase(value)) {
correctFound++;
} else {
throw new RuntimeException("WTF");
}
}
}
}
assertThat(correctFound).isEqualTo(result.getRedactionLog().getRedactionLogEntry().size());
System.out.println("correctFound " + correctFound);
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
SectionArea sectionArea = result.getText().getSectionTexts().get(3).getSectionAreas().get(5);
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(IOUtils.toByteArray(pdfFileResource.getInputStream())))) {
PDPage docPage = pdDocument.getPage(0);
PDFTextStripperByArea textStripper = new PDFTextStripperByArea();
PDRectangle cropBox = docPage.getCropBox();
PDRectangle mediaBox = docPage.getMediaBox();
// if (textPositions.get(0).getRotation() == 90) {
// posXEnd = textPositions.get(0).getYDirAdj() + 2;
// posYInit = getY1();
// posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
// } else {
// posXEnd = textPositions.get(textPositions.size() - 1)
// .getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
// posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
// posYEnd = textPositions.get(0).getPageHeight() - textPositions.get(textPositions.size() - 1)
// .getYDirAdj() + 2;
// }
Rectangle2D rect = new Rectangle2D.Float(sectionArea.getTopLeft()
.getY(), sectionArea.getTopLeft()
.getX() , sectionArea.getHeight(), sectionArea
.getWidth() + 0.001f);
textStripper.addRegion("region", rect);
textStripper.extractRegions(docPage);
String textForRegion = textStripper.getTextForRegion("region");
System.out.println(textForRegion);
// fill a rectangle
PDPageContentStream contents = new PDPageContentStream (pdDocument, docPage, PDPageContentStream.AppendMode.APPEND, false, false);
contents.setNonStrokingColor (Color.RED);
contents.addRect (sectionArea.getTopLeft().getX(), sectionArea.getTopLeft().getY(), sectionArea.getWidth(), sectionArea.getHeight());
contents.fill ();
contents.close ();
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
pdDocument.save(byteArrayOutputStream);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated2.pdf")) {
fileOutputStream.write(byteArrayOutputStream.toByteArray());
}
}
} catch (Exception e) {
throw new RedactionException(e);
}
}
@Test
public void testTableRedaction() throws IOException {
@@ -423,15 +681,21 @@ public class RedactionIntegrationTest {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
AnnotateResponse annotateResponse = redactionController.annotate(AnnotateRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.redactionLog(result.getRedactionLog())
.sectionGrid(result.getSectionGrid())
.build());
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated.pdf")) {
fileOutputStream.write(annotateResponse.getDocument());
}
long end = System.currentTimeMillis();
@@ -443,6 +707,8 @@ public class RedactionIntegrationTest {
@Test
public void testManualRedaction() throws IOException {
// 675eba69b0c2917de55462c817adaa05
System.out.println("testManualRedaction");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
@@ -457,8 +723,13 @@ public class RedactionIntegrationTest {
.text("This is a comment test")
.build();
manualRedactions.setIdsToRemove(Set.of(IdRemoval.builder()
.id("0836727c3508a0b2ea271da69c04cc2f")
.status(Status.REQUESTED)
.id("5b940b2cb401ed9f5be6fc24f6e77bcf")
.status(Status.DECLINED)
.build()));
manualRedactions.setForceRedacts(Set.of(ManualForceRedact.builder()
.id("675eba69b0c2917de55462c817adaa05")
.legalBasis("Something")
.status(Status.APPROVED)
.build()));
manualRedactions.getComments().put("e5be0f1d941bbb92a068e198648d06c4", List.of(comment));
@@ -473,18 +744,40 @@ public class RedactionIntegrationTest {
manualRedactionEntry.setReason("Manual Redaction");
manualRedactionEntry.setPositions(List.of(new Rectangle(new Point(375.61096f, 241.282f), 7.648041f, 43.72262f, 1), new Rectangle(new Point(384.83517f, 241.282f), 7.648041f, 17.043358f, 1)));
manualRedactions.getEntriesToAdd().add(manualRedactionEntry);
// manualRedactions.getEntriesToAdd().add(manualRedactionEntry);
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.manualRedactions(manualRedactions)
.build();
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
manualRedactions.getEntriesToAdd().add(manualRedactionEntry);
manualRedactions.setIdsToRemove(Set.of(IdRemoval.builder()
.id("5b940b2cb401ed9f5be6fc24f6e77bcf")
.status(Status.APPROVED)
.build()));
ReanalyzeResult reanalyzeResult = redactionController.reanalyze(RenalyzeRequest.builder()
.redactionLog(result.getRedactionLog())
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.manualRedactions(manualRedactions)
.text(result.getText())
.ruleSetId(TEST_RULESET_ID)
.build());
AnnotateResponse annotateResponse = redactionController.annotate(AnnotateRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.redactionLog(reanalyzeResult.getRedactionLog())
.sectionGrid(result.getSectionGrid())
.build());
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated.pdf")) {
fileOutputStream.write(annotateResponse.getDocument());
}
long end = System.currentTimeMillis();
@@ -533,7 +826,7 @@ public class RedactionIntegrationTest {
public void htmlTablesTest() throws IOException {
System.out.println("htmlTablesTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/52 Fludioxonil_RAR_07_Volume_3CA_B-5_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -570,13 +863,12 @@ public class RedactionIntegrationTest {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Phantom Cells.pdf");
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
if (!entry.isHint()) {
@@ -592,16 +884,21 @@ public class RedactionIntegrationTest {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/sponsor_companies.pdf");
RedactionRequest request = RedactionRequest.builder()
AnalyzeRequest request = AnalyzeRequest.builder()
.ruleSetId(TEST_RULESET_ID)
.flatRedaction(false)
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
RedactionResult result = redactionController.redact(request);
AnalyzeResult result = redactionController.analyze(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
AnnotateResponse annotateResponse = redactionController.annotate(AnnotateRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.redactionLog(result.getRedactionLog())
.sectionGrid(result.getSectionGrid())
.build());
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Annotated.pdf")) {
fileOutputStream.write(annotateResponse.getDocument());
}
long end = System.currentTimeMillis();
@@ -16,6 +16,7 @@ import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationSer
import org.apache.commons.io.IOUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.Before;
import org.junit.Ignore;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.kie.api.KieServices;
@@ -412,6 +413,7 @@ public class EntityRedactionServiceTest {
@Test
@Ignore
public void testNGuideline() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Empty Tabular Data.pdf");
@@ -1,15 +1,36 @@
package com.iqser.red.service.redaction.v1.server.redaction.utils;
import org.apache.commons.lang3.StringUtils;
import org.junit.Test;
import java.util.ArrayList;
import java.util.List;
import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
public class RegExPatternTest {
@Test
public void testExpand(){
String pattern = "(,? [A-Z]\\.?( ?[A-Z]\\.?)?( ?[A-Z]\\.?)?\\b\\.?)";
String text = ", G.R., Simoneaux,";
Pattern compiledPattern = Pattern.compile(pattern, 0);
Matcher matcher = compiledPattern.matcher(text);
while (matcher.find()) {
String match = matcher.group(1);
if (StringUtils.isNotBlank(match)) {
System.out.println(match);
}
}
}
@Test
public void testEmailRegEx(){
String text = "Address: Schwarzwaldalle " +
@@ -1,10 +1,238 @@
Long-term
(Parent
ADME - Bioanalyses, Vergèze,
Aeration: Gentle
Andrews mean
Andrews pKi
BASF study code
Billington –
Bond elute
Brown Liquid
Brown Orange
Brown Solid
Brown disco
Brown discoloured
Brown fat
Brown liquid
Brown solid
Buehler (1980),
Buehler (1980).
Buehler 9
Buehler Test.
Buehler assay
Buehler test
Buehler)
Buehler-like :
Calandra lark
Cary, NC,
Chambersas for
Class 4
Class III’
Class III’)
Class method,
Class2 (effectively
ClassMethod) Vischim
Clay =
Cochran-Armitage Test.
Cochran-armitage test.
Dunnett 2
Dunnett's "t''
Dunnett's Test
Dunnett's multiple
Dunnett's t-test
Dunnett's test
Dunnett) **
Dunnett) Dose
Dunnett) Food
Dunnett) Lower
Dunnett, 1964)
Dunnett’s Multiple
Dunnett’s T
Dunnett’s Test)
Dunnett’s adjustment
Dunnett’s multiple
Dunnett’s post
Dunnett’s t-test)
Fine vacuolation
Fine white
Finewhite powder
Fischer 344
Fischer 344)
Fischer F344
Fischer rat
Fischer) Contaminants:
Fischer-344 acceptability
Fischer-344/
Fischer-344/DuCrj acceptability
Fischer344 rats.
Fischer344) :
Fischerrat embryo
Fischer’s exact
Fisher 344
Fisher Exact Test
Fisher Exact test
Fisher exact test
Fisher's Exact
Fisher-344 rat
Fisher344 (F344/
Fisher344 rat
Fisher344 rats
Fisher344 rats.
Fisher344 rats/
Fisher344) group
Fisher’s exact.
Freeman-Tukey transform
Gentle aeration
Grade 2:
Grade 3)
Grade 3:
Green
Green Alga
Green as
Green-, blue-,
GreenAlga, Selenastrum
GreenAlga.. Springborn
Gross External
Gross Necropsy
Gross and
Gross necropsies
Gross necropsy
Gross pathology
Gross pathology:
Hall Farm
Hall Farm)
HallFarm (396
Ham's F12
Ham’s F10
Hand-held
HarlanWinkelmann, Borchem,
Hill International,
Hill Top
Hill criteria
Hill, Bracknell,
HillTop Chamber,
Hillmodel was
Hill’s model
Japanese White
Jerneplaque assay
Kent. Blood
Klimisch score
Litchfield and Wilcoxon
Long term
Long-Term Toxicity
Long-term
Longbw/
Major Macroscopic findings
Major biliary
Major defects
Major deviations
Major fetal
Major metabolic
Major metabolites
Major route
Major treatment-related
Mallard (Anas
Mallard with
Manual-Hand held
Manual-Hand held
Weight:
Marsh Frog
Marsh harrier
Masonjars (500
Meier formula)
Meier survival
Miles anti-Canine
Miles anti-dog
Netherlands, 2011
Netherlands, published:
Netherlands. Report
Penn State
Porewater:
Rattray study
Read across
Reed Bunting
Reichert-Jung Cryo
RifconGmbH. Oxon
Ritz and
S.P.-U/
San Diego,
Shaw's Farm,
Sprague
Strain JapaneseWhite
Strong morphological
Weight -
Weight 2744
Weight Gain
Weight On
Weight Range
Weight and
Weight and length
Aeration: Gentle
Weight at
Weight gain
Weight loss
Weight of
Weight only
Weight range
Weight, age
Weight, feed
Weight-of-evidence approach
Weight:
Weight; Adult
Weight; Whole
WeightGain 0-1
WeightRange:
Weightat dosing:
Weightrange 200-250g.
White (Tif:
White 10
White 3
White Age
White Powder
White Strain
White blood
White cell
White crystalline
White guideline
White guinea-pigs
White hygroscopic
White powder
White solid
White stork
White swan
White to off-white
White wagtail
White, solid
WhitePowder Lot/
Whiteguideline :
Whitehygroscopic solid
Whitepowder Lot/
Whitepowder Stability
Whitesolid Batch
Whiteswan)
Whiteto off-white
Wildlife International specifications
Wildlife Internationalspecifications Water:
Williams E
Williams medium
Williams' medium
WilliamsE medium
Williams’ test
Wilson's method.
Wilson. All foetal
Wilson.All foetal
Wood pigeons
York.
Zeneca dossier);
Zenecadossier); see
green algae
CTL
No details reported
Not given
Not indicated
Not reported
Not reported.
Not stated
Syngenta
Syngenta AG
N/A
No details reported
Not available
Test facility
TBD
@@ -0,0 +1,261 @@
package drools
import com.iqser.red.service.redaction.v1.server.redaction.model.Section
global Section section
// --------------------------------------- CBI rules -------------------------------------------------------------------
rule "0: Expand CBI Authors with firstname initials"
when
Section(matchesType("CBI_author") || matchesType("recommendation_CBI_author"))
then
section.expandByRegEx("CBI_author", "(,? [A-Z]\\.?( ?[A-Z]\\.?)?( ?[A-Z]\\.?)?\\b\\.?)", false, 1);
section.expandByRegEx("recommendation_CBI_author", "(,? [A-Z]\\.?( ?[A-Z]\\.?)?( ?[A-Z]\\.?)?\\b\\.?)", false, 1);
end
rule "1: Redact CBI Authors"
when
Section(matchesType("CBI_author") || matchesType("recommendation_CBI_author"))
then
section.redact("CBI_author", 1, "Author found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "2: Redact CBI Address"
when
Section(matchesType("CBI_address") || matchesType("recommendation_CBI_address"))
then
section.redact("CBI_address", 2, "Address found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "3: Do not redact genitive CBI_author"
when
Section(matchesType("CBI_author") || matchesType("recommendation_CBI_author"))
then
section.expandToHintAnnotationByRegEx("CBI_author", "['’’'ʼˈ´`‘′ʻ’']s", false, 0, "test_method");
section.expandToHintAnnotationByRegEx("recommendation_CBI_author", "['’’'ʼˈ´`‘′ʻ’']s", false, 0, "test_method");
end
rule "4: Redact Author(s) cells in Tables with Author(s) header"
when
Section(hasTableHeader("Author(s)") && !hasTableHeader("Vertebrate study Y/N"))
then
section.redactCell("Author(s)", 4, "CBI_author", false, "Author found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "5: Redact Author cells in Tables with Author header"
when
Section(hasTableHeader("Author") && !hasTableHeader("Vertebrate study Y/N"))
then
section.redactCell("Author", 5, "CBI_author", false, "Author found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "6: Redact and recommand Authors in Tables with Vertebrate study Y/N header"
when
Section(rowEquals("Vertebrate study Y/N", "Y") || rowEquals("Vertebrate study Y/N", "Yes") || rowEquals("Vertebrate study Y/N", "N") || rowEquals("Vertebrate study Y/N", "No"))
then
section.redactCell("Author(s)", 6, "CBI_author", true, "Author found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "7: Redact if CTL/* or BL/* was found"
when
Section(searchText.contains("CTL/") || searchText.contains("BL/"))
then
section.addRedaction("CTL", "must_redact", 7, "Laboratory for vertebrate studies found", "Article 39(1)(2) of Regulation (EC) No 178/2002" );
section.addRedaction("BL", "must_redact", 7, "Laboratory for vertebrate studies found", "Article 39(1)(2) of Regulation (EC) No 178/2002" );
end
rule "8: Redact and add recommendation for et al. author"
when
Section(searchText.contains("et al"))
then
section.redactAndRecommendByRegEx("\\b([A-ZÄÖÜ][^\\s\\.,]+( [A-ZÄÖÜ]{1,2}\\.?)?( ?[A-ZÄÖÜ]\\.?)?) et al\\.?", false, 1, "CBI_author", 8, "Author found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "9: Add recommendation for Addresses in Test Organism sections"
when
Section(searchText.contains("Species:") && searchText.contains("Source:"))
then
section.recommendLineAfter("Source:", "CBI_address");
end
rule "10: Add recommendation for Addresses in Test Animals sections"
when
Section(searchText.contains("Species") && searchText.contains("Source"))
then
section.recommendLineAfter("Source", "CBI_address");
end
rule "11: Do not redact Names and Addresses if Published Information found"
when
Section(matchesType("published_information"))
then
section.redactNot("CBI_author", 11, "Published Information found");
section.redactNot("CBI_address", 11, "Published Information found");
end
// --------------------------------------- PII rules -------------------------------------------------------------------
rule "12: Redacted PII Personal Identification Information"
when
Section(matchesType("PII"))
then
section.redact("PII", 12, "PII (Personal Identification Information) found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "13: Redact Emails by RegEx"
when
Section(searchText.contains("@"))
then
section.redactByRegEx("\\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,4}\\b", true, 0, "PII", 13, "PII (Personal Identification Information) found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "14: Redact contact information"
when
Section(text.contains("Contact point:")
|| text.contains("Phone:")
|| text.contains("Fax:")
|| text.contains("Tel.:")
|| text.contains("Tel:")
|| text.contains("E-mail:")
|| text.contains("Email:")
|| text.contains("e-mail:")
|| text.contains("E-mail address:")
|| text.contains("Alternative contact:")
|| text.contains("Telephone number:")
|| text.contains("Telephone No:")
|| text.contains("Fax number:")
|| text.contains("Telephone:")
|| text.contains("Phone No.")
|| text.contains("European contact:"))
then
section.redactLineAfter("Contact point:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Tel.:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Tel:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("E-mail:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Email:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("e-mail:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("E-mail address:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Contact:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Alternative contact:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone number:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone No:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax number:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone No.", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactBetween("No:", "Fax", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactBetween("Contact:", "Tel.:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("European contact:", "PII", 14, true, "Contact information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "15: Redact contact information if applicant is found"
when
Section(headlineContainsWord("applicant") || text.contains("Applicant") || headlineContainsWord("Primary contact") || headlineContainsWord("Alternative contact") || text.contains("Telephone number:"))
then
section.redactLineAfter("Contact point:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Tel.:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Tel:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("E-mail:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Email:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("e-mail:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("E-mail address:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Contact:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Alternative contact:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone number:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone No:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax number:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone No.", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactBetween("No:", "Fax", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactBetween("Contact:", "Tel.:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("European contact:", "PII", 15, true, "Applicant information was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "16: Redact contact information if Producer is found"
when
Section(text.toLowerCase().contains("producer of the plant protection") || text.toLowerCase().contains("producer of the active substance") || text.contains("Manufacturer of the active substance") || text.contains("Manufacturer:") || text.contains("Producer or producers of the active substance"))
then
section.redactLineAfter("Contact:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("E-mail:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Contact:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Fax number:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Telephone number:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Tel:", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactLineAfter("Phone No.", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
section.redactBetween("No:", "Fax", "PII", 16, true, "Producer was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "17: Redact AUTHOR(S)"
when
Section(searchText.contains("AUTHOR(S):"))
then
section.redactLinesBetween("AUTHOR(S):", "COMPLETION DATE:", "PII", 17, true, "AUTHOR(S) was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "18: Redact PERFORMING LABORATORY"
when
Section(searchText.contains("PERFORMING LABORATORY:"))
then
section.redactBetween("PERFORMING LABORATORY:", "LABORATORY PROJECT ID:", "PII", 18, true, "PERFORMING LABORATORY was found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "19: Redact On behalf of Sequani Ltd.:"
when
Section(searchText.contains("On behalf of Sequani Ltd.: Name Title"))
then
section.redactBetween("On behalf of Sequani Ltd.: Name Title", "On behalf of", "PII", 19, false , "PII (Personal Identification Information) found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
rule "20: Redact On behalf of Syngenta Ltd.:"
when
Section(searchText.contains("On behalf of Syngenta Ltd.: Name Title"))
then
section.redactBetween("On behalf of Syngenta Ltd.: Name Title", "Study dates", "PII", 20, false , "PII (Personal Identification Information) found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
// --------------------------------------- other rules -------------------------------------------------------------------
rule "21: Purity Hint"
when
Section(searchText.toLowerCase().contains("purity"))
then
section.addHintAnnotationByRegEx("(purity ?( of|\\(.{1,20}\\))?( ?:)?) .{0,5}[\\d\\.]+( .{0,4}\\.)? ?%", true, 1, "hint_only");
end
rule "22: Redact Must Redact"
when
Section(matchesType("must_redact"))
then
section.redact("must_redact", 22, "Must Redact found", "Article 39(1)(2) of Regulation (EC) No 178/2002");
end
@@ -7,6 +7,15 @@ global Section section
// --------------------------------------- CBI rules -------------------------------------------------------------------
rule "0: Expand CBI Authors with firstname initials"
when
Section(matchesType("CBI_author") || matchesType("recommendation_CBI_author"))
then
section.expandByRegEx("CBI_author", "(,? [A-Z]\\.?( ?[A-Z]\\.?)?( ?[A-Z]\\.?)?\\b\\.?)", false, 1);
section.expandByRegEx("recommendation_CBI_author", "(,? [A-Z]\\.?( ?[A-Z]\\.?)?( ?[A-Z]\\.?)?\\b\\.?)", false, 1);
end
rule "1: Redacted because Section contains Vertebrate"
when
Section(matchesType("vertebrate"))
@@ -34,21 +43,21 @@ rule "3: Do not redact Names and Addresses if no redaction Indicator is containe
end
rule "4: Do not redact Names and Addresses if no redaction Indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("published_information"))
then
section.redactNot("CBI_author", 4, "Vertebrate and Published Information found");
section.redactNot("CBI_address", 4, "Vertebrate and Published Information found");
end
rule "5: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
rule "4: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("no_redaction_indicator"), matchesType("redaction_indicator"))
then
section.redact("CBI_author", 5, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 5, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_author", 4, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("CBI_address", 4, "Vertebrate and Redaction Indicator found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "5: Do not redact Names and Addresses if no redaction Indicator is contained"
when
Section(matchesType("vertebrate"), matchesType("published_information"))
then
section.redactNot("CBI_author", 5, "Vertebrate and Published Information found");
section.redactNot("CBI_address", 5, "Vertebrate and Published Information found");
end
@@ -134,33 +143,49 @@ rule "11: Redact if CTL/* or BL/* was found"
end
rule "12: Add recommendation for et al. author"
rule "12: Redact and add recommendation for et al. author"
when
Section(searchText.contains("et al."))
Section(searchText.contains("et al"))
then
section.addRecommendationByRegEx("\\b([A-ZÄÖÜ][^\\s\\.,]+( [A-ZÄÖÜ]\\.?)?( [A-ZÄÖÜ]\\.?)?) et al\\.?", false, 1, "CBI_author");
section.redactAndRecommendByRegEx("\\b([A-ZÄÖÜ][^\\s\\.,]+( [A-ZÄÖÜ]{1,2}\\.?)?( ?[A-ZÄÖÜ]\\.?)?) et al\\.?", false, 1, "CBI_author", 12, "Author found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "13: Add recommendation for Addresses in Test Organism sections"
when
Section(searchText.contains("Species:") && searchText.contains("Source:"))
then
section.recommendLineAfter("Source:", "CBI_address");
end
rule "14: Add recommendation for Addresses in Test Animals sections"
when
Section(searchText.contains("Species") && searchText.contains("Source"))
then
section.recommendLineAfter("Source", "CBI_address");
end
// --------------------------------------- PII rules -------------------------------------------------------------------
rule "13: Redacted PII Personal Identification Information"
rule "14: Redacted PII Personal Identification Information"
when
Section(matchesType("PII"))
then
section.redact("PII", 13, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redact("PII", 14, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "14: Redact Emails by RegEx"
rule "15: Redact Emails by RegEx"
when
Section(searchText.contains("@"))
then
section.redactByRegEx("\\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,4}\\b", true, 0, "PII", 14, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactByRegEx("\\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,4}\\b", true, 0, "PII", 15, "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "15: Redact contact information"
rule "16: Redact contact information"
when
Section(text.contains("Contact point:")
|| text.contains("Phone:")
@@ -178,96 +203,105 @@ rule "15: Redact contact information"
|| text.contains("Telephone:")
|| text.contains("European contact:"))
then
section.redactLineAfter("Contact point:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 15, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact point:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 16, true, "Contact information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "16: Redact contact information if applicant is found"
rule "17: Redact contact information if applicant is found"
when
Section(headlineContainsWord("applicant") || text.contains("Applicant") || headlineContainsWord("Primary contact") || headlineContainsWord("Alternative contact") || text.contains("Telephone number:"))
then
section.redactLineAfter("Contact point:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 16, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact point:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel.:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Email:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("e-mail:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail address:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Alternative contact:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone No:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("Contact:", "Tel.:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("European contact:", "PII", 17, true, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "17: Redact contact information if Producer is found"
rule "18: Redact contact information if Producer is found"
when
Section(text.toLowerCase().contains("producer of the plant protection") || text.toLowerCase().contains("producer of the active substance") || text.contains("Manufacturer of the active substance") || text.contains("Manufacturer:") || text.contains("Producer or producers of the active substance"))
then
section.redactLineAfter("Contact:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 17, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Phone:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("E-mail:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Contact:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Fax number:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Telephone number:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLineAfter("Tel:", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("No:", "Fax", "PII", 18, true, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "18: Redact AUTHOR(S)"
rule "19: Redact AUTHOR(S)"
when
Section(searchText.contains("AUTHOR(S):"))
then
section.redactLinesBetween("AUTHOR(S):", "COMPLETION DATE:", "PII", 18, true, "AUTHOR(S) was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactLinesBetween("AUTHOR(S):", "COMPLETION DATE:", "PII", 19, true, "AUTHOR(S) was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "19: Redact PERFORMING LABORATORY"
rule "20: Redact PERFORMING LABORATORY"
when
Section(searchText.contains("PERFORMING LABORATORY:"))
then
section.redactBetween("PERFORMING LABORATORY:", "LABORATORY PROJECT ID:", "PII", 19, true, "PERFORMING LABORATORY was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("PERFORMING LABORATORY:", "LABORATORY PROJECT ID:", "PII", 20, true, "PERFORMING LABORATORY was found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "20: Redact On behalf of Sequani Ltd.:"
rule "21: Redact On behalf of Sequani Ltd.:"
when
Section(searchText.contains("On behalf of Sequani Ltd.: Name Title"))
then
section.redactBetween("On behalf of Sequani Ltd.: Name Title", "On behalf of", "PII", 20, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("On behalf of Sequani Ltd.: Name Title", "On behalf of", "PII", 21, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
rule "21: Redact On behalf of Syngenta Ltd.:"
rule "22: Redact On behalf of Syngenta Ltd.:"
when
Section(searchText.contains("On behalf of Syngenta Ltd.: Name Title"))
then
section.redactBetween("On behalf of Syngenta Ltd.: Name Title", "Study dates", "PII", 21, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
section.redactBetween("On behalf of Syngenta Ltd.: Name Title", "Study dates", "PII", 22, false , "PII (Personal Identification Information) found", "Reg (EC) No 1107/2009 Art. 63 (2e)");
end
// --------------------------------------- other rules -------------------------------------------------------------------
rule "25: Redact Purity"
when
Section(searchText.contains("purity"))
then
section.redactByRegEx("purity ?:? (([\\d\\.]+)( .{0,4}\\.)? ?%)", true, 1, "purity", 17, "Purity found", "Reg (EC) No 1107/2009 Art. 63 (2a)");
end
@@ -10,7 +10,7 @@
<Root level="warn">
<AppenderRef ref="CONSOLE" />
</Root>
<Logger name="com.iqser.gin4" level="info" />
<Logger name="com.iqser" level="info" />
</Loggers>
</Configuration>