Compare commits

...
Author SHA1 Message Date
deiflaender e178393c23 Fixed paragraph recognition 2020-11-16 16:00:45 +01:00
deiflaender f75aff5186 RED-692: Added matchedRule to RedactionLog 2020-11-16 11:55:58 +01:00
deiflaender 85d73dae47 RED-632: Set status DECLINED in RedactionLog for manual removals that are DECLINED 2020-11-12 16:14:45 +01:00
Dominique Eiflaender 1792eb554e Pull request #66: RED-629: Each annotation is one entry in the RedactionLog
Merge in RED/redaction-service from RED-629 to master

* commit '936683f94dd089d47b267cadb606c42b9e8b5515':
  RED-629: Each annotation is one entry in the RedactionLog
2020-11-11 15:39:08 +01:00
deiflaender 936683f94d RED-629: Each annotation is one entry in the RedactionLog 2020-11-11 15:29:28 +01:00
Dominique Eiflaender d2161e1b3f Pull request #65: RED-419: Avoid duplicate entries
Merge in RED/redaction-service from RED-419 to master

* commit 'efe49ac2c1ae8a27feb6e07c1b4614619a0b2ece':
  RED-419: Avoid duplicate entries
2020-11-05 12:34:20 +01:00
deiflaender efe49ac2c1 RED-419: Avoid duplicate entries 2020-11-05 12:27:24 +01:00
Dominique Eiflaender 61352b565d Pull request #64: RED-515: Added manualRedactionType to redactionLog, to know if it must added or not in pdftron-redaction-service if it is requested
Merge in RED/redaction-service from RED-515 to master

* commit 'c6d3e3a4cf8f34bd2f129e1ae15ce9b142fc284c':
  RED-515: Added manualRedactionType to redactionLog, to know if it must added or not in pdftron-redaction-service if it is requested
2020-11-04 15:54:20 +01:00
deiflaender c6d3e3a4cf RED-515: Added manualRedactionType to redactionLog, to know if it must added or not in pdftron-redaction-service if it is requested 2020-11-04 15:47:00 +01:00
Thierry Goeckel c650361ec2 Pull request #63: Fix sponsor companies rule and add corresponding test
Merge in RED/redaction-service from RED-473 to master

* commit '122d5a556e35b961ba2199280f6836221c6f768c':
  Bump conf service version and remove obsolete object init
  Fix test
  Fix rule in drools file, too
  Fix sponsor companies rule and add corresponding test
2020-11-04 11:59:15 +01:00
Thierry Göckel 122d5a556e Bump conf service version and remove obsolete object init 2020-11-04 11:49:13 +01:00
Thierry Göckel 355cb16679 Fix test 2020-11-03 21:10:17 +01:00
Thierry Göckel c285c384ce Fix rule in drools file, too 2020-11-03 20:48:23 +01:00
Thierry Göckel 02052fbb6a Fix sponsor companies rule and add corresponding test 2020-11-03 20:47:11 +01:00
Dominique Eiflaender 329ad98a21 Pull request #62: Temprorary exclude WebMvcMetricsAutoConfiguration till UNDERTOW-1743 is fixed
Merge in RED/redaction-service from ExcludeMetrics to master

* commit '843b32e68feae46f306c5e122ebfd98e6531fc86':
  Temprorary exclude WebMvcMetricsAutoConfiguration till UNDERTOW-1743 is fixed
2020-11-03 10:18:43 +01:00
deiflaender 843b32e68f Temprorary exclude WebMvcMetricsAutoConfiguration till UNDERTOW-1743 is fixed 2020-11-03 09:57:00 +01:00
Dominique Eiflaender 3371b825ef Pull request #61: Removed id prefix
Merge in RED/redaction-service from TimoTest to master

* commit '6e83040a50728c491667bdbe211636d245e0cf6a':
  Removed id prefix
2020-11-03 09:15:28 +01:00
deiflaender 6e83040a50 Removed id prefix 2020-10-30 16:18:44 +01:00
Dominique Eiflaender de66d2fda8 Pull request #60: RED-510: Do not add annotations for manual redactions that are approved and should be already in a dictionary
Merge in RED/redaction-service from RED-510 to master

* commit '1b2bd7fa209cba7e635bbaad7962396d931576ff':
  RED-510: Do not add annotations for manual redactions that are approved and should be already in a dictionary
2020-10-29 15:37:37 +01:00
deiflaender 1b2bd7fa20 RED-510: Do not add annotations for manual redactions that are approved and should be already in a dictionary 2020-10-29 15:16:30 +01:00
Dominique Eiflaender 013353d6ef Pull request #59: RED-499: Changed prefix for redactions, fixed legalbasis problem
Merge in RED/redaction-service from RED-499 to master

* commit '0015304eea9564d76765fcb1a76d2b6a293c1ebc':
  RED-499: Changed prefix for redactions, fixed legalbasis problem
2020-10-28 13:55:41 +01:00
deiflaender 0015304eea RED-499: Changed prefix for redactions, fixed legalbasis problem 2020-10-28 13:02:46 +01:00
Dominique Eiflaender 97257a9c08 Pull request #58: RED-427: Added possibility to add comments to all annotation
Merge in RED/redaction-service from RED-427 to master

* commit '612d67301bbf186f5144c0243f9a37fe21d3646c':
  RED-427: Added possibility to add comments to all annotation
2020-10-27 13:02:50 +01:00
deiflaender 612d67301b RED-427: Added possibility to add comments to all annotation 2020-10-27 12:32:11 +01:00
Thierry Goeckel 6986b14713 Pull request #57: RED-451 Legal Basis
Merge in RED/redaction-service from RED-451 to master

* commit '8be59e4ddf2432f47a57d50e2743886c56af3652':
  Add legal basis to redaction log entry
  Integrate legal basis to rules, redaction annotation and tests
2020-10-27 12:00:59 +01:00
Thierry Göckel 8be59e4ddf Add legal basis to redaction log entry 2020-10-26 15:03:21 +01:00
Thierry Göckel f0f748db1b Integrate legal basis to rules, redaction annotation and tests 2020-10-26 10:07:41 +01:00
Thierry Goeckel 04f0c29a49 Pull request #56: Streamline actuator settings
Merge in RED/redaction-service from actuator-settings to master

* commit 'b5192b7424745360dc2206b45ec7f7667b867e27':
  Streamline actuator settings
2020-10-22 09:43:11 +02:00
Thierry Göckel b5192b7424 Streamline actuator settings 2020-10-21 17:50:33 +02:00
Dominique Eiflaender 268dd0200e Pull request #55: RED-413: Added user to manual redactions, approved flag to Status.APPROVED
Merge in RED/redaction-service from RED-413 to master

* commit '638c4f9c65d35406bba1fd33e20c3bae8a6e6a13':
  RED-413: Added user to manual redactions, approved flag to Status.APPROVED
2020-10-12 12:32:30 +02:00
deiflaender 638c4f9c65 RED-413: Added user to manual redactions, approved flag to Status.APPROVED 2020-10-12 12:21:18 +02:00
Dominique Eiflaender 8f12695f16 Pull request #54: RED-414: Ids for manual redactions and comments must be created in file-management-service
Merge in RED/redaction-service from RED-414 to master

* commit '79db6e4bf9d821fe4716a6da31d3f6d6fcaff05b':
  RED-414: Ids for manual redactions and comments must be created in file-management-service
2020-10-09 12:17:38 +02:00
deiflaender 79db6e4bf9 RED-414: Ids for manual redactions and comments must be created in file-management-service 2020-10-09 12:06:19 +02:00
Dominique Eiflaender c8eac2ca3d Pull request #53: RED-240: Use HexColors instead of float[], RED-380: Load all system colors from dictionary-service
Merge in RED/redaction-service from RED-240 to master

* commit 'a01651b8278ed09f9409d6ec2e1615680bc846cb':
  RED-240: Use HexColors instead of float[], RED-380: Load all system colors from dictionary-service
2020-10-08 12:05:42 +02:00
deiflaender a01651b827 RED-240: Use HexColors instead of float[], RED-380: Load all system colors from dictionary-service 2020-10-08 11:51:35 +02:00
Dominique Eiflaender 943d40366e Pull request #52: RED-391: Support comments and approval of manual redactions, RED-385: Improve annotation IDs in PDFs
Merge in RED/redaction-service from RED-391 to master

* commit '365158a0084b2034aca91e6f77eb9df42bbcb246':
  RED-391: Support comments and approval of manual redactions, RED-385: Improve annotation IDs in PDFs
2020-10-05 15:36:06 +02:00
deiflaender 365158a008 RED-391: Support comments and approval of manual redactions, RED-385: Improve annotation IDs in PDFs 2020-10-05 15:14:23 +02:00
Thierry Goeckel e76b893c8d Pull request #51: RED-104
Merge in RED/redaction-service from RED-104 to master

* commit 'fa29f56d98d05fd348dd27e3bbb1098851c238bd':
  Test line break fix
2020-10-05 11:02:56 +02:00
Thierry Goeckel d91612a69a Pull request #50: Make sure section and table char indices match
Merge in RED/redaction-service from RED-381 to master

* commit '9a965dd68a72285af6bf6a28ec6aa67358c35dba':
  Make sure section and table char indices match
2020-10-05 11:02:44 +02:00
Thierry Göckel fa29f56d98 Test line break fix 2020-10-02 18:11:58 +02:00
Thierry Göckel 9a965dd68a Make sure section and table char indices match 2020-10-02 17:54:31 +02:00
Dominique Eiflaender 0f4838e60d Pull request #49: RED-373: Fixed multiline annotations view in pdftron
Merge in RED/redaction-service from RED-373-2 to master

* commit 'a4e012762e1adda234decc1137f221cfd70498d3':
  RED-373: Fixed multiline annotations view in pdftron
2020-10-01 15:31:25 +02:00
deiflaender a4e012762e RED-373: Fixed multiline annotations view in pdftron 2020-10-01 15:19:57 +02:00
Dominique Eiflaender 0c1d8ba33d Pull request #48: RED-372: Suffix cross page annotations with -1, -2, etc., to avoid duplicate annotation ids
Merge in RED/redaction-service from RED-372 to master

* commit 'ecdfd29803e09ac57cc74b19408db54d9a8247a7':
  RED-372: Suffix cross page annotations with -1, -2, etc., to avoid duplicate annotation ids
2020-10-01 13:01:24 +02:00
deiflaender ecdfd29803 RED-372: Suffix cross page annotations with -1, -2, etc., to avoid duplicate annotation ids 2020-10-01 12:48:44 +02:00
Dominique Eiflaender 107dbe1910 Pull request #47: RED-373: Use a single multiline annotation instead of seperate annotations per line for one Entity
Merge in RED/redaction-service from RED-373 to master

* commit 'dd02296face3c32bf7675301cf9969103e65f52f':
  Changed formating
  RED-373: Use a single multiline annotation instead of seperate annotations per line for one Entity
2020-10-01 12:47:58 +02:00
deiflaender dd02296fac Changed formating 2020-10-01 11:55:38 +02:00
Thierry Goeckel 589458a112 Pull request #46: Add rule redacting sponsor companies if preceded by prefix
Merge in RED/redaction-service from RED-301 to master

* commit '6a8f3665198ec2d7452ed0dac43b9a7d220c8bc5':
  Remove entries from must_redact dict, add test and refactor rules
  Add rule redacting sponsor companies if preceded by prefix
2020-10-01 11:44:41 +02:00
deiflaender ce22121802 RED-373: Use a single multiline annotation instead of seperate annotations per line for one Entity 2020-10-01 11:15:28 +02:00
Thierry Göckel 6a8f366519 Remove entries from must_redact dict, add test and refactor rules 2020-09-30 16:08:56 +02:00
Thierry Göckel cec5fd3d5e Add rule redacting sponsor companies if preceded by prefix 2020-09-30 14:13:17 +02:00
Dominique Eiflaender 99bde2956f Pull request #45: RED-277: Upgraded to pdfbox 2.0.21
Merge in RED/redaction-service from RED-277 to master

* commit 'ecc303e79e82ed698f1c8bdd37dbd521335428bc':
  RED-277: Upgraded to pdfbox 2.0.21
2020-09-29 16:17:32 +02:00
deiflaender ecc303e79e RED-277: Upgraded to pdfbox 2.0.21 2020-09-29 15:18:58 +02:00
Dominique Eiflaender 384cd1d0dc Pull request #44: Fix annotating single letter entities
Merge in RED/redaction-service from bugfix/single-letter-annotations to master

* commit '8e5ff318106e700e28abd14a0eb1337d3a44c109':
  Fix annotating single letter entities
2020-09-29 15:17:51 +02:00
Thierry Göckel 8e5ff31810 Fix annotating single letter entities 2020-09-29 15:10:52 +02:00
Dominique Eiflaender 8b1574845c Pull request #43: RED-264: Avoid phantom cells, by merging line till rounded value of biggest char height/width
Merge in RED/redaction-service from RED-264 to master

* commit 'e1adcfb02c4c7d46140884b3d38f8daf1e1aae92':
  Added missing unittest
  RED-264: Avoid phantom cells, by merging line till rounded value of biggest char height/width
2020-09-29 15:04:20 +02:00
deiflaender e1adcfb02c Added missing unittest 2020-09-29 14:54:35 +02:00
deiflaender 11164219c5 RED-264: Avoid phantom cells, by merging line till rounded value of biggest char height/width 2020-09-29 14:21:42 +02:00
Thierry Goeckel c9516205fe Pull request #42: RED-318
Merge in RED/redaction-service from RED-318 to master

* commit 'f5790bccab3fb8dce66aeb9fbe501655ae6b86c3':
  Log entries and manual redaction require one position per line
  Annotate line by line
  Refactor multiple positions to one
  Merge rectangles for single annotation/redaction log entry
2020-09-29 14:15:21 +02:00
Thierry Göckel f5790bccab Log entries and manual redaction require one position per line 2020-09-29 13:54:32 +02:00
Thierry Göckel 74e63ca292 Annotate line by line 2020-09-29 13:43:16 +02:00
Thierry Göckel 6638f0cf9e Refactor multiple positions to one 2020-09-28 17:47:55 +02:00
Thierry Göckel 5448153096 Merge rectangles for single annotation/redaction log entry 2020-09-28 17:31:40 +02:00
Dominique Eiflaender 03274ba1ab Pull request #41: RED-299: Redact complete Author(s) Field in Vertebrate study tables
Merge in RED/redaction-service from RED-299 to master

* commit '664b9b420605cb1adf1113b34c52a45e8024dbbc':
  RED-299: Redact complete Author(s) Field in Vertebrate study tables
2020-09-23 15:34:46 +02:00
deiflaender 664b9b4206 RED-299: Redact complete Author(s) Field in Vertebrate study tables 2020-09-23 15:20:52 +02:00
Thierry Goeckel 564d74e39d Pull request #40: Test modified rules 8 and 10
Merge in RED/redaction-service from RED-288 to master

* commit 'cfb7554c618b9482f529b8183bc0f1f95df7d00f':
  Test modified rules 8 and 9
2020-09-23 11:44:35 +02:00
Thierry Göckel cfb7554c61 Test modified rules 8 and 9 2020-09-22 21:36:05 +02:00
Christoph  Schabert ee40064b32 Pull request #39: RED-316: Switch to new red base image
Merge in RED/redaction-service from RED-316 to master

* commit 'f2fafc3db2ab8825eb4a2136e8eab28a12dbc2e6':
  Switch to rootless base-image
  Upgrade bamboo-specs to 7.1.2
2020-09-16 08:59:45 +02:00
Christoph  Schabert f2fafc3db2 Switch to rootless base-image 2020-09-15 16:02:08 +02:00
Christoph  Schabert d5b0638f11 Upgrade bamboo-specs to 7.1.2 2020-09-15 16:01:26 +02:00
Dominique Eiflaender 130cac0b77 Pull request #38: RED-293: Add unkown Textblocks to paragraphs
Merge in RED/redaction-service from RED-293 to master

* commit 'de310e8a654b601cb4bf258675904a25d4233783':
  RED-293: Add unkown Textblocks to paragraphs
2020-09-14 14:39:36 +02:00
deiflaender de310e8a65 RED-293: Add unkown Textblocks to paragraphs 2020-09-14 13:52:16 +02:00
Dominique Eiflaender a0a78440d8 Pull request #37: RED-305: Added ids to manual redaction entries, RED-306: set redacet to false in redactionLog for manual deletions
Merge in RED/redaction-service from RED-305 to master

* commit 'e52a54b0a3eb8491ed3e79b3f514364a5cf22f76':
  RED-305: Added ids to manual redaction entries, RED-306: set redacet to false in redactionLog for manual deletions
2020-09-11 11:18:42 +02:00
deiflaender e52a54b0a3 RED-305: Added ids to manual redaction entries, RED-306: set redacet to false in redactionLog for manual deletions 2020-09-11 11:10:10 +02:00
deiflaender 65a29fb3a5 Added new files and missing dictionary entries 2020-09-07 16:06:25 +02:00
Dominique Eiflaender ff0a73c635 Pull request #36: Added filename to redactionLog
Merge in RED/redaction-service from filename to master

* commit 'c27c9b2f5902d66f654982d7e5f005ecc176e01b':
  Added missing constuctor for RedactionLog
  Added filename to redactionLog
2020-09-02 15:22:34 +02:00
deiflaender c27c9b2f59 Added missing constuctor for RedactionLog 2020-09-02 14:56:48 +02:00
deiflaender d95ff5e5cf Added filename to redactionLog 2020-09-02 14:47:58 +02:00
Thierry Goeckel 34e058c4e4 Pull request #35: Fix redaction in single cell tables
Merge in RED/redaction-service from RED-279 to master

* commit '3d5455d7297e5805ad30a578d00d6d1737820c7d':
  Fix redaction in single cell tables
2020-09-02 14:39:08 +02:00
Thierry Göckel 3d5455d729 Fix redaction in single cell tables 2020-09-02 14:31:38 +02:00
Thierry Goeckel b07ebf78d2 Pull request #34: RED-267: Return dictionary and rules version in redactionLog
Merge in RED/redaction-service from RED-267 to master

* commit 'b5fa771c285554132edc32e1df59836fd7748b9a':
  RED-267: Return dictionary and rules version in redactionLog
2020-09-02 08:54:43 +02:00
deiflaender b5fa771c28 RED-267: Return dictionary and rules version in redactionLog 2020-09-02 08:47:05 +02:00
Thierry Goeckel 81ea2e91ef Pull request #33: Fix entity span in table rows and detection of headers in rotated tables
Merge in RED/redaction-service from bugfix/rowspan-and-header-in-rotated-table-fix to master

* commit '4954aafed78e06484531d6264bdf215176ee0ef2':
  Reduce log level.
  Remove unused import
  Adjust test to added rule and fix vertical header propagation for row > 2
  Fix entity span in table rows and detection of headers in rotated tables
2020-08-25 17:30:57 +02:00
Thierry Göckel 4954aafed7 Reduce log level. 2020-08-25 17:21:56 +02:00
Thierry Göckel 0baf48698d Remove unused import 2020-08-25 15:45:24 +02:00
Thierry Göckel 3c590dcf1d Adjust test to added rule and fix vertical header propagation for row >
2
2020-08-25 15:41:44 +02:00
Thierry Göckel 848c506c3f Fix entity span in table rows and detection of headers in rotated tables 2020-08-25 13:50:34 +02:00
Thierry Goeckel 6483e637c6 Pull request #32: Remove '-' in headernames
Merge in RED/redaction-service from tablesHeader3 to master

* commit '272c7cc2287b146664b8bb277c3731262ac78a19':
  Remove '-' in headernames
2020-08-24 16:15:44 +02:00
deiflaender 272c7cc228 Remove '-' in headernames 2020-08-24 15:39:50 +02:00
Thierry Goeckel 5586531c33 Pull request #31: Made rules for table more stabil and flexible
Merge in RED/redaction-service from tablesRules2 to master

* commit '559c42154276ea861c09cf29b71d9bb9c17e6e87':
  Made rules for table more stabil and flexible
2020-08-24 14:52:29 +02:00
deiflaender 559c421542 Made rules for table more stabil and flexible 2020-08-24 14:45:30 +02:00
Dominique Eiflaender d026f4e1db Pull request #30: Fix test and logic of same page table merge for one row tables
Merge in RED/redaction-service from bugfix/fix-test-table-merge to master

* commit '38ddb1d9c8dbf0e83d40ec916f1ebeb6d0f86f42':
  Fix test and logic of same page table merge for one row tables
2020-08-24 14:22:41 +02:00
Thierry Göckel 38ddb1d9c8 Fix test and logic of same page table merge for one row tables 2020-08-24 14:06:34 +02:00
Thierry Goeckel 92f0af9d21 Pull request #28: Highlight N in vertebrate study tables and fixed rule order
Merge in RED/redaction-service from tableRules to master

* commit 'afa26895c8e78f261ec3700c515f412d8cf605ce':
  Fixed test problem
  Fixed pmd
  Highlight N in vertebrate study tables and fixed rule order
2020-08-24 13:15:11 +02:00
Dominique Eiflaender b24922aa60 Pull request #29: Fix multi-page table merge
Merge in RED/redaction-service from bugfix/multi-page-table-merge to master

* commit 'baa703928f50e53a66f925fcf72082df188482a1':
  Fix multi-page table merge
2020-08-24 13:09:43 +02:00
Thierry Göckel baa703928f Fix multi-page table merge 2020-08-24 13:03:03 +02:00
64 changed files with 16985 additions and 430 deletions
+2 -2
View File
@@ -5,7 +5,7 @@
<parent>
<groupId>com.atlassian.bamboo</groupId>
<artifactId>bamboo-specs-parent</artifactId>
<version>7.0.4</version>
<version>7.1.2</version>
<relativePath/>
</parent>
@@ -34,4 +34,4 @@
<!-- run 'mvn test' to perform offline validation of the plan -->
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
</project>
</project>
@@ -1,4 +1,4 @@
FROM gin5/platform-base:5.2.0
FROM red/base-image:1.0.0
ARG PLATFORM_JAR
@@ -6,4 +6,4 @@ ENV PLATFORM_JAR ${PLATFORM_JAR}
ENV USES_ELASTICSEARCH false
COPY ["${PLATFORM_JAR}", "/"]
COPY ["${PLATFORM_JAR}", "/"]
+1 -1
View File
@@ -22,7 +22,7 @@
</modules>
<properties>
<pdfbox.version>2.0.16</pdfbox.version>
<pdfbox.version>2.0.21</pdfbox.version>
</properties>
@@ -0,0 +1,21 @@
package com.iqser.red.service.redaction.v1.model;
import java.time.OffsetDateTime;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class Comment {
private String id;
private OffsetDateTime date;
private String text;
private String user;
}
@@ -0,0 +1,19 @@
package com.iqser.red.service.redaction.v1.model;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class IdRemoval {
private String id;
private String user;
private Status status;
private boolean removeFromDictionary;
}
@@ -4,18 +4,25 @@ import java.util.ArrayList;
import java.util.List;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactionEntry {
private String id;
private String user;
private String type;
private String value;
private String reason;
private String legalBasis;
private List<Rectangle> positions = new ArrayList<>();
private Status status;
private boolean addToDictionary;
private String section;
private int sectionNumber;
@@ -0,0 +1,5 @@
package com.iqser.red.service.redaction.v1.model;
public enum ManualRedactionType {
ADD, REMOVE
}
@@ -1,17 +1,29 @@
package com.iqser.red.service.redaction.v1.model;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import lombok.AllArgsConstructor;
import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
@Data
@Builder
@AllArgsConstructor
@NoArgsConstructor
public class ManualRedactions {
private Set<String> idsToRemove = new HashSet<>();
@Builder.Default
private Set<IdRemoval> idsToRemove = new HashSet<>();
@Builder.Default
private Set<ManualRedactionEntry> entriesToAdd = new HashSet<>();
@Builder.Default
private Map<String, List<Comment>> comments = new HashMap<>();
}
@@ -11,6 +11,19 @@ import lombok.NoArgsConstructor;
@NoArgsConstructor
public class RedactionLog {
public RedactionLog(List<RedactionLogEntry> redactionLogEntry, long dictionaryVersion, long rulesVersion) {
this.redactionLogEntry = redactionLogEntry;
this.dictionaryVersion = dictionaryVersion;
this.rulesVersion = rulesVersion;
}
private List<RedactionLogEntry> redactionLogEntry;
private long dictionaryVersion = -1;
private long rulesVersion = -1;
private String filename;
}
@@ -18,6 +18,8 @@ public class RedactionLogEntry {
private String type;
private String value;
private String reason;
private int matchedRule;
private String legalBasis;
private boolean redacted;
private boolean isHint;
private String section;
@@ -27,5 +29,7 @@ public class RedactionLogEntry {
private List<Rectangle> positions = new ArrayList<>();
private int sectionNumber;
private boolean manual;
private Status status;
private ManualRedactionType manualRedactionType;
}
@@ -0,0 +1,5 @@
package com.iqser.red.service.redaction.v1.model;
public enum Status {
REQUESTED, APPROVED, DECLINED
}
@@ -11,25 +11,6 @@
<artifactId>redaction-service-server-v1</artifactId>
<properties>
<pdfbox.version>2.0.20</pdfbox.version>
</properties>
<dependencyManagement>
<dependencies>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>${pdfbox.version}</version>
</dependency>
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox-tools</artifactId>
<version>${pdfbox.version}</version>
</dependency>
</dependencies>
</dependencyManagement>
<dependencies>
<dependency>
<groupId>com.iqser.red.service</groupId>
@@ -39,7 +20,7 @@
<dependency>
<groupId>com.iqser.red.service</groupId>
<artifactId>configuration-service-api-v1</artifactId>
<version>1.0.12</version>
<version>1.2.0</version>
</dependency>
<dependency>
<groupId>org.drools</groupId>
@@ -12,6 +12,7 @@ import org.kie.api.builder.KieModule;
import org.kie.api.runtime.KieContainer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.actuate.autoconfigure.metrics.web.servlet.WebMvcMetricsAutoConfiguration;
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
@@ -29,7 +30,7 @@ import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettin
@Import({DefaultWebMvcConfiguration.class})
@EnableFeignClients(basePackageClasses = RulesClient.class)
@EnableConfigurationProperties(RedactionServiceSettings.class)
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class})
@SpringBootApplication(exclude = {SecurityAutoConfiguration.class, ManagementWebSecurityAutoConfiguration.class, WebMvcMetricsAutoConfiguration.class})
public class Application {
@Autowired
@@ -131,7 +131,7 @@ public class TextBlock extends AbstractTextContainer {
TextPositionSequence previous = null;
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -84,7 +84,7 @@ public class ClassificationService {
}
else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular() && textBlock.getMostPopularWordStyle().equals("italic") && !document.getFontStyleCounter().getMostPopular().equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
textBlock.setClassification("TextBlock Italic");
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getSequences().size() > 3){
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock)){
textBlock.setClassification("TextBlock Unknown");
}
}
@@ -3,18 +3,21 @@ package com.iqser.red.service.redaction.v1.server.controller;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.List;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RestController;
import com.iqser.red.service.redaction.v1.model.RedactionLog;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
import com.iqser.red.service.redaction.v1.server.redaction.service.EntityRedactionService;
import com.iqser.red.service.redaction.v1.server.segmentation.PdfSegmentationService;
@@ -36,6 +39,7 @@ public class RedactionController implements RedactionResource {
private final EntityRedactionService entityRedactionService;
private final PdfFlattenService pdfFlattenService;
private final DroolsExecutionService droolsExecutionService;
private final DictionaryService dictionaryService;
@Override
public RedactionResult redact(@RequestBody RedactionRequest redactionRequest) {
@@ -49,10 +53,10 @@ public class RedactionController implements RedactionResource {
if (redactionRequest.isFlatRedaction()) {
PDDocument flatDocument = pdfFlattenService.flattenPDF(pdDocument);
return convert(flatDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
return convert(flatDocument, classifiedDoc.getPages().size(), classifiedDoc.getRedactionLogEntities());
}
return convert(pdDocument, classifiedDoc.getPages().size(), new RedactionLog(classifiedDoc.getRedactionLogEntities()));
return convert(pdDocument, classifiedDoc.getPages().size(), classifiedDoc.getRedactionLogEntities());
} catch (IOException e) {
throw new RedactionException(e);
@@ -130,14 +134,14 @@ public class RedactionController implements RedactionResource {
return convert(document, numberOfPages, null);
}
private RedactionResult convert(PDDocument document, int numberOfPages, RedactionLog redactionLog) throws IOException {
private RedactionResult convert(PDDocument document, int numberOfPages, List<RedactionLogEntry> redactionLogEntities) throws IOException {
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
document.save(byteArrayOutputStream);
return RedactionResult.builder()
.document(byteArrayOutputStream.toByteArray())
.numberOfPages(numberOfPages)
.redactionLog(redactionLog)
.redactionLog(new RedactionLog(redactionLogEntities, dictionaryService.getDictionaryVersion(), droolsExecutionService.getRulesVersion()))
.build();
}
@@ -44,10 +44,10 @@ import lombok.extern.slf4j.Slf4j;
public class PDFLinesTextStripper extends PDFTextStripper {
@Getter
private float minCharWidth = Float.MAX_VALUE;
private int maxCharWidths;
@Getter
private float minCharHeight = Float.MAX_VALUE;
private int maxCharHeight;
@Getter
private final List<TextPositionSequence> textPositionSequences = new ArrayList<>();
@@ -201,8 +201,16 @@ public class PDFLinesTextStripper extends PDFTextStripper {
int startIndex = 0;
for (int i = 0; i <= textPositions.size() - 1; i++) {
minCharWidth = Math.min(minCharWidth, textPositions.get(i).getWidthDirAdj());
minCharHeight = Math.min(minCharHeight, textPositions.get(i).getHeightDir());
int charHeight = (int) textPositions.get(i).getHeightDir();
if(charHeight > maxCharHeight){
maxCharHeight = charHeight;
}
int charWidth = (int) textPositions.get(i).getWidthDirAdj();
if(charWidth > maxCharWidths){
maxCharWidths = charWidth;
}
if (i == 0 && textPositions.get(i).getUnicode().equals(" ")) {
startIndex++;
@@ -241,8 +249,8 @@ public class PDFLinesTextStripper extends PDFTextStripper {
@Override
public String getText(PDDocument doc) throws IOException {
minCharWidth = Float.MAX_VALUE;
minCharHeight = Float.MAX_VALUE;
maxCharWidths = 0;
maxCharWidths = 0;
textPositionSequences.clear();
rulings.clear();
graphicsPath.clear();
@@ -17,6 +17,6 @@ public class ParsedElements {
private boolean landscape;
private boolean rotated;
private float minCharWidth;
private float minCharHeight;
private float maxCharWidth;
private float maxCharHeight;
}
@@ -9,9 +9,7 @@ import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import lombok.Data;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.Setter;
@Data
@RequiredArgsConstructor
@@ -19,10 +17,6 @@ public class TextPositionSequence implements CharSequence {
private List<TextPosition> textPositions = new ArrayList<>();
@Getter
@Setter
private float[] annotationColor;
private final int page;
@@ -107,6 +101,9 @@ public class TextPositionSequence implements CharSequence {
}
}
public float getRotationAdjustedY() {
return textPositions.get(0).getY();
}
public float getY1() {
@@ -194,22 +191,18 @@ public class TextPositionSequence implements CharSequence {
public Rectangle getRectangle() {
float height = textPositions.get(0).getHeightDir() + 2;
float height = getTextHeight();
float posXInit;
float posXInit = getX1();
float posXEnd;
float posYInit;
float posYEnd;
if (textPositions.get(0).getRotation() == 90) {
posXEnd = textPositions.get(0).getYDirAdj() + 2;
posXInit = textPositions.get(0).getYDirAdj() - height;
posYInit = textPositions.get(0).getXDirAdj();
posYInit = getY1();
posYEnd = textPositions.get(textPositions.size() - 1).getXDirAdj() - height + 4;
} else {
posXInit = textPositions.get(0).getXDirAdj();
posXEnd = textPositions.get(textPositions.size() - 1)
.getXDirAdj() + textPositions.get(textPositions.size() - 1).getWidth() + 1;
posYInit = textPositions.get(0).getPageHeight() - textPositions.get(0).getYDirAdj() - 2;
@@ -220,4 +213,5 @@ public class TextPositionSequence implements CharSequence {
return new Rectangle(new Point(posXInit, posYInit), posXEnd - posXInit, posYEnd - posYInit + height, page);
}
}
@@ -0,0 +1,40 @@
package com.iqser.red.service.redaction.v1.server.redaction.model;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import lombok.Value;
@Value
public class CellValue {
TextBlock textBlock;
int rowSpanStart;
@Override
public String toString() {
StringBuilder sb = new StringBuilder();
TextPositionSequence previous = null;
for (TextPositionSequence word : textBlock.getSequences()) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
}
}
sb.append(word.toString());
previous = word;
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" {2}", " ");
}
}
@@ -17,6 +17,7 @@ public class Entity {
private final String type;
private boolean redaction;
private String redactionReason;
private String legalBasis;
private List<EntityPositionSequence> positionSequences = new ArrayList<>();
private List<TextPositionSequence> targetSequences;
private Integer start;
@@ -30,7 +31,7 @@ public class Entity {
private int sectionNumber;
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber) {
public Entity(String word, String type, boolean redaction, String redactionReason, List<EntityPositionSequence> positionSequences, String headline, int matchedRule, int sectionNumber, String legalBasis) {
this.word = word;
this.type = type;
@@ -40,6 +41,7 @@ public class Entity {
this.headline = headline;
this.matchedRule = matchedRule;
this.sectionNumber = sectionNumber;
this.legalBasis = legalBasis;
}
@@ -136,9 +136,11 @@ public class SearchableText {
private List<EntityPositionSequence> buildEntityPositionSequence(List<TextPositionSequence> crossSequenceParts) {
String id = IdBuilder.buildId(crossSequenceParts);
String plainId = IdBuilder.buildId(crossSequenceParts);
String id = plainId;
List<EntityPositionSequence> result = new ArrayList<>();
int currentPage = -1;
int idDiffentPageSuffix = 1;
EntityPositionSequence entityPositionSequence = new EntityPositionSequence(id);
for (TextPositionSequence textPositionSequence : crossSequenceParts) {
if (currentPage == -1) {
@@ -148,9 +150,13 @@ public class SearchableText {
} else if (currentPage == textPositionSequence.getPage()) {
entityPositionSequence.getSequences().add(textPositionSequence);
} else {
id = plainId + "-" + idDiffentPageSuffix;
idDiffentPageSuffix++;
result.add(entityPositionSequence);
entityPositionSequence = new EntityPositionSequence(id);
entityPositionSequence.setPageNumber(textPositionSequence.getPage());
entityPositionSequence.getSequences().add(textPositionSequence);
currentPage = textPositionSequence.getPage();
}
}
result.add(entityPositionSequence);
@@ -173,7 +179,7 @@ public class SearchableText {
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -197,7 +203,7 @@ public class SearchableText {
for (TextPositionSequence word : sequences) {
if (previous != null) {
if (Math.abs(previous.getY1() - word.getY1()) > word.getTextHeight()) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
@@ -9,8 +9,6 @@ import java.util.regex.Pattern;
import org.apache.commons.lang3.StringUtils;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import lombok.Builder;
import lombok.Data;
import lombok.extern.slf4j.Slf4j;
@@ -32,28 +30,21 @@ public class Section {
private int sectionNumber;
private Map<String, TextBlock> tabularData;
private Map<String, CellValue> tabularData;
public boolean isVertebrateStudy() {
return tabularData != null
&& (tabularData.containsKey("Vertebrate study Y/N")
&& tabularData.get("Vertebrate study Y/N").getText().equals("Y")
|| tabularData.containsKey("Verte brate study Y/N")
&& tabularData.get("Verte brate study Y/N").getText().equals("Y"));
public boolean rowEquals(String headerName, String value) {
String cleanHeaderName = headerName.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
return tabularData != null && tabularData.containsKey(cleanHeaderName) && tabularData.get(cleanHeaderName)
.getTextBlock()
.getText()
.equals(value);
}
public boolean isNotVertebrateStudy() {
return tabularData != null
&& (tabularData.containsKey("Vertebrate study Y/N")
&& tabularData.get("Vertebrate study Y/N").getText().equals("N")
|| tabularData.containsKey("Verte brate study Y/N")
&& tabularData.get("Verte brate study Y/N").getText().equals("N"));
}
public boolean contains(String type) {
public boolean matchesType(String type) {
return entities.stream().anyMatch(entity -> entity.getType().equals(type));
}
@@ -65,13 +56,14 @@ public class Section {
}
public void redact(String type, int ruleNumber, String reason) {
public void redact(String type, int ruleNumber, String reason, String legalBasis) {
entities.forEach(entity -> {
if (entity.getType().equals(type)) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
@@ -89,7 +81,20 @@ public class Section {
}
public void redactLineAfter(String start, String asType, int ruleNumber, String reason) {
public void redactIfPrecededBy(String prefix, String type, int ruleNumber, String reason, String legalBasis) {
entities.forEach(entity -> {
if (entity.getType().equals(type) && searchText.indexOf(prefix + entity.getWord()) != 1) {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
public void redactLineAfter(String start, String asType, int ruleNumber, String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(text, start, "\n");
@@ -108,13 +113,14 @@ public class Section {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason) {
public void redactBetween(String start, String stop, String asType, int ruleNumber, String reason, String legalBasis) {
String[] values = StringUtils.substringsBetween(searchText, start, stop);
@@ -133,6 +139,7 @@ public class Section {
entity.setRedaction(true);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(reason);
entity.setLegalBasis(legalBasis);
}
});
}
@@ -148,9 +155,15 @@ public class Section {
startIndex = searchText.indexOf(value, stopIndex);
stopIndex = startIndex + value.length();
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(searchText
.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex), asType, startIndex, stopIndex, headline, sectionNumber));
if (startIndex > -1 && (startIndex == 0 || Character.isWhitespace(searchText.charAt(startIndex - 1)) || isSeparator(
searchText.charAt(startIndex - 1))) && (stopIndex == searchText.length() || isSeparator(searchText.charAt(
stopIndex)))) {
found.add(new Entity(searchText.substring(startIndex, stopIndex),
asType,
startIndex,
stopIndex,
headline,
sectionNumber));
}
} while (startIndex > -1);
@@ -160,7 +173,8 @@ public class Section {
private boolean isSeparator(char c) {
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}", String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
return Character.isWhitespace(c) || Pattern.matches("\\p{Punct}",
String.valueOf(c)) || c == '\"' || c == '‘' || c == '’';
}
@@ -182,18 +196,50 @@ public class Section {
public void highlightCell(String cellHeader, int ruleNumber, String type) {
TextBlock value = tabularData.get(cellHeader);
annotateCell(cellHeader, ruleNumber, type, false, null, null);
}
public void redactCell(String cellHeader, int ruleNumber, String type, String reason, String legalBasis) {
annotateCell(cellHeader, ruleNumber, type, true, reason, legalBasis);
}
public void redactNotCell(String cellHeader, int ruleNumber, String type, String reason) {
annotateCell(cellHeader, ruleNumber, type, false, reason, null);
}
private void annotateCell(String cellHeader, int ruleNumber, String type, boolean redact, String reason, String legalBasis) {
String cleanHeaderName = cellHeader.replaceAll("\n", "").replaceAll(" ", "").replaceAll("-", "");
CellValue value = tabularData.get(cleanHeaderName);
if (value == null) {
log.warn("Could not find any data for {}.", cellHeader);
} else {
Entity entity = new Entity(value.getText(), type, 0, value.getText().length(), headline, sectionNumber);
entity.setRedaction(false);
String word = value.toString();
Entity entity = new Entity(word,
type,
value.getRowSpanStart(),
value.getRowSpanStart() + word.length(),
headline,
sectionNumber);
entity.setRedaction(redact);
entity.setMatchedRule(ruleNumber);
entity.setRedactionReason(cellHeader);
entity.setTargetSequences(value.getSequences()); // Make sure no other cells with same content are highlighted
entities.add(entity);
}
entity.setRedactionReason(reason);
entity.setTargetSequences(value.getTextBlock()
.getSequences()); // Make sure no other cells with same content are highlighted
entity.setLegalBasis(legalBasis);
// HashSet keeps the older value, but we want the new only.
entities.remove(entity);
entities.add(entity);
entities = removeEntitiesContainedInLarger(entities);
}
}
}
}
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server.redaction.service;
import java.awt.Color;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
@@ -11,6 +12,7 @@ import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
@@ -20,13 +22,14 @@ import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
@Slf4j
public class DictionaryService {
private final DictionaryClient dictionaryClient;
@Getter
private long dictionaryVersion = -1;
@Getter
@@ -44,6 +47,15 @@ public class DictionaryService {
@Getter
private float[] defaultColor;
@Getter
private float[] requestAddColor;
@Getter
private float[] requestRemoveColor;
@Getter
private float[] notRedactedColor;
public void updateDictionary() {
@@ -62,7 +74,7 @@ public class DictionaryService {
if (typeResponse != null && CollectionUtils.isNotEmpty(typeResponse.getTypes())) {
entryColors = typeResponse.getTypes()
.stream()
.collect(Collectors.toMap(TypeResult::getType, TypeResult::getColor));
.collect(Collectors.toMap(TypeResult::getType, t -> convertColor(t.getHexColor())));
hintTypes = typeResponse.getTypes()
.stream()
.filter(TypeResult::isHint)
@@ -73,8 +85,16 @@ public class DictionaryService {
.filter(TypeResult::isCaseInsensitive)
.map(TypeResult::getType)
.collect(Collectors.toList());
dictionary = entryColors.keySet().stream().collect(Collectors.toMap(type -> type, this::convertEntries));
defaultColor = dictionaryClient.getDefaultColor().getColor();
dictionary = entryColors.keySet()
.stream()
.collect(Collectors.toMap(type -> type, this::convertEntries));
Colors colors = dictionaryClient.getColors();
defaultColor = convertColor(colors.getDefaultColor());
requestAddColor = convertColor(colors.getRequestAdd());
requestRemoveColor = convertColor(colors.getRequestRemove());
notRedactedColor = convertColor(colors.getNotRedacted());
}
} catch (FeignException e) {
log.warn("Got some unknown feignException", e);
@@ -84,6 +104,7 @@ public class DictionaryService {
private Set<String> convertEntries(String s) {
if (caseInsensitiveTypes.contains(s)) {
return dictionaryClient.getDictionaryForType(s)
.getEntries()
@@ -94,4 +115,11 @@ public class DictionaryService {
return new HashSet<>(dictionaryClient.getDictionaryForType(s).getEntries());
}
private float[] convertColor(String hex) {
Color color = Color.decode(hex);
return new float[]{color.getRed() / 255f, color.getGreen() / 255f, color.getBlue() / 255f};
}
}
@@ -18,6 +18,7 @@ import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
import com.iqser.red.service.redaction.v1.server.redaction.model.Section;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
@Service
@@ -29,6 +30,7 @@ public class DroolsExecutionService {
@Autowired
private KieContainer kieContainer;
@Getter
private long rulesVersion = -1;
public Section executeRules(Section section) {
@@ -18,6 +18,7 @@ import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
@@ -26,9 +27,7 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class EntityRedactionService {
@@ -51,24 +50,30 @@ public class EntityRedactionService {
List<Table> tables = paragraph.getTables();
for (Table table : tables) {
boolean singleCellTable = table.getRowCount() == 1 && table.getColCount() == 1;
for (List<Cell> row : table.getRows()) {
SearchableText searchableRow = new SearchableText();
Map<String, TextBlock> tabularData = new HashMap<>();
Map<String, CellValue> tabularData = new HashMap<>();
int start = 0;
for (Cell cell : row) {
if (cell.isHeaderCell() || CollectionUtils.isEmpty(cell.getTextBlocks())) {
if (!singleCellTable && cell.isHeaderCell() || CollectionUtils.isEmpty(cell.getTextBlocks())) {
continue;
}
addSectionToManualRedactions(cell.getTextBlocks(), manualRedactions, table.getHeadline(), sectionNumber);
int cellStart = start;
cell.getHeaderCells().forEach(headerCell -> {
String headerName = headerCell.getTextBlocks().get(0).getText()
.replaceAll("\n", " ")
.replaceAll(" ", " ");
tabularData.put(headerName, cell.getTextBlocks().get(0));
StringBuilder headerBuilder = new StringBuilder();
headerCell.getTextBlocks().forEach(textBlock -> headerBuilder.append(textBlock.getText()));
String headerName = headerBuilder.toString()
.replaceAll("\n", "")
.replaceAll(" ", "")
.replaceAll("-", "");
tabularData.put(headerName, new CellValue(cell.getTextBlocks().get(0), cellStart));
});
start = start + cell.toString().length() + 1; // include automatically appended white space
for (TextBlock textBlock : cell.getTextBlocks()) {
searchableRow.addAll(textBlock.getSequences());
}
}
Set<Entity> rowEntities = findEntities(searchableRow, table.getHeadline(), sectionNumber);
@@ -112,7 +117,7 @@ public class EntityRedactionService {
classifiedDoc.getEntities()
.computeIfAbsent(entry.getKey(), (x) -> new ArrayList<>())
.add(new Entity(entity.getWord(), entity.getType(), entity.isRedaction(), entity.getRedactionReason(), entry
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber()));
.getValue(), entity.getHeadline(), entity.getMatchedRule(), entity.getSectionNumber(), entity.getLegalBasis()));
}
}
@@ -138,17 +143,17 @@ public class EntityRedactionService {
private Set<Entity> findEntities(SearchableText searchableText, String headline, int sectionNumber) {
Set<Entity> found = new HashSet<>();
if (StringUtils.isEmpty(searchableText.toString()) && StringUtils.isEmpty(headline)) {
String searchableString = searchableText.toString();
if (StringUtils.isEmpty(searchableString)) {
return found;
}
String inputString = searchableText.toString();
String lowercaseInputString = inputString.toLowerCase();
String lowercaseInputString = searchableString.toLowerCase();
for (Map.Entry<String, Set<String>> entry : dictionaryService.getDictionary().entrySet()) {
if (dictionaryService.getCaseInsensitiveTypes().contains(entry.getKey())) {
found.addAll(find(lowercaseInputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
} else {
found.addAll(find(inputString, entry.getValue(), entry.getKey(), headline, sectionNumber));
found.addAll(find(searchableString, entry.getValue(), entry.getKey(), headline, sectionNumber));
}
}
@@ -5,7 +5,6 @@ import java.util.List;
import com.google.common.hash.HashFunction;
import com.google.common.hash.Hashing;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import lombok.experimental.UtilityClass;
@@ -19,14 +18,9 @@ public class IdBuilder {
StringBuilder sb = new StringBuilder();
crossSequenceParts.forEach(sequencePart -> sequencePart.getTextPositions().forEach(textPosition -> {
sb.append(textPosition.getTextMatrix());
sb.append(textPosition.getTextMatrix()).append(sequencePart.getPage());
}));
return hashFunction.hashString(sb.toString(), StandardCharsets.UTF_8).toString();
}
public String buildId(ManualRedactionEntry manualRedactionEntry) {
return hashFunction.hashString(manualRedactionEntry.toString(), StandardCharsets.UTF_8).toString();
}
}
@@ -21,7 +21,6 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractT
import com.iqser.red.service.redaction.v1.server.tableextraction.model.CleanRulings;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.RulingCleaningService;
import com.iqser.red.service.redaction.v1.server.tableextraction.service.TableExtractionService;
import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@@ -37,6 +36,7 @@ public class PdfSegmentationService {
private final ClassificationService classificationService;
private final SectionsBuilderService sectionsBuilderService;
public Document parseDocument(PDDocument pdDocument) throws IOException {
Document document = new Document();
@@ -56,19 +56,21 @@ public class PdfSegmentationService {
int rotation = pdPage.getRotation();
boolean isRotated = rotation != 0 && rotation != 360;
ParsedElements parsedElements = ParsedElements
.builder()
ParsedElements parsedElements = ParsedElements.builder()
.rulings(stripper.getRulings())
.sequences(stripper.getTextPositionSequences())
.minCharWidth(Utils.round(stripper.getMinCharWidth(), 2))
.minCharHeight(Utils.round(stripper.getMinCharHeight(), 2))
.maxCharWidth(stripper.getMaxCharWidths())
.maxCharHeight(stripper.getMaxCharWidths())
.landscape(isLandscape)
.rotated(isRotated)
.build();
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), parsedElements.getMinCharWidth(), parsedElements.getMinCharHeight());
CleanRulings cleanRulings = rulingCleaningService.getCleanRulings(parsedElements.getRulings(), parsedElements
.getMaxCharWidth(), parsedElements.getMaxCharHeight());
Page page = blockificationService.blockify(parsedElements.getSequences(), cleanRulings.getHorizontal(), cleanRulings.getVertical());
Page page = blockificationService.blockify(parsedElements.getSequences(), cleanRulings.getHorizontal(), cleanRulings
.getVertical());
page.setRotation(rotation);
tableExtractionService.extractTables(cleanRulings, page);
@@ -91,7 +93,10 @@ public class PdfSegmentationService {
}
private void increaseDocumentStatistics(Page page, Document document) {
if (!page.isLandscape()) {
document.getFontSizeCounter().addAll(page.getFontSizeCounter().getCountPerValue());
}
@@ -100,6 +105,7 @@ public class PdfSegmentationService {
document.getFontStyleCounter().addAll(page.getFontStyleCounter().getCountPerValue());
}
private void buildPageStatistics(Page page) {
// Collect all statistics for the page, except from blocks inside tables, as tables will always be added to BodyTextFrame.
@@ -4,6 +4,7 @@ import java.util.ArrayList;
import java.util.Collections;
import java.util.Iterator;
import java.util.List;
import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import org.springframework.stereotype.Service;
@@ -38,23 +39,30 @@ public class SectionsBuilderService {
current.setPage(page.getPageNumber());
if (prev != null && current.getClassification().startsWith("H ") || !document.isHeadlines()) {
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
if (prev != null && current.getClassification().startsWith("H ") && !prev.getClassification().startsWith("H ") || !document.isHeadlines()) {
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
lastHeadline = current.getText();
if (CollectionUtils.isNotEmpty(chunkBlock.getTables())) {
previousTable = chunkBlock.getTables().get(0);
if(document.isHeadlines()) {
lastHeadline = current.getText();
}
chunkBlockList.add(chunkBlock);
chunkWords = new ArrayList<>();
if (CollectionUtils.isNotEmpty(chunkBlock.getTables())) {
previousTable = chunkBlock.getTables().get(chunkBlock.getTables().size() - 1);
}
}
if (current instanceof Table) {
Table table = (Table) current;
// Distribute header information for subsequent tables
mergeTableMetadata(table, previousTable);
previousTable = table;
}
chunkWords.add(current);
prev = current;
}
}
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline, previousTable);
Paragraph chunkBlock = buildTextBlock(chunkWords, lastHeadline);
chunkBlock.setHeadline(lastHeadline);
chunkBlockList.add(chunkBlock);
@@ -62,7 +70,38 @@ public class SectionsBuilderService {
}
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline, Table previousTable) {
private void mergeTableMetadata(Table currentTable, Table previousTable) {
// Distribute header information for subsequent tables
if (previousTable != null && hasInvalidHeaderInformation(currentTable) && hasValidHeaderInformation(previousTable)) {
List<Cell> previousTableNonHeaderRow = getRowWithNonHeaderCells(previousTable);
List<Cell> tableNonHeaderRow = getRowWithNonHeaderCells(currentTable);
// Allow merging of tables if header row is separated from first logical non-header row
if (previousTableNonHeaderRow.isEmpty() && previousTable.getRowCount() == 1 && previousTable.getRows()
.get(0)
.size() == tableNonHeaderRow.size()) {
previousTableNonHeaderRow = previousTable.getRows().get(0).stream().map(cell -> {
Cell fakeCell = new Cell(cell.getPoints()[0], cell.getPoints()[2]);
fakeCell.setHeaderCells(Collections.singletonList(cell));
return fakeCell;
}).collect(Collectors.toList());
}
if (previousTableNonHeaderRow.size() == tableNonHeaderRow.size()) {
for (int i = currentTable.getRowCount() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = currentTable.getRows().get(i);
if (row.size() == tableNonHeaderRow.size() && row.stream()
.allMatch(cell -> cell.getHeaderCells().isEmpty())) {
for (int j = 0; j < row.size(); j++) {
row.get(j).setHeaderCells(previousTableNonHeaderRow.get(j).getHeaderCells());
}
}
}
}
}
}
private Paragraph buildTextBlock(List<AbstractTextContainer> wordBlockList, String lastHeadline) {
Paragraph paragraph = new Paragraph();
TextBlock textBlock = null;
@@ -85,27 +124,6 @@ public class SectionsBuilderService {
} else {
table.setHeadline("Table in: " + lastHeadline);
}
// Distribute header information for subsequent tables
if (previousTable != null && hasInvalidHeaderInformation(table) && hasValidHeaderInformation(previousTable)) {
List<Cell> previousTableNonHeaderRow = getRowWithNonHeaderCells(previousTable);
List<Cell> tableNonHeaderRow = getRowWithNonHeaderCells(table);
// Allow merging of tables if header row is separated from first logical non-header row
if (previousTableNonHeaderRow.isEmpty() && previousTable.getRowCount() == 1
&& previousTable.getRows().get(0).size() == tableNonHeaderRow.size()) {
previousTableNonHeaderRow = previousTable.getRows().get(0);
}
if (previousTableNonHeaderRow.size() == tableNonHeaderRow.size()) {
for (int i = table.getRows().size() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = table.getRows().get(i);
if (row.size() == tableNonHeaderRow.size()
&& row.stream().allMatch(cell -> cell.getHeaderCells().isEmpty())) {
for (int j = 0; j < row.size(); j++) {
row.get(j).setHeaderCells(previousTableNonHeaderRow.get(j).getHeaderCells());
}
}
}
}
}
if (textBlock != null && !alreadyAdded) {
paragraph.getPageBlocks().add(textBlock);
@@ -168,7 +186,7 @@ public class SectionsBuilderService {
private List<Cell> getRowWithNonHeaderCells(Table table) {
for (int i = table.getRows().size() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
for (int i = table.getRowCount() - 1; i >= 0; i--) { // Non header rows are most likely at bottom of table
List<Cell> row = table.getRows().get(i);
boolean allNonHeader = true;
for (Cell cell : row) {
@@ -5,6 +5,8 @@ import java.util.ArrayList;
import java.util.List;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
import lombok.Data;
import lombok.EqualsAndHashCode;
@@ -20,10 +22,10 @@ public class Cell extends Rectangle {
private boolean isHeaderCell;
public Cell(Point2D topLeft, Point2D bottomRight) {
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()),
(float) (bottomRight
super((float) topLeft.getY(), (float) topLeft.getX(), (float) (bottomRight.getX() - topLeft.getX()), (float) (bottomRight
.getY() - topLeft.getY()));
}
@@ -33,4 +35,30 @@ public class Cell extends Rectangle {
textBlocks.add(textBlock);
}
@Override
public String toString() {
StringBuilder sb = new StringBuilder();
for (TextBlock textBlock : textBlocks) {
TextPositionSequence previous = null;
for (TextPositionSequence word : textBlock.getSequences()) {
if (previous != null) {
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
sb.append('\n');
} else {
sb.append(' ');
}
}
sb.append(word.toString());
previous = word;
}
}
return TextNormalizationUtilities.removeHyphenLineBreaks(sb.toString())
.replaceAll("\n", " ")
.replaceAll(" {2}", " ");
}
}
@@ -29,11 +29,13 @@ public class Table extends AbstractTextContainer {
@Setter
private String headline;
@Getter
private int rowCount;
private int unrotatedRowCount;
@Getter
private int colCount;
private int unrotatedColCount;
private int rowCount = -1;
private int colCount = -1;
private final int rotation;
@@ -65,6 +67,25 @@ public class Table extends AbstractTextContainer {
}
public int getRowCount() {
if (rowCount == -1) {
rowCount = getRows().size();
}
return rowCount;
}
public int getColCount() {
if (colCount == -1) {
colCount = getRows().stream().mapToInt(List::size).max().orElse(0);
}
return colCount;
}
/**
* Detect header cells (either first row or first column):
* Column is marked as header if cell text is bold and row cell text is not bold.
@@ -72,100 +93,54 @@ public class Table extends AbstractTextContainer {
*/
private void computeHeaders() {
if (rows == null) {
rows = computeRows();
}
// A bold cell is a header cell as long as every cell to the left/top is bold, too
cells.forEach((position, cell) -> {
List<Cell> cellsToTheLeft = getCellsToTheLeft(position);
Cell lastHeaderCell = null;
for (Cell leftCell : cellsToTheLeft) {
if (CollectionUtils.isNotEmpty(leftCell.getTextBlocks()) && leftCell.getTextBlocks()
// we move from left to right and top to bottom
for (int rowIndex = 0; rowIndex < rows.size(); rowIndex++) {
List<Cell> rowCells = rows.get(rowIndex);
for (int colIndex = 0; colIndex < rowCells.size(); colIndex++) {
Cell cell = rowCells.get(colIndex);
List<Cell> cellsToTheLeft = rowCells.subList(0, colIndex);
Cell lastHeaderCell = null;
for (Cell leftCell : cellsToTheLeft) {
if (leftCell.isHeaderCell()) {
lastHeaderCell = leftCell;
} else {
break;
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
List<Cell> cellsToTheTop = new ArrayList<>();
for (int i = 0; i < rowIndex; i++) {
try {
cellsToTheTop.add(rows.get(i).get(colIndex));
} catch (IndexOutOfBoundsException e) {
log.debug("No cell {} in row {}, ignoring.", colIndex, rowIndex);
}
}
for (Cell topCell : cellsToTheTop) {
if (topCell.isHeaderCell()) {
lastHeaderCell = topCell;
} else {
break;
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
if (CollectionUtils.isNotEmpty(cell.getTextBlocks()) && cell.getTextBlocks()
.get(0)
.getMostPopularWordStyle()
.equals("bold")) {
lastHeaderCell = leftCell;
} else {
break;
cell.setHeaderCell(true);
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
lastHeaderCell = null;
List<Cell> cellsToTheTop = getCellToTheTop(position);
for (Cell topCell : cellsToTheTop) {
if (CollectionUtils.isNotEmpty(topCell.getTextBlocks()) && topCell.getTextBlocks()
.get(0)
.getMostPopularWordStyle()
.equals("bold")) {
lastHeaderCell = topCell;
} else {
break;
}
}
if (lastHeaderCell != null) {
cell.getHeaderCells().add(lastHeaderCell);
}
if (CollectionUtils.isNotEmpty(cell.getTextBlocks()) && cell.getTextBlocks()
.get(0)
.getMostPopularWordStyle()
.equals("bold")) {
cell.setHeaderCell(true);
}
});
}
private List<Cell> getCellsToTheLeft(CellPosition cellPosition) {
List<Cell> result = new ArrayList<>();
if (cellPosition.getCol() == 0) {
return result;
}
int row = cellPosition.getRow();
for (int i = cellPosition.getCol() - 1; i >= 0; i--) {
if (cells.get(new CellPosition(row, i)) != null) {
result.add(cells.get(new CellPosition(row, i)));
} else {
Cell spanningCell = null;
while (spanningCell == null && row >= 0) {
row--;
spanningCell = cells.get(new CellPosition(row, i));
}
if (spanningCell != null) {
result.add(spanningCell);
}
row = cellPosition.getRow();
}
}
Collections.reverse(result);
return result;
}
private List<Cell> getCellToTheTop(CellPosition cellPosition) {
List<Cell> result = new ArrayList<>();
if (cellPosition.getRow() == 0) {
return result;
}
int col = cellPosition.getCol();
for (int i = cellPosition.getRow() - 1; i >= 0; i--) {
if (cells.get(new CellPosition(i, col)) != null) {
result.add(cells.get(new CellPosition(i, col)));
} else {
Cell spanningCell = null;
while (spanningCell == null && col >= 0) {
col--;
spanningCell = cells.get(new CellPosition(i, col));
}
if (spanningCell != null) {
result.add(spanningCell);
}
col = cellPosition.getCol();
}
}
Collections.reverse(result);
return result;
}
@@ -173,9 +148,9 @@ public class Table extends AbstractTextContainer {
List<List<Cell>> rows = new ArrayList<>();
if (rotation == 90) {
for (int i = 0; i < colCount; i++) { // rows
for (int i = 0; i < unrotatedColCount; i++) { // rows
List<Cell> lastRow = new ArrayList<>();
for (int j = rowCount - 1; j >= 0; j--) { // cols
for (int j = unrotatedRowCount - 1; j >= 0; j--) { // cols
Cell cell = cells.get(new CellPosition(j, i));
if (cell != null) {
lastRow.add(cell);
@@ -184,9 +159,9 @@ public class Table extends AbstractTextContainer {
rows.add(lastRow);
}
} else if (rotation == 270) {
for (int i = colCount - 1; i >= 0; i--) { // rows
for (int i = unrotatedColCount - 1; i >= 0; i--) { // rows
List<Cell> lastRow = new ArrayList<>();
for (int j = 0; j < rowCount; j++) { // cols
for (int j = 0; j < unrotatedRowCount; j++) { // cols
Cell cell = cells.get(new CellPosition(i, j));
if (cell != null) {
lastRow.add(cell);
@@ -195,9 +170,9 @@ public class Table extends AbstractTextContainer {
rows.add(lastRow);
}
} else {
for (int i = 0; i < rowCount; i++) {
for (int i = 0; i < unrotatedRowCount; i++) {
List<Cell> lastRow = new ArrayList<>();
for (int j = 0; j < colCount; j++) {
for (int j = 0; j < unrotatedColCount; j++) {
Cell cell = cells.get(new CellPosition(i, j)); // JAVA_8 use getOrDefault()
if (cell != null) {
lastRow.add(cell);
@@ -214,8 +189,8 @@ public class Table extends AbstractTextContainer {
private void add(Cell chunk, int row, int col) {
rowCount = Math.max(rowCount, row + 1);
colCount = Math.max(colCount, col + 1);
unrotatedRowCount = Math.max(unrotatedRowCount, row + 1);
unrotatedColCount = Math.max(unrotatedColCount, col + 1);
CellPosition cp = new CellPosition(row, col);
cells.put(cp, chunk);
@@ -18,9 +18,9 @@ import com.iqser.red.service.redaction.v1.server.tableextraction.utils.Utils;
@Service
public class RulingCleaningService {
public CleanRulings getCleanRulings(List<Ruling> rulings, float minCharWidth, float minCharHeight){
public CleanRulings getCleanRulings(List<Ruling> rulings, float maxCharWidth, float maxCharHeight){
if (!rulings.isEmpty()) {
snapPoints(rulings, minCharWidth , minCharHeight);
snapPoints(rulings, maxCharWidth , maxCharHeight);
}
List<Ruling> vrs = new ArrayList<>();
@@ -2,7 +2,12 @@ package com.iqser.red.service.redaction.v1.server.visualization.service;
import java.awt.Color;
import java.io.IOException;
import java.util.ArrayList;
import java.util.GregorianCalendar;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Collectors;
import org.apache.commons.collections4.CollectionUtils;
import org.apache.pdfbox.pdmodel.PDDocument;
@@ -13,13 +18,19 @@ import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.graphics.color.PDColor;
import org.apache.pdfbox.pdmodel.graphics.color.PDDeviceRGB;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationText;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationTextMarkup;
import org.apache.pdfbox.text.TextPosition;
import org.springframework.stereotype.Service;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactionType;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
import com.iqser.red.service.redaction.v1.server.classification.model.Paragraph;
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
@@ -27,15 +38,12 @@ import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSeque
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityPositionSequence;
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
@Slf4j
@Service
@RequiredArgsConstructor
public class AnnotationHighlightService {
@@ -43,7 +51,10 @@ public class AnnotationHighlightService {
private final DictionaryService dictionaryService;
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction, ManualRedactions manualRedactions) throws IOException {
public void highlight(PDDocument document, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions) throws IOException {
Set<Integer> manualRedactionPages = getManualRedactionPages(manualRedactions);
for (int page = 1; page <= document.getNumberOfPages(); page++) {
@@ -51,51 +62,147 @@ public class AnnotationHighlightService {
drawSectionFrames(document, classifiedDoc, flatRedaction, pdPage, page);
if (classifiedDoc.getEntities().get(page) == null) {
continue;
if (classifiedDoc.getEntities().get(page) != null) {
addAnnotations(pdPage, classifiedDoc, flatRedaction, manualRedactions, page);
}
addAnnotations(pdPage, classifiedDoc, flatRedaction, manualRedactions, page);
addManualAnnotations(pdPage, classifiedDoc, manualRedactions, page);
if (manualRedactionPages.contains(page)) {
addManualAnnotations(pdPage, classifiedDoc, manualRedactions, page);
}
}
}
private void addAnnotations(PDPage pdPage, Document classifiedDoc, boolean flatRedaction, ManualRedactions manualRedactions, int page) throws IOException {
private Set<Integer> getManualRedactionPages(ManualRedactions manualRedactions) {
Set<Integer> manualRedactionPages = new HashSet<>();
if (manualRedactions == null) {
return manualRedactionPages;
}
manualRedactions.getEntriesToAdd().forEach(entry -> {
entry.getPositions().forEach(pos -> {
manualRedactionPages.add(pos.getPage());
});
});
return manualRedactionPages;
}
private void addAnnotations(PDPage pdPage, Document classifiedDoc, boolean flatRedaction,
ManualRedactions manualRedactions, int page) throws IOException {
List<PDAnnotation> annotations = pdPage.getAnnotations();
// Duplicates can exist due table extraction colums over multiple rows.
Set<String> processedIds = new HashSet<>();
entityLoop:
for (Entity entity : classifiedDoc.getEntities().get(page)) {
if (flatRedaction && !isRedactionType(entity)) {
continue;
}
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(entity);
boolean requestedToRemove = false;
List<Comment> comments = null;
for (EntityPositionSequence entityPositionSequence : entity.getPositionSequences()) {
if (manualRedactions != null && manualRedactions.getIdsToRemove()
.contains(entityPositionSequence.getId())) {
entity.setRedaction(false);
entity.setRedactionReason(entity.getRedactionReason() + ", removed by manual override");
redactionLogEntry.setManual(true);
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(entity);
if (processedIds.contains(entityPositionSequence.getId())) {
// TODO refactor this outer loop jump as soon as we have the time.
continue entityLoop;
} else {
processedIds.add(entityPositionSequence.getId());
}
for (TextPositionSequence textPositions : entityPositionSequence.getSequences()) {
if (manualRedactions != null && !manualRedactions.getIdsToRemove().isEmpty()) {
for (IdRemoval manualRemoval : manualRedactions.getIdsToRemove()) {
if (manualRemoval.getId().equals(entityPositionSequence.getId())) {
comments = manualRedactions.getComments().get(manualRemoval.getId());
String manualOverrideReason = null;
if (manualRemoval.getStatus().equals(Status.APPROVED)) {
entity.setRedaction(false);
redactionLogEntry.setRedacted(false);
redactionLogEntry.setStatus(Status.APPROVED);
manualOverrideReason = entity.getRedactionReason() + ", removed by manual override";
} else if (manualRemoval.getStatus().equals(Status.REQUESTED)) {
requestedToRemove = true;
manualOverrideReason = entity.getRedactionReason() + ", requested to remove";
redactionLogEntry.setStatus(Status.REQUESTED);
} else {
redactionLogEntry.setStatus(Status.DECLINED);
}
entity.setRedactionReason(manualOverrideReason != null ? manualOverrideReason : entity.getRedactionReason());
redactionLogEntry.setReason(manualOverrideReason);
redactionLogEntry.setManual(true);
redactionLogEntry.setManualRedactionType(ManualRedactionType.REMOVE);
}
}
Rectangle rectangle = textPositions.getRectangle();
redactionLogEntry.getPositions().add(rectangle);
annotations.add(createAnnotation(rectangle, entityPositionSequence.getId(), createAnnotationContent(entity), getColor(entity), !flatRedaction && !isHint(entity)));
}
if (CollectionUtils.isNotEmpty(entityPositionSequence.getSequences())) {
List<Rectangle> rectanglesPerLine = getRectanglesPerLine(entityPositionSequence.getSequences()
.stream()
.flatMap(seq -> seq.getTextPositions().stream())
.collect(Collectors.toList()), page);
if (manualRedactions != null) {
comments = manualRedactions.getComments().get(entityPositionSequence.getId());
}
redactionLogEntry.getPositions().addAll(rectanglesPerLine);
annotations.addAll(createAnnotation(rectanglesPerLine, entityPositionSequence.getId(), createAnnotationContent(entity), getColor(entity, requestedToRemove), comments, !isHint(entity)));
}
redactionLogEntry.setId(entityPositionSequence.getId());
// FIXME ids should never be null. Figure out why this happens.
if (redactionLogEntry.getId() != null) {
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
private void addManualAnnotations(PDPage pdPage, Document classifiedDoc, ManualRedactions manualRedactions, int page) throws IOException {
private List<Rectangle> getRectanglesPerLine(List<TextPosition> textPositions, int page) {
List<Rectangle> rectangles = new ArrayList<>();
if (textPositions.size() == 1) {
rectangles.add(new TextPositionSequence(textPositions, page).getRectangle());
} else {
float y = textPositions.get(0).getYDirAdj();
int startIndex = 0;
for (int i = 1; i < textPositions.size(); i++) {
float yDirAdj = textPositions.get(i).getYDirAdj();
if (yDirAdj != y) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, i), page).getRectangle());
y = yDirAdj;
startIndex = i;
}
}
if (startIndex != textPositions.size() - 1) {
rectangles.add(new TextPositionSequence(textPositions.subList(startIndex, textPositions.size()), page).getRectangle());
}
}
return rectangles;
}
private void addManualAnnotations(PDPage pdPage, Document classifiedDoc, ManualRedactions manualRedactions,
int page) throws IOException {
if (manualRedactions == null) {
return;
@@ -105,32 +212,40 @@ public class AnnotationHighlightService {
for (ManualRedactionEntry manualRedactionEntry : manualRedactions.getEntriesToAdd()) {
String id = IdBuilder.buildId(manualRedactionEntry);
String id = manualRedactionEntry.getId();
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(manualRedactionEntry);
RedactionLogEntry redactionLogEntry = createRedactionLogEntry(manualRedactionEntry, id);
List<Rectangle> rectanglesOnPage = new ArrayList<>();
for (Rectangle rectangle : manualRedactionEntry.getPositions()) {
if (page != rectangle.getPage()) {
continue;
if (page == rectangle.getPage()) {
rectanglesOnPage.add(rectangle);
redactionLogEntry.getPositions().add(rectangle);
}
PDAnnotationTextMarkup highlight = createAnnotation(rectangle, id, createAnnotationContent(manualRedactionEntry), getColor(manualRedactionEntry
.getType()), true);
annotations.add(highlight);
redactionLogEntry.getPositions().add(rectangle);
}
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
if (!rectanglesOnPage.isEmpty() && !approvedAndShouldBeInDictionary(manualRedactionEntry)) {
annotations.addAll(createAnnotation(rectanglesOnPage, id, createAnnotationContent(manualRedactionEntry), getColorForManualAdd(manualRedactionEntry
.getType(), manualRedactionEntry.getStatus()), manualRedactions.getComments().get(id), true));
classifiedDoc.getRedactionLogEntities().add(redactionLogEntry);
}
}
}
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry) {
private boolean approvedAndShouldBeInDictionary(ManualRedactionEntry manualRedactionEntry) {
return manualRedactionEntry.getStatus().equals(Status.APPROVED) && manualRedactionEntry.isAddToDictionary();
}
private RedactionLogEntry createRedactionLogEntry(ManualRedactionEntry manualRedactionEntry, String id) {
return RedactionLogEntry.builder()
.id(id)
.color(getColor(manualRedactionEntry.getType()))
.reason(manualRedactionEntry.getReason())
.legalBasis(manualRedactionEntry.getLegalBasis())
.value(manualRedactionEntry.getValue())
.type(manualRedactionEntry.getType())
.redacted(true)
@@ -138,6 +253,8 @@ public class AnnotationHighlightService {
.section(manualRedactionEntry.getSection())
.sectionNumber(manualRedactionEntry.getSectionNumber())
.manual(true)
.status(manualRedactionEntry.getStatus())
.manualRedactionType(ManualRedactionType.ADD)
.build();
}
@@ -145,70 +262,117 @@ public class AnnotationHighlightService {
private RedactionLogEntry createRedactionLogEntry(Entity entity) {
return RedactionLogEntry.builder()
.color(getColor(entity))
.color(getColor(entity, false))
.reason(entity.getRedactionReason())
.legalBasis(entity.getLegalBasis())
.value(entity.getWord())
.type(entity.getType())
.redacted(entity.isRedaction())
.isHint(isHint(entity))
.section(entity.getHeadline())
.sectionNumber(entity.getSectionNumber())
.matchedRule(entity.getMatchedRule())
.build();
}
private PDAnnotationTextMarkup createAnnotation(Rectangle rectangle, String id, String content, float[] color, boolean popup) {
private List<PDAnnotation> createAnnotation(List<Rectangle> rectangles, String id, String content, float[] color,
List<Comment> comments, boolean popup) {
List<PDAnnotation> annotations = new ArrayList<>();
PDAnnotationTextMarkup annotation = new PDAnnotationTextMarkup(PDAnnotationTextMarkup.SUB_TYPE_HIGHLIGHT);
annotation.constructAppearances();
annotation.setRectangle(toPDRectangle(rectangle));
annotation.setQuadPoints(toQuadPoints(rectangle));
PDRectangle pdRectangle = toPDRectangle(rectangles);
annotation.setRectangle(pdRectangle);
annotation.setQuadPoints(toQuadPoints(rectangles));
if (popup) {
annotation.setAnnotationName(id);
annotation.setTitlePopup(id);
annotation.setContents(content);
}
annotation.setTitlePopup(id);
annotation.setAnnotationName(id);
annotation.setColor(new PDColor(color, PDDeviceRGB.INSTANCE));
return annotation;
annotations.add(annotation);
if (comments != null) {
for (Comment comment : comments) {
PDAnnotationText txtAnnot = new PDAnnotationText();
txtAnnot.setAnnotationName(comment.getId());
txtAnnot.setInReplyTo(annotation); // Reference to highlight annotation
txtAnnot.setName(PDAnnotationText.NAME_COMMENT);
txtAnnot.setCreationDate(GregorianCalendar.from(comment.getDate().toZonedDateTime()));
txtAnnot.setTitlePopup(comment.getUser());
txtAnnot.setContents(comment.getText());
txtAnnot.setRectangle(pdRectangle);
annotations.add(txtAnnot);
}
}
return annotations;
}
private String createAnnotationContent(Entity entity) {
return new StringBuilder().append("\nRule ")
.append(entity.getMatchedRule())
.append(" matched")
.append("\n\n")
.append(entity.getRedactionReason())
.append("\n\nIn Section : \"")
.append(entity.getHeadline())
.append("\"")
.toString();
return "\nRule " + entity.getMatchedRule() + " matched\n\n" + entity.getRedactionReason() + "\n\nLegal basis:" + entity
.getLegalBasis() + "\n\nIn section: \"" + entity.getHeadline() + "\"";
}
private String createAnnotationContent(ManualRedactionEntry entry) {
return new StringBuilder().append("\nManual Redaction")
.append("\n\nIn Section : \"")
.append(entry.getSection())
.append("\"")
.toString();
return "\nManual Redaction\n\nIn Section : \"" + entry.getSection() + "\"";
}
private PDRectangle toPDRectangle(Rectangle rectangle) {
private PDRectangle toPDRectangle(List<Rectangle> rectangles) {
float lowerLeftX = Float.MAX_VALUE;
float upperRightX = 0;
float lowerLeftY = 0;
float upperRightY = Float.MAX_VALUE;
for (Rectangle rectangle : rectangles) {
if (rectangle.getTopLeft().getX() < lowerLeftX) {
lowerLeftX = rectangle.getTopLeft().getX();
}
if (rectangle.getTopLeft().getX() + rectangle.getWidth() > upperRightX) {
upperRightX = rectangle.getTopLeft().getX() + rectangle.getWidth();
}
if (rectangle.getTopLeft().getY() + rectangle.getHeight() > lowerLeftY) {
lowerLeftY = rectangle.getTopLeft().getY() + rectangle.getHeight();
}
if (rectangle.getTopLeft().getY() < upperRightY) {
upperRightY = rectangle.getTopLeft().getY();
}
}
PDRectangle annotationPosition = new PDRectangle();
annotationPosition.setLowerLeftX(rectangle.getTopLeft().getX());
annotationPosition.setLowerLeftY(rectangle.getTopLeft().getY() + rectangle.getHeight());
annotationPosition.setUpperRightX(rectangle.getTopLeft().getX() + rectangle.getWidth());
annotationPosition.setUpperRightY(rectangle.getTopLeft().getY());
annotationPosition.setLowerLeftX(lowerLeftX);
annotationPosition.setLowerLeftY(lowerLeftY);
annotationPosition.setUpperRightX(upperRightX);
annotationPosition.setUpperRightY(upperRightY);
return annotationPosition;
}
private float[] toQuadPoints(Rectangle rectangle) {
private float[] toQuadPoints(List<Rectangle> rectangles) {
float[] quadPoints = new float[rectangles.size() * 8];
int i = 0;
for (Rectangle rectangle : rectangles) {
float[] quadPoint = toQuadPoint(rectangle);
for (int j = 0; j <= 7; j++) {
quadPoints[i + j] = quadPoint[j];
}
i += 8;
}
return quadPoints;
}
private float[] toQuadPoint(Rectangle rectangle) {
// quadPoints is array of x,y coordinates in Z-like order (top-left, top-right, bottom-left,bottom-right)
// of the area to be highlighted
@@ -224,17 +388,17 @@ public class AnnotationHighlightService {
if (!entity.isRedaction()) {
return false;
}
if (isHint(entity)) {
return false;
}
return true;
return !isHint(entity);
}
private float[] getColor(Entity entity) {
private float[] getColor(Entity entity, boolean requestedToRemove) {
if (requestedToRemove) {
return dictionaryService.getRequestRemoveColor();
}
if (!entity.isRedaction() && !isHint(entity)) {
return new float[]{0.627f, 0.627f, 0.627f};
return dictionaryService.getNotRedactedColor();
}
if (!dictionaryService.getEntryColors().containsKey(entity.getType())) {
return dictionaryService.getDefaultColor();
@@ -243,6 +407,17 @@ public class AnnotationHighlightService {
}
private float[] getColorForManualAdd(String type, Status status) {
if (status.equals(Status.REQUESTED)) {
return dictionaryService.getRequestAddColor();
} else if (status.equals(Status.DECLINED)) {
return dictionaryService.getNotRedactedColor();
}
return getColor(type);
}
private float[] getColor(String type) {
if (!dictionaryService.getEntryColors().containsKey(type)) {
@@ -255,14 +430,12 @@ public class AnnotationHighlightService {
private boolean isHint(Entity entity) {
List<String> hintTypes = dictionaryService.getHintTypes();
if (CollectionUtils.isNotEmpty(hintTypes) && hintTypes.contains(entity.getType())) {
return true;
}
return false;
return CollectionUtils.isNotEmpty(hintTypes) && hintTypes.contains(entity.getType());
}
private void drawSectionFrames(PDDocument document, Document classifiedDoc, boolean flatRedaction, PDPage pdPage, int page) throws IOException {
private void drawSectionFrames(PDDocument document, Document classifiedDoc, boolean flatRedaction, PDPage pdPage,
int page) throws IOException {
if (flatRedaction) {
return;
@@ -334,4 +507,4 @@ public class AnnotationHighlightService {
}
}
}
}
@@ -12,5 +12,6 @@ management:
endpoint:
metrics.enabled: ${monitoring.enabled:false}
prometheus.enabled: ${monitoring.enabled:false}
health.enabled: true
endpoints.web.exposure.include: prometheus, health
metrics.export.prometheus.enabled: ${monitoring.enabled:false}
@@ -2,10 +2,6 @@ spring:
application:
name: redaction-service-v1
management:
endpoints:
web:
base-path: /
path-mapping:
health: "health"
management.endpoints:
web.base-path: /
enabled-by-default: false
@@ -1,5 +1,6 @@
package com.iqser.red.service.redaction.v1.server;
import static org.assertj.core.api.Assertions.assertThat;
import static org.mockito.Mockito.when;
import static org.springframework.boot.test.context.SpringBootTest.WebEnvironment.DEFINED_PORT;
@@ -13,11 +14,13 @@ import java.io.InputStream;
import java.io.InputStreamReader;
import java.net.URL;
import java.nio.charset.StandardCharsets;
import java.time.OffsetDateTime;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.UUID;
import java.util.stream.Collectors;
import org.apache.commons.io.IOUtils;
@@ -37,17 +40,21 @@ import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResult;
import com.iqser.red.service.redaction.v1.model.Comment;
import com.iqser.red.service.redaction.v1.model.IdRemoval;
import com.iqser.red.service.redaction.v1.model.ManualRedactionEntry;
import com.iqser.red.service.redaction.v1.model.ManualRedactions;
import com.iqser.red.service.redaction.v1.model.Point;
import com.iqser.red.service.redaction.v1.model.Rectangle;
import com.iqser.red.service.redaction.v1.model.RedactionLogEntry;
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
import com.iqser.red.service.redaction.v1.model.RedactionResult;
import com.iqser.red.service.redaction.v1.model.Status;
import com.iqser.red.service.redaction.v1.server.client.DictionaryClient;
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
import com.iqser.red.service.redaction.v1.server.controller.RedactionController;
@@ -62,6 +69,7 @@ public class RedactionIntegrationTest {
private static final String VERTEBRATES_CODE = "vertebrate";
private static final String ADDRESS_CODE = "address";
private static final String NAME_CODE = "name";
private static final String SPONSOR = "sponsor";
private static final String NO_REDACTION_INDICATOR = "no_redaction_indicator";
private static final String REDACTION_INDICATOR = "redaction_indicator";
private static final String HINT_ONLY = "hint_only";
@@ -77,9 +85,10 @@ public class RedactionIntegrationTest {
private DictionaryClient dictionaryClient;
private final Map<String, List<String>> dictionary = new HashMap<>();
private final Map<String, float[]> typeColorMap = new HashMap<>();
private final Map<String, String> typeColorMap = new HashMap<>();
private final Map<String, Boolean> hintTypeMap = new HashMap<>();
private final Map<String, Boolean> caseInSensitiveMap = new HashMap<>();
private final Colors colors = new Colors();
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@@ -116,11 +125,12 @@ public class RedactionIntegrationTest {
when(dictionaryClient.getDictionaryForType(VERTEBRATES_CODE)).thenReturn(getDictionaryResponse(VERTEBRATES_CODE));
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(getDictionaryResponse(ADDRESS_CODE));
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(getDictionaryResponse(NAME_CODE));
when(dictionaryClient.getDictionaryForType(SPONSOR)).thenReturn(getDictionaryResponse(SPONSOR));
when(dictionaryClient.getDictionaryForType(NO_REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(NO_REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(REDACTION_INDICATOR)).thenReturn(getDictionaryResponse(REDACTION_INDICATOR));
when(dictionaryClient.getDictionaryForType(HINT_ONLY)).thenReturn(getDictionaryResponse(HINT_ONLY));
when(dictionaryClient.getDictionaryForType(MUST_REDACT)).thenReturn(getDictionaryResponse(MUST_REDACT));
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor(new float[]{1f, 0.502f, 0f}));
when(dictionaryClient.getColors()).thenReturn(colors);
}
@@ -131,6 +141,11 @@ public class RedactionIntegrationTest {
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(SPONSOR, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/sponsor_companies.txt")
.stream()
.map(this::cleanDictionaryEntry)
.collect(Collectors.toSet()));
dictionary.computeIfAbsent(VERTEBRATES_CODE, v -> new ArrayList<>())
.addAll(ResourceLoader.load("dictionaries/vertebrates.txt")
.stream()
@@ -172,17 +187,19 @@ public class RedactionIntegrationTest {
private void loadTypeForTest() {
typeColorMap.put(VERTEBRATES_CODE, new float[]{0, 1, 0});
typeColorMap.put(ADDRESS_CODE, new float[]{0, 1, 1});
typeColorMap.put(NAME_CODE, new float[]{1, 1, 0});
typeColorMap.put(NO_REDACTION_INDICATOR, new float[]{0.8f, 0, 0.8f});
typeColorMap.put(REDACTION_INDICATOR, new float[]{1, 0.502f, 0.1f});
typeColorMap.put(HINT_ONLY, new float[]{0.8f, 1, 0.8f});
typeColorMap.put(MUST_REDACT, new float[]{1, 0, 0});
typeColorMap.put(VERTEBRATES_CODE, "#00ff00");
typeColorMap.put(ADDRESS_CODE, "#00ffff");
typeColorMap.put(NAME_CODE, "#ffff00");
typeColorMap.put(SPONSOR, "#2c21fc");
typeColorMap.put(NO_REDACTION_INDICATOR, "#e600ff");
typeColorMap.put(REDACTION_INDICATOR, "#ff7700");
typeColorMap.put(HINT_ONLY, "#00fcb1");
typeColorMap.put(MUST_REDACT, "#ff0000");
hintTypeMap.put(VERTEBRATES_CODE, true);
hintTypeMap.put(ADDRESS_CODE, false);
hintTypeMap.put(NAME_CODE, false);
hintTypeMap.put(SPONSOR, false);
hintTypeMap.put(NO_REDACTION_INDICATOR, true);
hintTypeMap.put(REDACTION_INDICATOR, true);
hintTypeMap.put(HINT_ONLY, true);
@@ -191,10 +208,16 @@ public class RedactionIntegrationTest {
caseInSensitiveMap.put(VERTEBRATES_CODE, true);
caseInSensitiveMap.put(ADDRESS_CODE, false);
caseInSensitiveMap.put(NAME_CODE, false);
caseInSensitiveMap.put(SPONSOR, false);
caseInSensitiveMap.put(NO_REDACTION_INDICATOR, true);
caseInSensitiveMap.put(REDACTION_INDICATOR, true);
caseInSensitiveMap.put(HINT_ONLY, true);
caseInSensitiveMap.put(MUST_REDACT, true);
colors.setDefaultColor("#acfc00");
colors.setNotRedacted("#cccccc");
colors.setRequestAdd("#04b093");
colors.setRequestRemove("#04b093");
}
@@ -204,7 +227,7 @@ public class RedactionIntegrationTest {
.stream()
.map(typeColor -> TypeResult.builder()
.type(typeColor.getKey())
.color(typeColor.getValue())
.hexColor(typeColor.getValue())
.isHint(hintTypeMap.get(typeColor.getKey()))
.isCaseInsensitive(caseInSensitiveMap.get(typeColor.getKey()))
.build())
@@ -216,7 +239,7 @@ public class RedactionIntegrationTest {
private DictionaryResponse getDictionaryResponse(String type) {
return DictionaryResponse.builder()
.color(typeColorMap.get(type))
.hexColor(typeColorMap.get(type))
.entries(dictionary.get(type))
.isHint(hintTypeMap.get(type))
.isCaseInsensitive(caseInSensitiveMap.get(type))
@@ -240,7 +263,16 @@ public class RedactionIntegrationTest {
.document(IOUtils.toByteArray(new FileInputStream(path)))
.build();
System.out.println("Redacting file : " + path.getName());
redactionController.redact(request);
RedactionResult result = redactionController.redact(request);
Map<String, List<RedactionLogEntry>> duplicates = new HashMap<>();
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
duplicates.computeIfAbsent(entry.getId(), v -> new ArrayList<>()).add(entry);
});
duplicates.entrySet().forEach(entry -> {
assertThat(entry.getValue().size()).isEqualTo(1);
});
}
}
@@ -269,7 +301,7 @@ public class RedactionIntegrationTest {
System.out.println("redactionTest");
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Trinexapac/96 Trinexapac-ethyl_RAR_09_Volume_3CA_B-7_2018-02-23.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_01_Volume_1_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -319,9 +351,26 @@ public class RedactionIntegrationTest {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Single Table.pdf");
ManualRedactions manualRedactions = new ManualRedactions();
manualRedactions.setIdsToRemove(Set.of("0836727c3508a0b2ea271da69c04cc2f"));
String manualAddId = UUID.randomUUID().toString();
Comment comment = Comment.builder()
.date(OffsetDateTime.now())
.user("TEST_USER")
.text("This is a comment test")
.build();
manualRedactions.setIdsToRemove(Set.of(IdRemoval.builder()
.id("0836727c3508a0b2ea271da69c04cc2f")
.status(Status.REQUESTED)
.build()));
manualRedactions.getComments().put("e5be0f1d941bbb92a068e198648d06c4", List.of(comment));
manualRedactions.getComments().put("0836727c3508a0b2ea271da69c04cc2f", List.of(comment));
manualRedactions.getComments().put(manualAddId, List.of(comment));
ManualRedactionEntry manualRedactionEntry = new ManualRedactionEntry();
manualRedactionEntry.setId(manualAddId);
manualRedactionEntry.setStatus(Status.REQUESTED);
manualRedactionEntry.setType("name");
manualRedactionEntry.setValue("O'Loughlin C.K.");
manualRedactionEntry.setReason("Manual Redaction");
@@ -350,8 +399,7 @@ public class RedactionIntegrationTest {
public void classificationTest() throws IOException {
System.out.println("classificationTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -369,8 +417,7 @@ public class RedactionIntegrationTest {
public void sectionsTest() throws IOException {
System.out.println("sectionsTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " + "Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -388,8 +435,7 @@ public class RedactionIntegrationTest {
public void htmlTablesTest() throws IOException {
System.out.println("htmlTablesTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Fludioxonil/51 " +
"Fludioxonil_RAR_02_Volume_2_2018-02-21.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/line_breaks.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -407,8 +453,7 @@ public class RedactionIntegrationTest {
public void htmlTableRotationTest() throws IOException {
System.out.println("htmlTableRotationTest");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S" +
"-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
ClassPathResource pdfFileResource = new ClassPathResource("files/Metolachlor/S-Metolachlor_RAR_02_Volume_2_2018-09-06.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
@@ -422,14 +467,56 @@ public class RedactionIntegrationTest {
}
@Test
public void phantomCellsDocumentTest() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Phantom Cells.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
result.getRedactionLog().getRedactionLogEntry().forEach(entry -> {
if (!entry.isHint()) {
assertThat(entry.getReason()).isEqualTo("Not redacted because row is not a vertebrate study");
}
});
}
@Test
public void sponsorCompanyTest() throws IOException {
long start = System.currentTimeMillis();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/sponsor_companies.pdf");
RedactionRequest request = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
request.setFlatRedaction(false);
RedactionResult result = redactionController.redact(request);
try (FileOutputStream fileOutputStream = new FileOutputStream("/tmp/Redacted.pdf")) {
fileOutputStream.write(result.getDocument());
}
long end = System.currentTimeMillis();
System.out.println("duration: " + (end - start));
System.out.println("numberOfPages: " + result.getNumberOfPages());
}
private static String loadFromClassPath(String path) {
URL resource = ResourceLoader.class.getClassLoader().getResource(path);
if (resource == null) {
throw new IllegalArgumentException("could not load classpath resource: drools/rules.drl");
}
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(),
StandardCharsets.UTF_8))) {
try (BufferedReader br = new BufferedReader(new InputStreamReader(resource.openStream(), StandardCharsets.UTF_8))) {
StringBuilder sb = new StringBuilder();
String str;
while ((str = br.readLine()) != null) {
@@ -35,7 +35,7 @@ import org.springframework.context.annotation.Bean;
import org.springframework.core.io.ClassPathResource;
import org.springframework.test.context.junit4.SpringRunner;
import com.iqser.red.service.configuration.v1.api.model.DefaultColor;
import com.iqser.red.service.configuration.v1.api.model.Colors;
import com.iqser.red.service.configuration.v1.api.model.DictionaryResponse;
import com.iqser.red.service.configuration.v1.api.model.RulesResponse;
import com.iqser.red.service.configuration.v1.api.model.TypeResponse;
@@ -55,8 +55,11 @@ public class EntityRedactionServiceTest {
private static final String DEFAULT_RULES = loadFromClassPath("drools/rules.drl");
private static final String NAME_CODE = "name";
private static final String ADDRESS_CODE = "address";
private static final String SPONSOR_CODE = "sponsor";
private static final AtomicLong DICTIONARY_VERSION = new AtomicLong();
private static final AtomicLong RULES_VERSION = new AtomicLong();
@MockBean
private DictionaryClient dictionaryClient;
@@ -69,6 +72,9 @@ public class EntityRedactionServiceTest {
@Autowired
private PdfSegmentationService pdfSegmentationService;
@Autowired
private DroolsExecutionService droolsExecutionService;
@TestConfiguration
public static class RedactionIntegrationTestConfiguration {
@@ -130,7 +136,35 @@ public class EntityRedactionServiceTest {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1)).hasSize(5); // 4 out of 5 entities recognized on page 1
assertThat(classifiedDoc.getEntities().get(1)).hasSize(7);// 3 author cells, 1 address, 1 Y and 2 N entities
}
}
@Test
public void testNestedRedaction() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/nested_redaction.pdf");
RedactionRequest redactionRequest = RedactionRequest.builder()
.document(IOUtils.toByteArray(pdfFileResource.getInputStream()))
.build();
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Casey, H.W.", "O’Loughlin, C.K.", "Salamon, C.M.", "Smith, S.H."))
.build();
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Toxigenics, Inc., Decatur, IL 62526, USA"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
try (PDDocument pdDocument = PDDocument.load(new ByteArrayInputStream(redactionRequest.getDocument()))) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1)).hasSize(7);// 3 author cells, 1 address, 1 Y and 2 N entities
}
}
@@ -185,7 +219,7 @@ public class EntityRedactionServiceTest {
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // two pages
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 9)
.count()).isEqualTo(10);
@@ -193,6 +227,105 @@ public class EntityRedactionServiceTest {
}
@Test
public void testApplicantInTableRedaction() throws IOException {
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" +
"rule \"6: Redact contact information if applicant is found\"\n" +
" when\n" +
" eval(section.headlineContainsWord(\"applicant\") || section.getText().contains(\"Applicant\"));\n" +
" then\n" +
" section.redactLineAfter(\"Name:\", \"address\", 6, \"Applicant information was found\", \"Reg" +
" (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"Address:\", \"Contact\", \"address\", 6, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Contact point:\", \"address\", 6, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Phone:\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Fax:\", \"address\", 6, \"Applicant information was found\", \"Reg " +
"(EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Tel.:\", \"address\", 6, \"Applicant information was found\", \"Reg" +
" (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Tel:\", \"address\", 6, \"Applicant information was found\", \"Reg " +
"(EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"E-mail:\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Email:\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Contact:\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Telephone number:\", \"address\", 6, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Fax number:\", \"address\", 6, \"Applicant information was found\"," +
" \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactLineAfter(\"Telephone:\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"No:\", \"Fax\", \"address\", 6, \"Applicant information was found\", " +
"\"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redactBetween(\"Contact:\", \"Tel.:\", \"address\", 6, \"Applicant information was found\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" end";
when(rulesClient.getVersion()).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules()).thenReturn(new RulesResponse(tableRules));
droolsExecutionService.updateRules();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Applicant Producer Table.pdf");
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/names.txt")))
.build();
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/addresses.txt")))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 6)
.count()).isEqualTo(18);
}
}
@Test
public void testSponsorInCell() throws IOException {
String tableRules = "package drools\n" +
"\n" +
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" + "rule \"11: Redact sponsor company\"\n" + " when\n" + " " +
"Section(searchText.toLowerCase().contains(\"batches produced at\"))\n" + " then\n" + " section" +
".redactIfPrecededBy(\"batches produced at\", \"sponsor\", 11, \"Redacted because it represents a " +
"sponsor company\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" + " end";
when(rulesClient.getVersion()).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules()).thenReturn(new RulesResponse(tableRules));
droolsExecutionService.updateRules();
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/batches_new_line.pdf");
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(new ArrayList<>(ResourceLoader.load("dictionaries/sponsor_companies.txt")))
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE)).thenReturn(dictionaryResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream()
.filter(entity -> entity.getMatchedRule() == 11)
.count()).isEqualTo(1);
}
}
@Test
public void headerPropagation() throws IOException {
@@ -214,7 +347,53 @@ public class EntityRedactionServiceTest {
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(2); // two pages
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(8);
assertThat(classifiedDoc.getEntities().get(2).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(4);
assertThat(classifiedDoc.getEntities().get(2).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(5); // 2 names, 1 address, 2 Y
}
pdfFileResource = new ClassPathResource("files/Minimal Examples/Header Propagation2.pdf");
dictionaryResponse = DictionaryResponse.builder()
.entries(Arrays.asList("Tribolet, R.", "Muir, G.", "Kühne-Thu, H.", "Close, C."))
.build();
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 9).count()).isEqualTo(3);
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 8).count()).isEqualTo(8);
}
}
@Test
public void testNGuideline() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Empty Tabular Data.pdf");
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Aldershof S."))
.build();
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.entries(Collections.singletonList("Novartis Crop Protection AG, Basel, Switzerland"))
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document classifiedDoc = pdfSegmentationService.parseDocument(pdDocument);
entityRedactionService.processDocument(classifiedDoc, null);
assertThat(classifiedDoc.getEntities()).hasSize(1); // one page
assertThat(classifiedDoc.getEntities().get(1).stream().filter(entity -> entity.getMatchedRule() == 8).count()).isEqualTo(6);
}
}
@@ -226,24 +405,53 @@ public class EntityRedactionServiceTest {
"import com.iqser.red.service.redaction.v1.server.redaction.model.Section\n" +
"\n" +
"global Section section\n" +
"rule \"8: Not redacted because Vertebrate Study = N\"\n" +
" when\n" +
" Section(rowEquals(\"Vertebrate study Y/N\", \"N\") || rowEquals(\"Vertebrate study Y/N\", \"No\"))\n" +
" then\n" +
" section.redactNotCell(\"Author(s)\", 8, \"name\", \"Not redacted because row is not a vertebrate study\");\n" +
" section.redactNot(\"address\", 8, \"Not redacted because row is not a vertebrate study\");\n" +
" section.highlightCell(\"Vertebrate study Y/N\", 8, \"hint_only\");\n" +
" end\n" +
"rule \"9: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study\"\n" +
" when\n" +
" Section(isVertebrateStudy())\n" +
" Section(rowEquals(\"Vertebrate study Y/N\", \"Y\") || rowEquals(\"Vertebrate study Y/N\", " +
"\"Yes\"))\n" +
" then\n" +
" section.redact(\"name\", 9, \"Redacted because row is a vertebrate study\");\n" +
" section.redact(\"address\", 9, \"Redacted because rows is a vertebrate study\");\n" +
" section.redactCell(\"Author(s)\", 9, \"name\", \"Redacted because row is a vertebrate study\", \"Reg (EC) No 1107/2009 Art. 63 (2g)\");\n" +
" section.redact(\"address\", 9, \"Redacted because row is a vertebrate study\", \"Reg (EC) No" +
" 1107/2009 Art. 63 (2g)\");\n" +
" section.highlightCell(\"Vertebrate study Y/N\", 9, \"must_redact\");\n" +
" end";
when(rulesClient.getVersion()).thenReturn(1L);
when(rulesClient.getVersion()).thenReturn(RULES_VERSION.incrementAndGet());
when(rulesClient.getRules()).thenReturn(new RulesResponse(tableRules));
TypeResponse typeResponse = TypeResponse.builder()
.types(Arrays.asList(
TypeResult.builder().type(NAME_CODE).color(new float[]{1, 1, 0}).build(),
TypeResult.builder().type(ADDRESS_CODE).color(new float[]{0, 1, 1}).build()))
TypeResult.builder().type(NAME_CODE).hexColor("#ffff00").build(),
TypeResult.builder().type(ADDRESS_CODE).hexColor("#ff00ff").build(),
TypeResult.builder().type(SPONSOR_CODE).hexColor("#00ffff").build()))
.build();
when(dictionaryClient.getVersion()).thenReturn(DICTIONARY_VERSION.incrementAndGet());
when(dictionaryClient.getAllTypes()).thenReturn(typeResponse);
when(dictionaryClient.getDefaultColor()).thenReturn(new DefaultColor());
// Default empty return to prevent NPEs
DictionaryResponse dictionaryResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(NAME_CODE)).thenReturn(dictionaryResponse);
DictionaryResponse addressResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(ADDRESS_CODE)).thenReturn(addressResponse);
DictionaryResponse sponsorResponse = DictionaryResponse.builder()
.build();
when(dictionaryClient.getDictionaryForType(SPONSOR_CODE)).thenReturn(sponsorResponse);
Colors colors = new Colors();
colors.setDefaultColor("#acfc00");
colors.setNotRedacted("#cccccc");
colors.setRequestAdd("#04b093");
colors.setRequestRemove("#04b093");
when(dictionaryClient.getColors()).thenReturn(colors);
}
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.segmentation;
import static org.assertj.core.api.Assertions.assertThat;
import java.io.IOException;
import java.util.Collections;
import java.util.List;
import java.util.stream.Collectors;
@@ -94,6 +95,46 @@ public class PdfSegmentationServiceTest {
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(0)
.stream()
.map(Collections::singletonList)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
.allMatch(row -> row.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList())
.equals(firstTableHeaderCells)))
.isTrue();
}
}
@Test
public void testMultiPageMetadataPropagation() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Merge Multi Page Table.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table firstTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(firstTable.getColCount()).isEqualTo(9);
assertThat(firstTable.getRowCount()).isEqualTo(5);
Table secondTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(1);
assertThat(secondTable.getColCount()).isEqualTo(9);
assertThat(secondTable.getRowCount()).isEqualTo(6);
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(firstTable.getRowCount() - 1)
.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
@@ -105,4 +146,44 @@ public class PdfSegmentationServiceTest {
}
}
@Test
public void testHeaderCellsForRotatedTable() throws IOException {
ClassPathResource pdfFileResource = new ClassPathResource("files/Minimal Examples/Rotated Table Headers.pdf");
try (PDDocument pdDocument = PDDocument.load(pdfFileResource.getInputStream())) {
Document document = pdfSegmentationService.parseDocument(pdDocument);
assertThat(document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())).isNotEmpty();
Table firstTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(0);
assertThat(firstTable.getColCount()).isEqualTo(8);
assertThat(firstTable.getRowCount()).isEqualTo(1);
Table secondTable = document.getParagraphs()
.stream()
.flatMap(paragraph -> paragraph.getTables().stream())
.collect(Collectors.toList())
.get(1);
assertThat(secondTable.getColCount()).isEqualTo(8);
assertThat(secondTable.getRowCount()).isEqualTo(6);
List<List<Cell>> firstTableHeaderCells = firstTable.getRows()
.get(0)
.stream()
.map(Collections::singletonList)
.collect(Collectors.toList());
assertThat(secondTable.getRows().stream()
.allMatch(row -> row.stream()
.map(Cell::getHeaderCells)
.collect(Collectors.toList())
.equals(firstTableHeaderCells)))
.isTrue();
}
}
}
@@ -227,6 +227,7 @@ CIBA-GEIGY Limited, Toxicology Services, Short-term Toxicology, 4332 Stein, Swit
CIBA-GEIGY Ltd. CH-4002 Basle, Switzerland
CIBA-GEIGY Ltd., Product Safety, Ecotoxicology, CH-4002 Basel, Switzerland
CIBA-GEIGY Ltd., Switzerland
CibaGeigy Ltd., CH-4002, Basel, Switzerland
CIP Chemisches Institut Pforzheim GmbH, Pforzheim, Germany
CL PHARMA AG
CL PHARMA AG, Linz, Austria
@@ -1563,4 +1564,36 @@ Zyma SA
Zyma SA, Nyon, Switzerland
“Bayer CropScience AG” Monheim
“Biologische Bundesanstalt (BBA)”, Berlin-Dahlem
“W. Neudorff GmbH KG”, An der Mühle 3, D- 31860 Emmertal
“W. Neudorff GmbH KG”, An der Mühle 3, D- 31860 Emmertal
Syngenta Ltd Jealott’s Hill Int. Research Centre Bracknell, Berkshire, RG42 6EY, United Kingdom
MITOX Consultants Science Park 408, 1098XH Amsterdam, NL
Eurofins Agrosciences Services Eco-Chem GmbH, Eutinger Str. 24, 75233 Niefern-Öschelbronn, Germany
Fraunhofer Institute for Molecular Biology and Applied Ecology (IME), 57392 Schmallenberg, Germany
Syngenta Crop Protection AGRCC Cytotest Cell Research GmbH, Germany
BASF SE, Ludwigshafen, Germany
AGRCC Ltd., Switzerland
AGWIL Research Lab. Inc., USA
Charles River Laboratories, UK
Ciba-Geigy Corporation Environmental Health Centre, Farmington
Syngenta Crop Protection, Inc., Greensboro, United States
Experimental Toxicology and Ecology, BASF SE
Wildlife International, Ltd. 8598 Commerce Drive Easton, MD 21601 USA
Küberich, Wiesentheid / Geesdorf, Germany
CiToxLAB Hungary Ltd. H8200 Veszprém, Szabadságpuszta, Hungary
Forellenzucht Trostadt GbR, Trostadt, Germany
Osage Catfisheries Inc., Osage Beach, MO, USA
Fischzucht Rhönforelle, Gersfeld, Germany
Aquatic Biosystems, Ft. Collins, Colorado, USA
West Aquarium GmbH, 37431 Bad Lauterberg, Germany
Aquatic BioSystems, Fort Collins, Colorado, USA
“U.S. Environmental Research Laboratory”, Duluth, Minnesota, USA
Continuous laboratory cultures
Wildlife International, Ltd. 8598 Commerce Drive Easton, MD 21601 USA
Osage Catfisheries, Inc., Osage Beach, MO, USA
Osage catfisheries, Osage Beach (MO), USA
Ciba-Geigy Corporation, Environmental Health Centre, Farmington
Novartis Crop Protection Inc., Greensboro, United States
Syngenta, Jealott’s Hill International Research Station, UK
Hillandale Farms, Kenansville, Florida, USA
Jealott’s Hill International Research Station, UK
L & C Dairy, Avon Park, Florida, UK
@@ -1,3 +1,2 @@
Batches Produced at
CTL
for determination of residues
determination of residues
@@ -5290,6 +5290,7 @@ Naeb O.
Nagai K.
Nagashima Yoshikazu
Nagata H.|Tanaka T.
Nagra, B
Nagra B.S
Nagra B.|Kingdom N.
Nagy K
@@ -5640,6 +5641,7 @@ Pedersen Carol
Pedersen Carol|DuCharme Darlene
Pedersen C|Ducharme D.
Pederson C.A.
Pederson
Peffer
Peffer R
Peffer R.
@@ -6249,6 +6251,7 @@ Roth M.
Roth Markus
Roth RN|Heilman RD
Roth, M.
Roth
Rotroff
Rouaud, J. L.
Roussel C
@@ -6848,7 +6851,7 @@ Sokolowshi Andrea
Sokolowski A
Sokolowski A.
Sokolowski Andrea
Sokolowski,
Sokolowski
Sole
Sole C
Sole, C.
@@ -7248,6 +7251,7 @@ Thanei, P.
Tharp BE
Thede B.
Theophilidis G
Thevenaz
Thevenaz P
Thevenaz P.
Thevenaz Ph.
@@ -8193,4 +8197,113 @@ Zoriki Hosomi R.
Zoriki Hosomi Rosana
Zuberer D
Zubrod J
Zwicker R.E.
Zwicker R.E.
Ott
Argus et al
Mihail
Clapp
Murfitt R
Yang
Swarbrick
Hertl
Breitwieser
Seville, A
Trumper, C.
Gasso-Brown, D.
Váliczkó, É
Váliczkó, É.
Durando, J.
Tarcai, Z.
Pooles, A
Roig, J.
Schäfers, C.
Nixon, W.
Zok, S.
Durjava M.K. et al
Jatzek J
Salinas E
Sousa J. V
Caferella, M.A.
Salina, E.
Walraven J
Kaspers et al
Schneider et al
Lowrie C
Macdougall J
Roberts S
Madrid, S.O.
Maynard, M.S.
Ray, W.
Ray, W.J.
Váliczkó É
Tarcai Z
Orovecz B
Kaspers
Tarcai Z.
Gizler, A
Whitlow
Johnson et al
Bryden et al
Váliczkó
Höger
Dorgerloh
Schäfers C
Caferella M.A
Salinas, E
Zok S
Tummon OJ
Eckstein CL
Francis PD
Ott M
Richeux F
Rajsekhar PV
Webbley
O’Hagan
MacDonald
Jewkes
Harris
Tomlinson
Tomlinson et al
Nagy
Török-Bathó
Hargitai
Shearer
Robertson
Strepka
Hackford
Blunt
Sieber
Elcombe
Haines
Kang
Sayers
Harris J
Baxter A
Macdonald M
PetusArpasy M
TorokBatho M.
van Ravenzwaa y B
Cords S
Dymarkows ka K.
Bahig M
Pekari K
Bercz J
Russell L
Kitchen K
Miyagawa M.
Sasaki Y
Yin D
Huff J
Blackburn K
Carlson G
Pereira M
Stoner G
Zelenak V
Valiczko E
Tomlinso n J
Trumper C
Senciuc M
Vance C.
Przeorska J.
Melville S.
Wicksted G.
@@ -0,0 +1,7 @@
Monthey Syngenta Crop Protection AG, Basel, Switzerland
Syngenta Crop Protection, Monthey, Switzerland
Fine Organics Limited, Middlesbrough, United Kingdom
Syngenta Monthey Switzerland
Hunan Haili Chemical Industry Co., Ltd., Hunan, China
Syngenta, Switzerland
Syngenta Nantong, China
@@ -7,16 +7,16 @@ global Section section
rule "1: Redacted because Section contains Vertebrate"
when
eval(section.contains("vertebrate")==true);
Section(matchesType("vertebrate"))
then
section.redact("name", 1, "Redacted because Section contains Vertebrate");
section.redact("address", 1, "Redacted because Section contains Vertebrate");
section.redact("name", 1, "Redacted because Section contains Vertebrate", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("address", 1, "Redacted because Section contains Vertebrate", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "2: Not Redacted because Section contains no Vertebrate"
when
eval(section.contains("vertebrate")==false);
Section(!matchesType("vertebrate"))
then
section.redactNot("name", 2, "Not Redacted because Section contains no Vertebrate");
section.redactNot("address", 2, "Not Redacted because Section contains no Vertebrate");
@@ -25,7 +25,7 @@ rule "2: Not Redacted because Section contains no Vertebrate"
rule "3: Do not redact Names and Addresses if no redaction Indicator is contained"
when
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true);
Section(matchesType("vertebrate"), matchesType("no_redaction_indicator"))
then
section.redactNot("name", 3, "Vertebrate was found, but also a no redaction indicator");
section.redactNot("address", 3, "Vertebrate was found, but also a no redaction indicator");
@@ -34,88 +34,102 @@ rule "3: Do not redact Names and Addresses if no redaction Indicator is containe
rule "4: Redact Names and Addresses if no_redaction_indicator and redaction_indicator is contained"
when
eval(section.contains("vertebrate")==true && section.contains("no_redaction_indicator")==true && section.contains("redaction_indicator")==true);
Section(matchesType("vertebrate"), matchesType("no_redaction_indicator"), matchesType("redaction_indicator"))
then
section.redact("name", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
section.redact("address", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator");
section.redact("name", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("address", 4, "Vertebrate was found and no_redaction_indicator and redaction_indicator", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "5: Do not redact in guideline sections"
when
eval(section.headlineContainsWord("guideline") || section.headlineContainsWord("Guidance"));
Section(headlineContainsWord("guideline") || headlineContainsWord("Guidance"))
then
section.redactNot("name", 5, "Section is a guideline section.");
section.redactNot("address", 5, "Section is a guideline section.");
end
rule "6: Redact contact information, if applicant is found"
rule "6: Redact contact information if applicant is found"
when
eval(section.headlineContainsWord("applicant") || section.getText().contains("Applicant"));
Section(headlineContainsWord("applicant") || text.contains("Applicant") || headlineContainsWord("Primary contact") || headlineContainsWord("Alternative contact"))
then
section.redactLineAfter("Name:", "address", 6, "Applicant information was found");
section.redactBetween("Address:", "Contact", "address", 6, "Applicant information was found");
section.redactLineAfter("Contact point:", "address", 6, "Applicant information was found");
section.redactLineAfter("Phone:", "address", 6, "Applicant information was found");
section.redactLineAfter("Fax:", "address", 6, "Applicant information was found");
section.redactLineAfter("Tel.:", "address", 6, "Applicant information was found");
section.redactLineAfter("Tel:", "address", 6, "Applicant information was found");
section.redactLineAfter("E-mail:", "address", 6, "Applicant information was found");
section.redactLineAfter("Email:", "address", 6, "Applicant information was found");
section.redactLineAfter("Contact:", "address", 6, "Applicant information was found");
section.redactLineAfter("Telephone number:", "address", 6, "Applicant information was found");
section.redactLineAfter("Fax number:", "address", 6, "Applicant information was found");
section.redactLineAfter("Telephone:", "address", 6, "Applicant information was found");
section.redactBetween("No:", "Fax", "address", 6, "Applicant information was found");
section.redactBetween("Contact:", "Tel.:", "address", 6, "Applicant information was found");
section.redactLineAfter("Name:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Address:", "Contact", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Contact point:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Phone:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Fax:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Tel.:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Tel:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("E-mail:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Email:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("e-mail:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("E-mail address:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Contact:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Alternative contact:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Telephone number:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Telephone No:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Fax number:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Telephone:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Company:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("No:", "Fax", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Contact:", "Tel.:", "address", 6, "Applicant information was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "7: Redact contact information, if Producer is found"
rule "7: Redact contact information if Producer is found"
when
eval(section.getText().toLowerCase().contains("producer of the plant protection") || section.getText().toLowerCase().contains("producer of the active substance") || section.getText().contains("Manufacturer of the active substance") || section.getText().contains("Manufacturer:") || section.getText().contains("Producer or producers of the active substance"));
Section(text.toLowerCase().contains("producer of the plant protection") || text.toLowerCase().contains("producer of the active substance") || text.contains("Manufacturer of the active substance") || text.contains("Manufacturer:") || text.contains("Producer or producers of the active substance"))
then
section.redactLineAfter("Name:", "address", 7, "Producer was found");
section.redactBetween("Address:", "Contact", "address", 7, "Producer was found");
section.redactBetween("Contact:", "Phone", "address", 7, "Producer was found");
section.redactBetween("Contact:", "Telephone number:", "address", 7, "Producer was found");
section.redactBetween("Address:", "Manufacturing", "address", 7, "Producer was found");
section.redactLineAfter("Telephone:", "address", 7, "Producer was found");
section.redactLineAfter("Phone:", "address", 7, "Producer was found");
section.redactLineAfter("Fax:", "address", 7, "Producer was found");
section.redactLineAfter("E-mail:", "address", 7, "Producer was found");
section.redactLineAfter("Contact:", "address", 7, "Producer was found");
section.redactLineAfter("Fax number:", "address", 7, "Producer was found");
section.redactLineAfter("Telephone number:", "address", 7, "Producer was found");
section.redactLineAfter("Tel:", "address", 7, "Producer was found");
section.redactBetween("No:", "Fax", "address", 7, "Producer was found");
section.redactLineAfter("Name:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Address:", "Contact", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Contact:", "Phone", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Contact:", "Telephone number:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("Address:", "Manufacturing", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Telephone:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Phone:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Fax:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("E-mail:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Contact:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Fax number:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Telephone number:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactLineAfter("Tel:", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redactBetween("No:", "Fax", "address", 7, "Producer was found", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "8: Not redacted because Vertebrate Study = N"
when
Section(isNotVertebrateStudy())
Section(rowEquals("Vertebrate study Y/N", "N") || rowEquals("Vertebrate study Y/N", "No"))
then
section.redactNot("name", 8, "Not redacted because row is not a vertebrate study");
section.redactNotCell("Author(s)", 8, "name", "Not redacted because row is not a vertebrate study");
section.redactNot("address", 8, "Not redacted because row is not a vertebrate study");
section.highlightCell("Vertebrate study Y/N", 8, "hint_only");
section.highlightCell("Verte brate study Y/N", 8, "hint_only");
end
rule "9: Redact if must redact entry is found"
when
eval(section.contains("must_redact")==true);
Section(matchesType("must_redact"))
then
section.redact("name", 9, "must_redact entry was found.");
section.redact("address", 9, "must_redact entry was found.");
section.redact("name", 9, "must_redact entry was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("address", 9, "must_redact entry was found.", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
rule "10: Redact Authors and Addresses in Reference Table, if it is a Vertebrate study"
rule "10: Redact Authors and Addresses in Reference Table if it is a Vertebrate study"
when
Section(isVertebrateStudy())
Section(rowEquals("Vertebrate study Y/N", "Y") || rowEquals("Vertebrate study Y/N", "Yes"))
then
section.redact("name", 10, "Redacted because row is a vertebrate study");
section.redact("address", 10, "Redacted because row is a vertebrate study");
section.redactCell("Author(s)", 10, "name", "Redacted because row is a vertebrate study", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.redact("address", 10, "Redacted because row is a vertebrate study", "Reg (EC) No 1107/2009 Art. 63 (2g)");
section.highlightCell("Vertebrate study Y/N", 10, "must_redact");
section.highlightCell("Verte brate study Y/N", 10, "must_redact");
end
rule "11: Redact sponsor company"
when
Section(searchText.toLowerCase().contains("batches produced at"))
then
section.redactIfPrecededBy("batches produced at", "sponsor", 11, "Redacted because it represents a sponsor company", "Reg (EC) No 1107/2009 Art. 63 (2g)");
end
File diff suppressed because one or more lines are too long