Compare commits
312
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d4e728350d | ||
|
|
b410067b8c | ||
|
|
73d3ed625a | ||
|
|
e9176db88f | ||
|
|
ef72cc6861 | ||
|
|
66b2d52d40 | ||
|
|
dd1b838a5c | ||
|
|
1fca62f578 | ||
|
|
0e925f2f24 | ||
|
|
e23432096c | ||
|
|
c1e2b8da29 | ||
|
|
16b04b5918 | ||
|
|
c16b6d41d5 | ||
|
|
b839c4e3ae | ||
|
|
6fd6caa8ad | ||
|
|
41282c0edd | ||
|
|
3c79e65345 | ||
|
|
d2eeaa91a6 | ||
|
|
53a375b832 | ||
|
|
d233c18d33 | ||
|
|
b74673ae63 | ||
|
|
faa702d3f4 | ||
|
|
05bf95d62b | ||
|
|
776de8392a | ||
|
|
660abb318f | ||
|
|
8b74142d31 | ||
|
|
467a242a3d | ||
|
|
87b11842a6 | ||
|
|
8f56d50322 | ||
|
|
38ce801f2d | ||
|
|
91e227248d | ||
|
|
623b8df5e6 | ||
|
|
20ab65afd2 | ||
|
|
1ce47b7fbc | ||
|
|
5feb6891e2 | ||
|
|
02bdbbc2d1 | ||
|
|
19e607e8a8 | ||
|
|
5c38150d34 | ||
|
|
18487c639b | ||
|
|
e2234dc52a | ||
|
|
503df78f88 | ||
|
|
5527eaec4e | ||
|
|
b4f079c3c2 | ||
|
|
1e6e5e2154 | ||
|
|
17bdcf8d24 | ||
|
|
aa43453206 | ||
|
|
ddbf80e4a6 | ||
|
|
2e3d4ad361 | ||
|
|
b8320bd000 | ||
|
|
76fda2b573 | ||
|
|
69540bcd5e | ||
|
|
8ab2738cd0 | ||
|
|
8d88b19915 | ||
|
|
9d88925ff1 | ||
|
|
f6bc49d42c | ||
|
|
e0dd06c6bf | ||
|
|
97209a3508 | ||
|
|
074205aa4d | ||
|
|
2f88d8083c | ||
|
|
a510a8bb9f | ||
|
|
c027924b19 | ||
|
|
313cf68044 | ||
|
|
3c3666cd06 | ||
|
|
5aba8fe881 | ||
|
|
ec60d0eab4 | ||
|
|
0e71613ffc | ||
|
|
142f5256ae | ||
|
|
ceee64ce59 | ||
|
|
d2f2cd975c | ||
|
|
5503dbfffc | ||
|
|
369710dbd4 | ||
|
|
b9b3e8c8f5 | ||
|
|
c478935b86 | ||
|
|
d92757cda4 | ||
|
|
5828e19422 | ||
|
|
b7bf84c323 | ||
|
|
fe7e8a83df | ||
|
|
a71a7818c2 | ||
|
|
ee5da81299 | ||
|
|
e486711755 | ||
|
|
58c3d15b78 | ||
|
|
ce15faf072 | ||
|
|
430a08e611 | ||
|
|
4cc40d8381 | ||
|
|
317a8a9af9 | ||
|
|
28e437a037 | ||
|
|
a5f27cfa4c | ||
|
|
34be42cd45 | ||
|
|
f1b5d605cc | ||
|
|
c3dadd6906 | ||
|
|
c29c5eef0d | ||
|
|
9867cd6848 | ||
|
|
64bd25a900 | ||
|
|
9b9b0ab271 | ||
|
|
da21d7da4b | ||
|
|
b6471f904c | ||
|
|
29451db72d | ||
|
|
87b76cdae0 | ||
|
|
85cad66ade | ||
|
|
1cc93d3a57 | ||
|
|
a1ef711bc3 | ||
|
|
b6a95244d8 | ||
|
|
4c06ab958c | ||
|
|
f5fb8f7c07 | ||
|
|
e962992b79 | ||
|
|
a9ae01ab32 | ||
|
|
e77180f6ac | ||
|
|
269716125c | ||
|
|
dc21e14921 | ||
|
|
6bc5a6a135 | ||
|
|
4f66f5acf7 | ||
|
|
a9ff46a3fd | ||
|
|
07aaa9722a | ||
|
|
cba81ce061 | ||
|
|
f84a366328 | ||
|
|
85c44374d9 | ||
|
|
1d39c150c7 | ||
|
|
b2f1201d92 | ||
|
|
b592fee500 | ||
|
|
4f36b8b43e | ||
|
|
c7a789ada6 | ||
|
|
6de3a3b043 | ||
|
|
43ff331a42 | ||
|
|
3013dc95f6 | ||
|
|
435f75996f | ||
|
|
ef04b7168a | ||
|
|
6e46cad2c5 | ||
|
|
c41e230f85 | ||
|
|
cadbc3c7a4 | ||
|
|
f4655c4050 | ||
|
|
16ea8364df | ||
|
|
3ec1e20e41 | ||
|
|
ce1f8e9117 | ||
|
|
b7e97890be | ||
|
|
ec49945682 | ||
|
|
b64222938c | ||
|
|
e4823e20d4 | ||
|
|
53428ec652 | ||
|
|
ca270fb7de | ||
|
|
06fe0dbd97 | ||
|
|
47de526b4b | ||
|
|
1d8e86e4f6 | ||
|
|
2b7972315f | ||
|
|
0af853dbfc | ||
|
|
0544a42117 | ||
|
|
55bebe8541 | ||
|
|
8bdfd6747d | ||
|
|
80c0c25c06 | ||
|
|
1cc66a3092 | ||
|
|
17baf9c0eb | ||
|
|
090a619600 | ||
|
|
9c8e2f047e | ||
|
|
24612e3471 | ||
|
|
9be8d46f4e | ||
|
|
3e32540616 | ||
|
|
5403829fd8 | ||
|
|
3e53e156a5 | ||
|
|
4c73d2841b | ||
|
|
db8eedc9a3 | ||
|
|
58e16fe39a | ||
|
|
0fe2b627f0 | ||
|
|
e4b5bb93e1 | ||
|
|
99a8b9788e | ||
|
|
822e1f096f | ||
|
|
8c5ede3fde | ||
|
|
40380a12a3 | ||
|
|
f48bdb6267 | ||
|
|
d3a5265b7f | ||
|
|
708b7f992c | ||
|
|
2b0a667424 | ||
|
|
c330de8624 | ||
|
|
32e73dc9d9 | ||
|
|
5e61f3370d | ||
|
|
bfadac7a3f | ||
|
|
df80ff68cd | ||
|
|
13352e7d16 | ||
|
|
3c9d59f73b | ||
|
|
333793021a | ||
|
|
264d7e3a87 | ||
|
|
6b89572c87 | ||
|
|
a74d049cf8 | ||
|
|
76dc08d0e6 | ||
|
|
663c361707 | ||
|
|
c81b062784 | ||
|
|
57ef7ead72 | ||
|
|
3269470446 | ||
|
|
c7e10cefba | ||
|
|
2301849815 | ||
|
|
2d5ba8f447 | ||
|
|
2a6ee02f9f | ||
|
|
7dad091e19 | ||
|
|
f7b584617b | ||
|
|
02a264d586 | ||
|
|
dacd7b725d | ||
|
|
7ee338a067 | ||
|
|
19d48e207b | ||
|
|
e22b625e9c | ||
|
|
95a2d71af0 | ||
|
|
ec8d3183dc | ||
|
|
6fec4cc209 | ||
|
|
fa3cbc938c | ||
|
|
f6b4d2a067 | ||
|
|
0967e8e64a | ||
|
|
20c7e19343 | ||
|
|
6fb7600de2 | ||
|
|
fbd55c15ef | ||
|
|
ffa251643a | ||
|
|
ac19defa1c | ||
|
|
21ca25f330 | ||
|
|
e9b9fbb91b | ||
|
|
3f76401c7d | ||
|
|
a16f9c6d6e | ||
|
|
7ca3416d1d | ||
|
|
85db826f89 | ||
|
|
6fac40fd8d | ||
|
|
23fe93630b | ||
|
|
778c975649 | ||
|
|
50b2908763 | ||
|
|
c1192ceefd | ||
|
|
77584c9a5a | ||
|
|
21d717f083 | ||
|
|
c85ce25ed4 | ||
|
|
f608d991a3 | ||
|
|
e914b09d1e | ||
|
|
87d64b6d12 | ||
|
|
eb71373152 | ||
|
|
a65e23e45d | ||
|
|
4aaef260d7 | ||
|
|
c669f9a09a | ||
|
|
b1bad03090 | ||
|
|
0a0d9fe31a | ||
|
|
fd37e3deab | ||
|
|
7fb04cfa48 | ||
|
|
0f5f7974cc | ||
|
|
0601ec4862 | ||
|
|
d2807e0ff1 | ||
|
|
e49bbfca88 | ||
|
|
7967453c67 | ||
|
|
a3d002929d | ||
|
|
8a44d6299c | ||
|
|
78de7371a0 | ||
|
|
924f791e4d | ||
|
|
542bbb294a | ||
|
|
df2fcd2e5e | ||
|
|
8e819a0d26 | ||
|
|
01e7115d30 | ||
|
|
747323f882 | ||
|
|
729a7334d9 | ||
|
|
36967cab2c | ||
|
|
b820463fff | ||
|
|
cfd9d7665b | ||
|
|
23d84980cd | ||
|
|
314f7ebdd9 | ||
|
|
f666a03301 | ||
|
|
313ec32d30 | ||
|
|
76dca41299 | ||
|
|
f21e48e3d7 | ||
|
|
bbd60e4bf4 | ||
|
|
ecf794ff37 | ||
|
|
5d8f44d55d | ||
|
|
513fbbb38d | ||
|
|
6178efb7ef | ||
|
|
7a9bb45b6d | ||
|
|
00f2a0cb82 | ||
|
|
1eb0485dca | ||
|
|
be15d38065 | ||
|
|
f154065824 | ||
|
|
d4b0de9184 | ||
|
|
fbd9bc71a8 | ||
|
|
c963c7ac32 | ||
|
|
404fd25be0 | ||
|
|
efd53066c4 | ||
|
|
5b5d898b93 | ||
|
|
cf75ac56cc | ||
|
|
1a6a1a008f | ||
|
|
e19b4573bb | ||
|
|
150ee92c1e | ||
|
|
7988a8de66 | ||
|
|
0014d7725a | ||
|
|
55f608cf99 | ||
|
|
f3ad0749b8 | ||
|
|
0f0aee9480 | ||
|
|
ff22dd1a49 | ||
|
|
52e5c58d6e | ||
|
|
941448492b | ||
|
|
49d47a0682 | ||
|
|
9bab1205b7 | ||
|
|
7963c57552 | ||
|
|
1427087d8f | ||
|
|
b83400ffa3 | ||
|
|
f05f2b21a4 | ||
|
|
02d9ebc24c | ||
|
|
ae3e4570c1 | ||
|
|
a975596446 | ||
|
|
a5b96f1489 | ||
|
|
4814ce3650 | ||
|
|
7fccdb32c5 | ||
|
|
1d389b78fd | ||
|
|
b4916b65b6 | ||
|
|
e1db383463 | ||
|
|
4d2a7d454d | ||
|
|
3758720683 | ||
|
|
22bb0b44db | ||
|
|
e93f5d147f | ||
|
|
88e6ac3d22 | ||
|
|
17b885d901 | ||
|
|
50d9f4fec2 | ||
|
|
1bcf0a3f07 | ||
|
|
1f6a9aa14d | ||
|
|
c8cf0078eb | ||
|
|
b13a695c64 | ||
|
|
32a907f697 |
@@ -1,4 +1,5 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to this project will be documented in this file.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
+28
-28
@@ -1,37 +1,37 @@
|
||||
<project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
|
||||
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/maven-v4_0_0.xsd">
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<parent>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs-parent</artifactId>
|
||||
<version>8.0.3</version>
|
||||
<relativePath/>
|
||||
</parent>
|
||||
<parent>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs-parent</artifactId>
|
||||
<version>8.1.3</version>
|
||||
<relativePath/>
|
||||
</parent>
|
||||
|
||||
<artifactId>bamboo-specs</artifactId>
|
||||
<version>1.0.0-SNAPSHOT</version>
|
||||
<packaging>jar</packaging>
|
||||
<artifactId>bamboo-specs</artifactId>
|
||||
<version>1.0.0-SNAPSHOT</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<dependencies>
|
||||
<dependency>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs-api</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs</artifactId>
|
||||
</dependency>
|
||||
<dependencies>
|
||||
<dependency>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs-api</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.atlassian.bamboo</groupId>
|
||||
<artifactId>bamboo-specs</artifactId>
|
||||
</dependency>
|
||||
|
||||
<!-- Test dependencies -->
|
||||
<dependency>
|
||||
<groupId>junit</groupId>
|
||||
<artifactId>junit</artifactId>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
<!-- Test dependencies -->
|
||||
<dependency>
|
||||
<groupId>junit</groupId>
|
||||
<artifactId>junit</artifactId>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
|
||||
<!-- run 'mvn test' to perform offline validation of the plan -->
|
||||
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
|
||||
<!-- run 'mvn test' to perform offline validation of the plan -->
|
||||
<!-- run 'mvn -Ppublish-specs' to upload the plan to your Bamboo server -->
|
||||
</project>
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
package buildjob;
|
||||
|
||||
import java.time.DayOfWeek;
|
||||
import static com.atlassian.bamboo.specs.builders.task.TestParserTask.createJUnitParserTask;
|
||||
|
||||
import java.time.LocalTime;
|
||||
|
||||
import com.atlassian.bamboo.specs.api.BambooSpec;
|
||||
import com.atlassian.bamboo.specs.api.builders.BambooKey;
|
||||
import com.atlassian.bamboo.specs.api.builders.Variable;
|
||||
import com.atlassian.bamboo.specs.api.builders.docker.DockerConfiguration;
|
||||
import com.atlassian.bamboo.specs.api.builders.permission.PermissionType;
|
||||
import com.atlassian.bamboo.specs.api.builders.permission.Permissions;
|
||||
@@ -17,19 +19,16 @@ import com.atlassian.bamboo.specs.api.builders.plan.branches.BranchCleanup;
|
||||
import com.atlassian.bamboo.specs.api.builders.plan.branches.PlanBranchManagement;
|
||||
import com.atlassian.bamboo.specs.api.builders.project.Project;
|
||||
import com.atlassian.bamboo.specs.builders.task.CheckoutItem;
|
||||
import com.atlassian.bamboo.specs.builders.task.CleanWorkingDirectoryTask;
|
||||
import com.atlassian.bamboo.specs.builders.task.InjectVariablesTask;
|
||||
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
|
||||
import com.atlassian.bamboo.specs.builders.task.VcsCheckoutTask;
|
||||
import com.atlassian.bamboo.specs.builders.task.VcsTagTask;
|
||||
import com.atlassian.bamboo.specs.builders.trigger.BitbucketServerTrigger;
|
||||
import com.atlassian.bamboo.specs.builders.trigger.RepositoryPollingTrigger;
|
||||
import com.atlassian.bamboo.specs.builders.trigger.ScheduledTrigger;
|
||||
import com.atlassian.bamboo.specs.model.task.InjectVariablesScope;
|
||||
import com.atlassian.bamboo.specs.util.BambooServer;
|
||||
import com.atlassian.bamboo.specs.builders.task.ScriptTask;
|
||||
import com.atlassian.bamboo.specs.model.task.ScriptTaskProperties.Location;
|
||||
|
||||
import static com.atlassian.bamboo.specs.builders.task.TestParserTask.createJUnitParserTask;
|
||||
import com.atlassian.bamboo.specs.util.BambooServer;
|
||||
|
||||
/**
|
||||
* Plan configuration for Bamboo.
|
||||
@@ -40,14 +39,15 @@ public class PlanSpec {
|
||||
|
||||
private static final String SERVICE_NAME = "redaction-service";
|
||||
|
||||
private static final String JVM_ARGS =" -Xmx4g -XX:+ExitOnOutOfMemoryError -XX:SurvivorRatio=2 -XX:NewRatio=1 -XX:InitialTenuringThreshold=16 -XX:MaxTenuringThreshold=16 -XX:InitiatingHeapOccupancyPercent=35 ";
|
||||
private static final String JVM_ARGS = " -Xmx4g -XX:+ExitOnOutOfMemoryError -XX:SurvivorRatio=2 -XX:NewRatio=1 -XX:InitialTenuringThreshold=16 -XX:MaxTenuringThreshold=16 -XX:InitiatingHeapOccupancyPercent=35 ";
|
||||
|
||||
private static final String SERVICE_KEY = SERVICE_NAME.toUpperCase().replaceAll("-", "");
|
||||
|
||||
|
||||
/**
|
||||
* Run main to publish plan on Bamboo
|
||||
*/
|
||||
public static void main(final String[] args) throws Exception {
|
||||
public static void main(final String[] args) {
|
||||
//By default credentials are read from the '.credentials' file.
|
||||
BambooServer bambooServer = new BambooServer("http://localhost:8085");
|
||||
|
||||
@@ -56,15 +56,26 @@ public class PlanSpec {
|
||||
PlanPermissions planPermission = new PlanSpec().createPlanPermission(plan.getIdentifier());
|
||||
bambooServer.publish(planPermission);
|
||||
|
||||
Plan nightPlan = new PlanSpec().createNightPlan();
|
||||
bambooServer.publish(nightPlan);
|
||||
PlanPermissions nightPlanPermission = new PlanSpec().createPlanPermission(nightPlan.getIdentifier());
|
||||
bambooServer.publish(nightPlanPermission);
|
||||
|
||||
Plan secPlan = new PlanSpec().createSecBuild();
|
||||
bambooServer.publish(secPlan);
|
||||
PlanPermissions secPlanPermission = new PlanSpec().createPlanPermission(secPlan.getIdentifier());
|
||||
bambooServer.publish(secPlanPermission);
|
||||
}
|
||||
|
||||
|
||||
private PlanPermissions createPlanPermission(PlanIdentifier planIdentifier) {
|
||||
Permissions permission = new Permissions()
|
||||
.userPermissions("atlbamboo", PermissionType.EDIT, PermissionType.VIEW, PermissionType.ADMIN, PermissionType.CLONE, PermissionType.BUILD)
|
||||
|
||||
Permissions permission = new Permissions().userPermissions("atlbamboo",
|
||||
PermissionType.EDIT,
|
||||
PermissionType.VIEW,
|
||||
PermissionType.ADMIN,
|
||||
PermissionType.CLONE,
|
||||
PermissionType.BUILD)
|
||||
.groupPermissions("development", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
|
||||
.groupPermissions("devplant", PermissionType.EDIT, PermissionType.VIEW, PermissionType.CLONE, PermissionType.BUILD)
|
||||
.loggedInUserPermissions(PermissionType.VIEW)
|
||||
@@ -72,103 +83,74 @@ public class PlanSpec {
|
||||
return new PlanPermissions(planIdentifier.getProjectKey(), planIdentifier.getPlanKey()).permissions(permission);
|
||||
}
|
||||
|
||||
|
||||
private Project project() {
|
||||
return new Project()
|
||||
.name("RED")
|
||||
.key(new BambooKey("RED"));
|
||||
|
||||
return new Project().name("RED").key(new BambooKey("RED"));
|
||||
}
|
||||
|
||||
|
||||
public Plan createPlan() {
|
||||
return new Plan(
|
||||
project(),
|
||||
SERVICE_NAME, new BambooKey(SERVICE_KEY))
|
||||
.description("Plan created from (enter repository url of your plan)")
|
||||
.stages(new Stage("Default Stage")
|
||||
.jobs(new Job("Default Job",
|
||||
new BambooKey("JOB1"))
|
||||
.tasks(
|
||||
new ScriptTask()
|
||||
.description("Clean")
|
||||
.inlineBody("#!/bin/bash\n" +
|
||||
"set -e\n" +
|
||||
"rm -rf ./*"),
|
||||
new VcsCheckoutTask()
|
||||
.description("Checkout Default Repository")
|
||||
.checkoutItems(new CheckoutItem().defaultRepository()),
|
||||
new ScriptTask()
|
||||
.description("Build")
|
||||
.location(Location.FILE)
|
||||
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
|
||||
.argument(SERVICE_NAME),
|
||||
createJUnitParserTask()
|
||||
.description("Resultparser")
|
||||
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
|
||||
.enabled(true),
|
||||
new InjectVariablesTask()
|
||||
.description("Inject git Tag")
|
||||
.path("git.tag")
|
||||
.namespace("g")
|
||||
.scope(InjectVariablesScope.LOCAL),
|
||||
new VcsTagTask()
|
||||
.description("${bamboo.g.gitTag}")
|
||||
.tagName("${bamboo.g.gitTag}")
|
||||
.defaultRepository())
|
||||
.dockerConfiguration(
|
||||
new DockerConfiguration()
|
||||
.image("nexus.iqser.com:5001/infra/maven:3.6.2-jdk-13-3.0.0")
|
||||
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
|
||||
.volume("/var/run/docker.sock", "/var/run/docker.sock")
|
||||
)
|
||||
)
|
||||
)
|
||||
.linkedRepositories("RED / " + SERVICE_NAME)
|
||||
|
||||
return new Plan(project(), SERVICE_NAME, new BambooKey(SERVICE_KEY)).description("Plan created from (enter repository url of your plan)")
|
||||
.variables(new Variable("maven_add_param", ""))
|
||||
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new CleanWorkingDirectoryTask().description("Clean working directory.")
|
||||
.enabled(true),
|
||||
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
|
||||
new ScriptTask().description("Build").location(Location.FILE).fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh").argument(SERVICE_NAME),
|
||||
createJUnitParserTask().description("Resultparser")
|
||||
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
|
||||
.enabled(true),
|
||||
new InjectVariablesTask().description("Inject git Tag").path("git.tag").namespace("g").scope(InjectVariablesScope.LOCAL),
|
||||
new VcsTagTask().description("${bamboo.g.gitTag}").tagName("${bamboo.g.gitTag}").defaultRepository())
|
||||
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
|
||||
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
|
||||
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
|
||||
.linkedRepositories("RED / " + SERVICE_NAME)
|
||||
.triggers(new BitbucketServerTrigger())
|
||||
.planBranchManagement(new PlanBranchManagement()
|
||||
.createForVcsBranch()
|
||||
.delete(new BranchCleanup()
|
||||
.whenInactiveInRepositoryAfterDays(14))
|
||||
.planBranchManagement(new PlanBranchManagement().createForVcsBranch()
|
||||
.delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14))
|
||||
.notificationForCommitters());
|
||||
}
|
||||
|
||||
|
||||
public Plan createNightPlan() {
|
||||
|
||||
return new Plan(project(), SERVICE_NAME + "-Night", new BambooKey(SERVICE_KEY + "NIGHT")).description("Long running nightly Plan for tests")
|
||||
.variables(new Variable("maven_add_param", "-Dtest-groups=rules-test"))
|
||||
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new CleanWorkingDirectoryTask().description("Clean working directory.")
|
||||
.enabled(true),
|
||||
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
|
||||
new ScriptTask().description("Build")
|
||||
.location(Location.FILE)
|
||||
.fileFromPath("bamboo-specs/src/main/resources/scripts/build-java.sh")
|
||||
.argument(SERVICE_NAME + " verify"),
|
||||
createJUnitParserTask().description("Resultparser")
|
||||
.resultDirectories("**/test-reports/*.xml, **/target/surefire-reports/*.xml, **/target/failsafe-reports/*.xml")
|
||||
.enabled(true))
|
||||
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
|
||||
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
|
||||
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
|
||||
.linkedRepositories("RED / " + SERVICE_NAME)
|
||||
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
|
||||
.planBranchManagement(new PlanBranchManagement().delete(new BranchCleanup().whenInactiveInRepositoryAfterDays(14)).notificationForCommitters());
|
||||
}
|
||||
|
||||
|
||||
public Plan createSecBuild() {
|
||||
return new Plan(
|
||||
project(),
|
||||
SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC"))
|
||||
.description("Security Analysis Plan")
|
||||
.stages(new Stage("Default Stage")
|
||||
.jobs(new Job("Default Job",
|
||||
new BambooKey("JOB1"))
|
||||
.tasks(
|
||||
new ScriptTask()
|
||||
.description("Clean")
|
||||
.inlineBody("#!/bin/bash\n" +
|
||||
"set -e\n" +
|
||||
"rm -rf ./*"),
|
||||
new VcsCheckoutTask()
|
||||
.description("Checkout Default Repository")
|
||||
.checkoutItems(new CheckoutItem().defaultRepository()),
|
||||
new ScriptTask()
|
||||
.description("Sonar")
|
||||
.location(Location.FILE)
|
||||
.fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh")
|
||||
.argument(SERVICE_NAME))
|
||||
.dockerConfiguration(
|
||||
new DockerConfiguration()
|
||||
.image("nexus.iqser.com:5001/infra/maven:3.6.2-jdk-13-3.0.0")
|
||||
.dockerRunArguments("--net=host")
|
||||
.volume("/etc/maven/settings.xml", "/usr/share/maven/ref/settings.xml")
|
||||
.volume("/var/run/docker.sock", "/var/run/docker.sock")
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
return new Plan(project(), SERVICE_NAME + "-Sec", new BambooKey(SERVICE_KEY + "SEC")).description("Security Analysis Plan")
|
||||
.stages(new Stage("Default Stage").jobs(new Job("Default Job", new BambooKey("JOB1")).tasks(new ScriptTask().description("Clean")
|
||||
.inlineBody("#!/bin/bash\n" + "set -e\n" + "rm -rf ./*"),
|
||||
new VcsCheckoutTask().description("Checkout Default Repository").cleanCheckout(true).checkoutItems(new CheckoutItem().defaultRepository()),
|
||||
new ScriptTask().description("Sonar").location(Location.FILE).fileFromPath("bamboo-specs/src/main/resources/scripts/sonar-java.sh").argument(SERVICE_NAME))
|
||||
.dockerConfiguration(new DockerConfiguration().image("nexus.iqser.com:5001/infra/maven:3.8.4-openjdk-17-slim")
|
||||
.dockerRunArguments("--net=host")
|
||||
.volume("/etc/maven/settings.xml", "/usr/share/maven/conf/settings.xml")
|
||||
.volume("/var/run/docker.sock", "/var/run/docker.sock"))))
|
||||
.linkedRepositories("RED / " + SERVICE_NAME)
|
||||
.triggers(
|
||||
new ScheduledTrigger()
|
||||
.scheduleOnceDaily(LocalTime.of(12, 00)),
|
||||
new BitbucketServerTrigger())
|
||||
.planBranchManagement(new PlanBranchManagement()
|
||||
.createForVcsBranchMatching("release.*")
|
||||
.notificationForCommitters());
|
||||
.triggers(new ScheduledTrigger().scheduleOnceDaily(LocalTime.of(23, 00)))
|
||||
.planBranchManagement(new PlanBranchManagement().createForVcsBranchMatching("release.*").notificationForCommitters());
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -2,50 +2,64 @@
|
||||
set -e
|
||||
|
||||
SERVICE_NAME=$1
|
||||
MVN_TARGET=${2:-deploy}
|
||||
|
||||
if [[ "${bamboo_version_tag}" = "dev" ]]
|
||||
then
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
if [[ "$bamboo_planRepository_branchName" == "master" ]]
|
||||
then
|
||||
branchVersion=$(cat pom.xml | grep -Eo " <version>.*-SNAPSHOT</version>" | sed -s 's|<version>\(.*\)\..*\(-*.*\)</version>|\1|' | tr -d ' ')
|
||||
latestVersion=$( semver $(git tag -l "${branchVersion}.*" ) | tail -n1 )
|
||||
newVersion="$(semver $latestVersion -p -i minor)"
|
||||
echo "new release on master with version $newVersion"
|
||||
elif [[ "$bamboo_planRepository_branchName" == release* ]]
|
||||
then
|
||||
branchVersion=$(echo $bamboo_planRepository_branchName | sed -s 's|release\/\([0-9]\+\.[0-9]\+\)\.x|\1|')
|
||||
latestVersion=$( semver $(git tag -l "${branchVersion}.*" ) | tail -n1 )
|
||||
newVersion="$(semver $latestVersion -p -i patch)"
|
||||
echo "new release on $bamboo_planRepository_branchName with version $newVersion"
|
||||
elif [[ "${bamboo_version_tag}" != "dev" ]]
|
||||
then
|
||||
newVersion="${bamboo_version_tag}"
|
||||
echo "new special version bild with $newVersion"
|
||||
else
|
||||
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
--no-transfer-progress \
|
||||
${bamboo_maven_add_param} \
|
||||
clean install \
|
||||
-Djava.security.egd=file:/dev/./urandomelse
|
||||
else
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
versions:set \
|
||||
-DnewVersion=${bamboo_version_tag}
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
|
||||
versions:set \
|
||||
-DnewVersion=${bamboo_version_tag}
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
--no-transfer-progress \
|
||||
clean deploy \
|
||||
-e \
|
||||
-DdeployAtEnd=true \
|
||||
-Dmaven.wagon.http.ssl.insecure=true \
|
||||
-Dmaven.wagon.http.ssl.allowall=true \
|
||||
-Dmaven.wagon.http.ssl.ignore.validity.dates=true \
|
||||
-DaltDeploymentRepository=iqser_release::default::https://nexus.iqser.com/repository/red-platform-releases
|
||||
echo "dev build with tag ${bamboo_planRepository_1_branch}_${bamboo_buildNumber}"
|
||||
echo "gitTag=${bamboo_planRepository_1_branch}_${bamboo_buildNumber}" > git.tag
|
||||
exit 0
|
||||
fi
|
||||
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
echo "gitTag=${newVersion}" > git.tag
|
||||
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
${bamboo_maven_add_param} \
|
||||
versions:set \
|
||||
-DnewVersion=${newVersion}
|
||||
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
|
||||
${bamboo_maven_add_param} \
|
||||
versions:set \
|
||||
-DnewVersion=${newVersion}
|
||||
|
||||
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
--no-transfer-progress \
|
||||
clean $MVN_TARGET \
|
||||
${bamboo_maven_add_param} \
|
||||
-e \
|
||||
-DdeployAtEnd=true \
|
||||
-Dmaven.wagon.http.ssl.insecure=true \
|
||||
-Dmaven.wagon.http.ssl.allowall=true \
|
||||
-Dmaven.wagon.http.ssl.ignore.validity.dates=true \
|
||||
-DaltDeploymentRepository=iqser_release::default::https://nexus.iqser.com/repository/red-platform-releases
|
||||
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
|
||||
package
|
||||
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-image-v1/pom.xml \
|
||||
docker:push
|
||||
|
||||
if [[ "${bamboo_version_tag}" = "dev" ]]
|
||||
then
|
||||
echo "gitTag=${bamboo_planRepository_1_branch}_${bamboo_buildNumber}" > git.tag
|
||||
else
|
||||
echo "gitTag=${bamboo_version_tag}" > git.tag
|
||||
fi
|
||||
|
||||
@@ -3,46 +3,42 @@ set -e
|
||||
|
||||
SERVICE_NAME=$1
|
||||
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
echo "build jar binaries"
|
||||
mvn -f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
--no-transfer-progress \
|
||||
clean install \
|
||||
-Djava.security.egd=file:/dev/./urandomelse
|
||||
|
||||
echo "dependency-check:aggregate"
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
org.owasp:dependency-check-maven:aggregate
|
||||
|
||||
if [[ -z "${bamboo_repository_pr_key}" ]]
|
||||
then
|
||||
echo "Sonar Scan for branch: ${bamboo_planRepository_1_branch}"
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
sonar:sonar \
|
||||
-Dsonar.projectKey=RED_$SERVICE_NAME \
|
||||
-Dsonar.host.url=https://sonarqube.iqser.com \
|
||||
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
|
||||
-Dsonar.branch.name=${bamboo_planRepository_1_branch} \
|
||||
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
|
||||
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
|
||||
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
|
||||
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
sonar:sonar \
|
||||
-Dsonar.projectKey=RED_$SERVICE_NAME \
|
||||
-Dsonar.host.url=https://sonarqube.iqser.com \
|
||||
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
|
||||
-Dsonar.branch.name=${bamboo_planRepository_1_branch} \
|
||||
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
|
||||
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
|
||||
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
|
||||
else
|
||||
echo "Sonar Scan for PR with key1: ${bamboo_repository_pr_key}"
|
||||
${bamboo_capability_system_builder_mvn3_Maven_3}/bin/mvn \
|
||||
--no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
sonar:sonar \
|
||||
-Dsonar.projectKey=RED_$SERVICE_NAME \
|
||||
-Dsonar.host.url=https://sonarqube.iqser.com \
|
||||
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
|
||||
-Dsonar.pullrequest.key=${bamboo_repository_pr_key} \
|
||||
-Dsonar.pullrequest.branch=${bamboo_repository_pr_sourceBranch} \
|
||||
-Dsonar.pullrequest.base=${bamboo_repository_pr_targetBranch} \
|
||||
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
|
||||
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
|
||||
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
|
||||
fi
|
||||
mvn --no-transfer-progress \
|
||||
-f ${bamboo_build_working_directory}/$SERVICE_NAME-v1/pom.xml \
|
||||
sonar:sonar \
|
||||
-Dsonar.projectKey=RED_$SERVICE_NAME \
|
||||
-Dsonar.host.url=https://sonarqube.iqser.com \
|
||||
-Dsonar.login=${bamboo_sonarqube_api_token_secret} \
|
||||
-Dsonar.pullrequest.key=${bamboo_repository_pr_key} \
|
||||
-Dsonar.pullrequest.branch=${bamboo_repository_pr_sourceBranch} \
|
||||
-Dsonar.pullrequest.base=${bamboo_repository_pr_targetBranch} \
|
||||
-Dsonar.dependencyCheck.jsonReportPath=target/dependency-check-report.json \
|
||||
-Dsonar.dependencyCheck.xmlReportPath=target/dependency-check-report.xml \
|
||||
-Dsonar.dependencyCheck.htmlReportPath=target/dependency-check-report.html
|
||||
fi
|
||||
@@ -1,6 +1,5 @@
|
||||
package buildjob;
|
||||
|
||||
|
||||
import org.junit.Test;
|
||||
|
||||
import com.atlassian.bamboo.specs.api.builders.plan.Plan;
|
||||
@@ -8,12 +7,18 @@ import com.atlassian.bamboo.specs.api.exceptions.PropertiesValidationException;
|
||||
import com.atlassian.bamboo.specs.api.util.EntityPropertiesBuilders;
|
||||
|
||||
public class PlanSpecTest {
|
||||
|
||||
@Test
|
||||
public void checkYourPlanOffline() throws PropertiesValidationException {
|
||||
|
||||
Plan plan = new PlanSpec().createPlan();
|
||||
EntityPropertiesBuilders.build(plan);
|
||||
|
||||
Plan nightPlan = new PlanSpec().createNightPlan();
|
||||
EntityPropertiesBuilders.build(nightPlan);
|
||||
|
||||
Plan secPlan = new PlanSpec().createSecBuild();
|
||||
EntityPropertiesBuilders.build(secPlan);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
<artifactId>redaction-service</artifactId>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<version>1.0-SNAPSHOT</version>
|
||||
<version>3.0-SNAPSHOT</version>
|
||||
|
||||
|
||||
<packaging>pom</packaging>
|
||||
@@ -18,4 +18,4 @@
|
||||
<module>redaction-service-image-v1</module>
|
||||
</modules>
|
||||
|
||||
</project>
|
||||
</project>
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
<parent>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<artifactId>platform-docker-dependency</artifactId>
|
||||
<version>1.1.0</version>
|
||||
<relativePath />
|
||||
<version>1.2.0</version>
|
||||
<relativePath/>
|
||||
</parent>
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
@@ -42,7 +42,7 @@
|
||||
<artifactId>docker-maven-plugin</artifactId>
|
||||
</plugin>
|
||||
</plugins>
|
||||
|
||||
|
||||
<pluginManagement>
|
||||
<plugins>
|
||||
<plugin>
|
||||
@@ -95,4 +95,4 @@
|
||||
</plugins>
|
||||
</pluginManagement>
|
||||
</build>
|
||||
</project>
|
||||
</project>
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
FROM red/redaction-service-base-v1:1.0.0
|
||||
FROM red/redaction-service-base-v1:2.0.0
|
||||
|
||||
ARG PLATFORM_JAR
|
||||
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
<parent>
|
||||
<artifactId>platform-dependency</artifactId>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<version>1.6.0</version>
|
||||
<relativePath />
|
||||
<version>1.17.0</version>
|
||||
<relativePath/>
|
||||
</parent>
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
@@ -23,20 +23,19 @@
|
||||
|
||||
<properties>
|
||||
<pdfbox.version>2.0.24</pdfbox.version>
|
||||
<dsljson.version>1.9.9</dsljson.version>
|
||||
</properties>
|
||||
|
||||
|
||||
<dependencyManagement>
|
||||
|
||||
<dependencies>
|
||||
<dependency>
|
||||
<groupId>com.iqser.red</groupId>
|
||||
<artifactId>platform-commons-dependency</artifactId>
|
||||
<version>1.11.0</version>
|
||||
<version>1.21.0</version>
|
||||
<scope>import</scope>
|
||||
<type>pom</type>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>pdfbox</artifactId>
|
||||
@@ -47,9 +46,7 @@
|
||||
<artifactId>pdfbox-tools</artifactId>
|
||||
<version>${pdfbox.version}</version>
|
||||
</dependency>
|
||||
|
||||
</dependencies>
|
||||
|
||||
</dependencyManagement>
|
||||
|
||||
<build>
|
||||
@@ -58,12 +55,10 @@
|
||||
<plugin>
|
||||
<groupId>org.sonarsource.scanner.maven</groupId>
|
||||
<artifactId>sonar-maven-plugin</artifactId>
|
||||
<version>3.9.0.2155</version>
|
||||
</plugin>
|
||||
</plugin>
|
||||
<plugin>
|
||||
<groupId>org.owasp</groupId>
|
||||
<artifactId>dependency-check-maven</artifactId>
|
||||
<version>6.3.1</version>
|
||||
<configuration>
|
||||
<format>ALL</format>
|
||||
</configuration>
|
||||
@@ -71,6 +66,12 @@
|
||||
<plugin>
|
||||
<groupId>org.jacoco</groupId>
|
||||
<artifactId>jacoco-maven-plugin</artifactId>
|
||||
<version>0.8.8</version>
|
||||
<configuration>
|
||||
<excludes>
|
||||
<exclude>org/drools/**/*</exclude>
|
||||
</excludes>
|
||||
</configuration>
|
||||
<executions>
|
||||
<execution>
|
||||
<id>prepare-agent</id>
|
||||
@@ -88,27 +89,5 @@
|
||||
</plugin>
|
||||
</plugins>
|
||||
</pluginManagement>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.jacoco</groupId>
|
||||
<artifactId>jacoco-maven-plugin</artifactId>
|
||||
<version>0.8.7</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<id>prepare-agent</id>
|
||||
<goals>
|
||||
<goal>prepare-agent</goal>
|
||||
</goals>
|
||||
</execution>
|
||||
<execution>
|
||||
<id>report</id>
|
||||
<goals>
|
||||
<goal>report-aggregate</goal>
|
||||
</goals>
|
||||
<phase>verify</phase>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
</project>
|
||||
|
||||
@@ -12,19 +12,50 @@
|
||||
<artifactId>redaction-service-api-v1</artifactId>
|
||||
|
||||
<properties>
|
||||
<persistence-service.version>1.85.0</persistence-service.version>
|
||||
<persistence-service.version>1.299.0</persistence-service.version>
|
||||
</properties>
|
||||
|
||||
<dependencies>
|
||||
|
||||
<!-- https://mvnrepository.com/artifact/com.dslplatform/dsl-json-java8 -->
|
||||
<dependency>
|
||||
<groupId>com.dslplatform</groupId>
|
||||
<artifactId>dsl-json-java8</artifactId>
|
||||
<version>${dsljson.version}</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.springframework</groupId>
|
||||
<artifactId>spring-web</artifactId>
|
||||
<optional>true</optional>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>persistence-service-api-v1</artifactId>
|
||||
<version>${persistence-service.version}</version>
|
||||
<exclusions>
|
||||
<exclusion>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>redaction-service-api-v1</artifactId>
|
||||
</exclusion>
|
||||
</exclusions>
|
||||
</dependency>
|
||||
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-compiler-plugin</artifactId>
|
||||
<configuration>
|
||||
<annotationProcessors>
|
||||
<annotationProcessor>lombok.launch.AnnotationProcessorHider$AnnotationProcessor</annotationProcessor>
|
||||
<annotationProcessor>com.dslplatform.json.processor.CompiledJsonAnnotationProcessor</annotationProcessor>
|
||||
</annotationProcessors>
|
||||
</configuration>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
</project>
|
||||
|
||||
+6
-6
@@ -1,11 +1,5 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
@@ -13,6 +7,12 @@ import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
|
||||
+3
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
@@ -32,6 +34,7 @@ public class AnalyzeResult {
|
||||
|
||||
private ManualRedactions manualRedactions;
|
||||
|
||||
private Set<FileAttribute> addedFileAttributes;
|
||||
|
||||
}
|
||||
|
||||
|
||||
+9
-1
@@ -2,6 +2,14 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum ArgumentType {
|
||||
|
||||
INTEGER, BOOLEAN, STRING, FILE_ATTRIBUTE, REGEX, TYPE, RULE_NUMBER, LEGAL_BASIS, REFERENCE_TYPE
|
||||
INTEGER,
|
||||
BOOLEAN,
|
||||
STRING,
|
||||
FILE_ATTRIBUTE,
|
||||
REGEX,
|
||||
TYPE,
|
||||
RULE_NUMBER,
|
||||
LEGAL_BASIS,
|
||||
REFERENCE_TYPE
|
||||
|
||||
}
|
||||
|
||||
+3
-2
@@ -1,12 +1,12 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@@ -16,4 +16,5 @@ public class Change {
|
||||
private int analysisNumber;
|
||||
private ChangeType type;
|
||||
private OffsetDateTime dateTime;
|
||||
|
||||
}
|
||||
|
||||
+3
-1
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum ChangeType {
|
||||
ADDED, REMOVED, CHANGED
|
||||
ADDED,
|
||||
REMOVED,
|
||||
CHANGED
|
||||
}
|
||||
|
||||
+3
-1
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum Engine {
|
||||
DICTIONARY, NER, RULE
|
||||
DICTIONARY,
|
||||
NER,
|
||||
RULE
|
||||
}
|
||||
|
||||
+4
-3
@@ -1,13 +1,13 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@@ -18,4 +18,5 @@ public class ImportedRedaction {
|
||||
|
||||
@Builder.Default
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
|
||||
}
|
||||
|
||||
+7
-3
@@ -1,20 +1,24 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class ImportedRedactions {
|
||||
|
||||
@Builder.Default
|
||||
private Map<Integer, List<ImportedRedaction>> importedRedactions = new HashMap<>();
|
||||
|
||||
}
|
||||
|
||||
+16
-5
@@ -2,6 +2,7 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.BaseAnnotation;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
@@ -20,27 +21,37 @@ public class ManualChange {
|
||||
private AnnotationStatus annotationStatus;
|
||||
private ManualRedactionType manualRedactionType;
|
||||
private OffsetDateTime processedDate;
|
||||
private OffsetDateTime requestedDate;
|
||||
private String userId;
|
||||
private Map<String, Object> propertyChanges = new HashMap<>();
|
||||
private Map<String, String> propertyChanges = new HashMap<>();
|
||||
|
||||
public boolean isProcessed() {
|
||||
return processedDate != null;
|
||||
}
|
||||
|
||||
public static ManualChange from(BaseAnnotation baseAnnotation) {
|
||||
|
||||
ManualChange manualChange = new ManualChange();
|
||||
manualChange.annotationStatus = baseAnnotation.getStatus();
|
||||
manualChange.processedDate = baseAnnotation.getProcessedDate();
|
||||
manualChange.requestedDate = baseAnnotation.getRequestDate();
|
||||
manualChange.userId = baseAnnotation.getUser();
|
||||
return manualChange;
|
||||
}
|
||||
|
||||
|
||||
public boolean isProcessed() {
|
||||
|
||||
return processedDate != null;
|
||||
}
|
||||
|
||||
|
||||
public ManualChange withManualRedactionType(ManualRedactionType manualRedactionType) {
|
||||
|
||||
this.manualRedactionType = manualRedactionType;
|
||||
return this;
|
||||
}
|
||||
|
||||
public ManualChange withChange(String property, Object value) {
|
||||
|
||||
public ManualChange withChange(String property, String value) {
|
||||
|
||||
this.propertyChanges.put(property, value);
|
||||
return this;
|
||||
}
|
||||
|
||||
+4
-1
@@ -2,6 +2,9 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
public enum MessageType {
|
||||
|
||||
ANALYSE, REANALYSE, STRUCTURE_ANALYSE, SURROUNDING_TEXT
|
||||
ANALYSE,
|
||||
REANALYSE,
|
||||
STRUCTURE_ANALYSE,
|
||||
SURROUNDING_TEXT
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -12,4 +12,5 @@ import lombok.NoArgsConstructor;
|
||||
public class ReanalyzeResult {
|
||||
|
||||
private RedactionLog redactionLog;
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -14,4 +14,5 @@ public class Rectangle {
|
||||
private float height;
|
||||
|
||||
private int page;
|
||||
|
||||
}
|
||||
|
||||
+8
-6
@@ -1,17 +1,20 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.legalbasis.LegalBasis;
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
|
||||
@Data
|
||||
@CompiledJson
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class RedactionLog {
|
||||
|
||||
|
||||
/**
|
||||
* Version 0 Redaction Logs have manual redactions merged inside them
|
||||
* Version 1 Redaction Logs only contain system ( rule/dictionary ) redactions. Manual Redactions are merged in at runtime.
|
||||
@@ -23,13 +26,12 @@ public class RedactionLog {
|
||||
*/
|
||||
private int analysisNumber;
|
||||
|
||||
private List<RedactionLogEntry> redactionLogEntry;
|
||||
private List<LegalBasis> legalBasis;
|
||||
private List<RedactionLogEntry> redactionLogEntry = new ArrayList<>();
|
||||
private List<RedactionLogLegalBasis> legalBasis = new ArrayList<>();
|
||||
|
||||
private long dictionaryVersion = -1;
|
||||
private long dossierDictionaryVersion = -1;
|
||||
private long rulesVersion = -1;
|
||||
private long legalBasisVersion = -1;
|
||||
|
||||
|
||||
}
|
||||
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.OffsetDateTime;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionLogComment {
|
||||
|
||||
private long id;
|
||||
private String user;
|
||||
private String text;
|
||||
private String annotationId;
|
||||
private String fileId;
|
||||
private OffsetDateTime date;
|
||||
private OffsetDateTime softDeletedTime;
|
||||
|
||||
}
|
||||
+18
-8
@@ -1,12 +1,11 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.Comment;
|
||||
|
||||
import lombok.*;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@@ -27,6 +26,7 @@ public class RedactionLogEntry {
|
||||
private boolean redacted;
|
||||
private boolean isHint;
|
||||
private boolean isRecommendation;
|
||||
private boolean isFalsePositive;
|
||||
|
||||
private String section;
|
||||
private float[] color;
|
||||
@@ -35,12 +35,11 @@ public class RedactionLogEntry {
|
||||
private List<Rectangle> positions = new ArrayList<>();
|
||||
private int sectionNumber;
|
||||
|
||||
|
||||
private String textBefore;
|
||||
private String textAfter;
|
||||
|
||||
@Builder.Default
|
||||
private List<Comment> comments = new ArrayList<>();
|
||||
private List<RedactionLogComment> comments = new ArrayList<>();
|
||||
|
||||
private int startOffset;
|
||||
private int endOffset;
|
||||
@@ -53,6 +52,8 @@ public class RedactionLogEntry {
|
||||
|
||||
private boolean excluded;
|
||||
|
||||
private String sourceId;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
@Builder.Default
|
||||
private List<Change> changes = new ArrayList<>();
|
||||
@@ -65,21 +66,30 @@ public class RedactionLogEntry {
|
||||
|
||||
private Set<String> reference = new HashSet<>();
|
||||
|
||||
@Builder.Default
|
||||
private Set<String> importedRedactionIntersections = new HashSet<>();
|
||||
|
||||
|
||||
public boolean lastChangeIsRemoved() {
|
||||
|
||||
return last(changes).map(c -> c.getType() == ChangeType.REMOVED).orElse(false);
|
||||
}
|
||||
|
||||
|
||||
public boolean isLocalManualRedaction() {
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY &&
|
||||
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.ADD_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
}
|
||||
|
||||
|
||||
public boolean isManuallyRemoved() {
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY &&
|
||||
mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
|
||||
return manualChanges.stream().anyMatch(mc -> mc.getManualRedactionType() == ManualRedactionType.REMOVE_LOCALLY && mc.getAnnotationStatus() == AnnotationStatus.APPROVED);
|
||||
}
|
||||
|
||||
|
||||
private <T> Optional<T> last(List<T> list) {
|
||||
|
||||
return list.isEmpty() ? Optional.empty() : Optional.of(list.get(list.size() - 1));
|
||||
}
|
||||
|
||||
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class RedactionLogLegalBasis {
|
||||
|
||||
private String name;
|
||||
private String description;
|
||||
private String reason;
|
||||
|
||||
}
|
||||
+13
-4
@@ -1,15 +1,18 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.configuration.Colors;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.type.Type;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@@ -22,4 +25,10 @@ public class RedactionRequest {
|
||||
private ManualRedactions manualRedactions;
|
||||
@Builder.Default
|
||||
private Set<Integer> excludedPages = new HashSet<>();
|
||||
|
||||
private Colors colors;
|
||||
private List<Type> types;
|
||||
|
||||
private boolean includeFalsePositives;
|
||||
|
||||
}
|
||||
|
||||
+9
-13
@@ -3,36 +3,32 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.NonNull;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Data
|
||||
@RequiredArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class SectionArea {
|
||||
|
||||
@NonNull
|
||||
private Point topLeft;
|
||||
|
||||
@NonNull
|
||||
private float width;
|
||||
|
||||
@NonNull
|
||||
private float height;
|
||||
|
||||
@NonNull
|
||||
private int page;
|
||||
|
||||
private String header;
|
||||
|
||||
|
||||
public boolean contains(Rectangle other) {
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft().getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft().getY() + other.getHeight();
|
||||
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeft().getX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeft()
|
||||
.getX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeft().getY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeft()
|
||||
.getY() + other.getHeight();
|
||||
}
|
||||
|
||||
|
||||
// TODO we should only use one rectangle class.
|
||||
public boolean contains(com.iqser.red.service.persistence.service.v1.api.model.annotations.Rectangle other) {
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft().getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
|
||||
|
||||
return page == other.getPage() && this.topLeft.getX() <= other.getTopLeftX() && this.topLeft.getX() + this.getWidth() >= other.getTopLeftX() + other.getWidth() && this.getTopLeft()
|
||||
.getY() <= other.getTopLeftY() && this.getTopLeft().getY() + this.getHeight() >= other.getTopLeftY() + other.getHeight();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+10
-6
@@ -1,13 +1,15 @@
|
||||
package com.iqser.red.service.redaction.v1.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
@Data
|
||||
@CompiledJson
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class SectionGrid {
|
||||
@@ -17,13 +19,15 @@ public class SectionGrid {
|
||||
private List<SectionGridSection> sections = new ArrayList<>();
|
||||
|
||||
@Data
|
||||
@RequiredArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public static class SectionGridSection {
|
||||
|
||||
private final int sectionNumber;
|
||||
private final String headline;
|
||||
private final Set<Integer> pages;
|
||||
private final List<SectionArea> sectionAreas;
|
||||
private int sectionNumber;
|
||||
private String headline;
|
||||
private Set<Integer> pages;
|
||||
private List<SectionArea> sectionAreas;
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-12
@@ -3,30 +3,18 @@ package com.iqser.red.service.redaction.v1.model;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
import lombok.NonNull;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
@RequiredArgsConstructor
|
||||
public class SectionRectangle {
|
||||
|
||||
@NonNull
|
||||
private Point topLeft;
|
||||
|
||||
@NonNull
|
||||
private float width;
|
||||
|
||||
@NonNull
|
||||
private float height;
|
||||
|
||||
@NonNull
|
||||
private int part;
|
||||
|
||||
@NonNull
|
||||
private int numberOfParts;
|
||||
|
||||
private List<CellRectangle> tableCells;
|
||||
|
||||
+4
-9
@@ -1,20 +1,17 @@
|
||||
package com.iqser.red.service.redaction.v1.resources;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.redaction.v1.model.*;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLog;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionResult;
|
||||
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.web.bind.annotation.PathVariable;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RequestParam;
|
||||
|
||||
public interface RedactionResource {
|
||||
|
||||
@PostMapping(value = "/annotate", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest);
|
||||
|
||||
|
||||
@PostMapping(value = "/debug/classifications", produces = MediaType.APPLICATION_JSON_VALUE, consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
RedactionResult classify(@RequestBody RedactionRequest redactionRequest);
|
||||
|
||||
@@ -36,8 +33,6 @@ public interface RedactionResource {
|
||||
|
||||
|
||||
@PostMapping(value = "/manual/surrounding-text/{dossierId}/{fileId}", consumes = MediaType.APPLICATION_JSON_VALUE, produces = MediaType.APPLICATION_JSON_VALUE)
|
||||
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
|
||||
@PathVariable("fileId") String fileId,
|
||||
@RequestBody ManualRedactions manualRedactions);
|
||||
ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions);
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -1,6 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.resources;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
|
||||
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
|
||||
|
||||
@@ -12,18 +12,50 @@
|
||||
<artifactId>redaction-service-server-v1</artifactId>
|
||||
|
||||
<properties>
|
||||
<drools.version>7.59.0.Final</drools.version>
|
||||
<kie.version>7.59.0.Final</kie.version>
|
||||
<locationtech.version>1.16.1</locationtech.version>
|
||||
<pdfbox.jbig2-imageio.version>3.0.3</pdfbox.jbig2-imageio.version>
|
||||
<jai-imageio.version>1.4.0</jai-imageio.version>
|
||||
<drools.version>7.73.0.Final</drools.version>
|
||||
<kie.version>7.73.0.Final</kie.version>
|
||||
<locationtech.version>1.18.2</locationtech.version>
|
||||
<javaassist.version>3.28.0-GA</javaassist.version>
|
||||
<ahocorasick.version>0.6.3</ahocorasick.version>
|
||||
<jackson.version>2.13.2</jackson.version>
|
||||
</properties>
|
||||
|
||||
<dependencies>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.springframework.boot</groupId>
|
||||
<artifactId>spring-boot-starter-aop</artifactId>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
<artifactId>storage-commons</artifactId>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.module</groupId>
|
||||
<artifactId>jackson-module-afterburner</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.datatype</groupId>
|
||||
<artifactId>jackson-datatype-jsr310</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.ahocorasick</groupId>
|
||||
<artifactId>ahocorasick</artifactId>
|
||||
<version>${ahocorasick.version}</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.javassist</groupId>
|
||||
<artifactId>javassist</artifactId>
|
||||
<version>${javaassist.version}</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.iqser.red.service</groupId>
|
||||
<artifactId>redaction-service-api-v1</artifactId>
|
||||
@@ -49,22 +81,6 @@
|
||||
<artifactId>guava</artifactId>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.apache.pdfbox</groupId>
|
||||
<artifactId>jbig2-imageio</artifactId>
|
||||
<version>${pdfbox.jbig2-imageio.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.github.jai-imageio</groupId>
|
||||
<artifactId>jai-imageio-core</artifactId>
|
||||
<version>${jai-imageio.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.github.jai-imageio</groupId>
|
||||
<artifactId>jai-imageio-jpeg2000</artifactId>
|
||||
<version>${jai-imageio.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- commons -->
|
||||
<dependency>
|
||||
<groupId>com.iqser.red.commons</groupId>
|
||||
@@ -113,6 +129,18 @@
|
||||
|
||||
<build>
|
||||
<plugins>
|
||||
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-compiler-plugin</artifactId>
|
||||
<configuration>
|
||||
<annotationProcessors>
|
||||
<annotationProcessor>lombok.launch.AnnotationProcessorHider$AnnotationProcessor</annotationProcessor>
|
||||
<annotationProcessor>com.dslplatform.json.processor.CompiledJsonAnnotationProcessor</annotationProcessor>
|
||||
</annotationProcessors>
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<plugin>
|
||||
<!-- generate git.properties for exposure in /info -->
|
||||
<groupId>pl.project13.maven</groupId>
|
||||
|
||||
+12
@@ -3,14 +3,19 @@ package com.iqser.red.service.redaction.v1.server;
|
||||
import com.iqser.red.commons.spring.DefaultWebMvcConfiguration;
|
||||
import com.iqser.red.service.redaction.v1.server.client.RulesClient;
|
||||
import com.iqser.red.service.redaction.v1.server.settings.RedactionServiceSettings;
|
||||
|
||||
import org.springframework.boot.SpringApplication;
|
||||
import org.springframework.boot.actuate.autoconfigure.security.servlet.ManagementWebSecurityAutoConfiguration;
|
||||
import org.springframework.boot.autoconfigure.SpringBootApplication;
|
||||
import org.springframework.boot.autoconfigure.security.servlet.SecurityAutoConfiguration;
|
||||
import org.springframework.boot.context.properties.EnableConfigurationProperties;
|
||||
import org.springframework.cloud.openfeign.EnableFeignClients;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Import;
|
||||
|
||||
import io.micrometer.core.aop.TimedAspect;
|
||||
import io.micrometer.core.instrument.MeterRegistry;
|
||||
|
||||
@Import({DefaultWebMvcConfiguration.class})
|
||||
@EnableFeignClients(basePackageClasses = RulesClient.class)
|
||||
@EnableConfigurationProperties(RedactionServiceSettings.class)
|
||||
@@ -18,9 +23,16 @@ import org.springframework.context.annotation.Import;
|
||||
public class Application {
|
||||
|
||||
public static void main(String[] args) {
|
||||
|
||||
System.setProperty("org.apache.pdfbox.rendering.UsePureJavaCMYKConversion", "true");
|
||||
SpringApplication.run(Application.class, args);
|
||||
}
|
||||
|
||||
|
||||
@Bean
|
||||
public TimedAspect timedAspect(MeterRegistry registry) {
|
||||
|
||||
return new TimedAspect(registry);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+1
-1
@@ -14,7 +14,7 @@ import lombok.NoArgsConstructor;
|
||||
public class Document {
|
||||
|
||||
private List<Page> pages = new ArrayList<>();
|
||||
private List<Paragraph> paragraphs = new ArrayList<>();
|
||||
private List<Section> sections = new ArrayList<>();
|
||||
private List<Header> headers = new ArrayList<>();
|
||||
private List<Footer> footers = new ArrayList<>();
|
||||
private List<UnclassifiedText> unclassifiedTexts = new ArrayList<>();
|
||||
|
||||
+10
-6
@@ -14,7 +14,9 @@ public class FloatFrequencyCounter {
|
||||
@Getter
|
||||
Map<Float, Integer> countPerValue = new HashMap<>();
|
||||
|
||||
|
||||
public void add(float value) {
|
||||
|
||||
if (!countPerValue.containsKey(value)) {
|
||||
countPerValue.put(value, 1);
|
||||
} else {
|
||||
@@ -22,7 +24,9 @@ public class FloatFrequencyCounter {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public void addAll(Map<Float, Integer> otherCounter) {
|
||||
|
||||
for (Map.Entry<Float, Integer> entry : otherCounter.entrySet()) {
|
||||
if (countPerValue.containsKey(entry.getKey())) {
|
||||
countPerValue.put(entry.getKey(), countPerValue.get(entry.getKey()) + entry.getValue());
|
||||
@@ -32,12 +36,12 @@ public class FloatFrequencyCounter {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public Float getMostPopular() {
|
||||
|
||||
Map.Entry<Float, Integer> mostPopular = null;
|
||||
for (Map.Entry<Float, Integer> entry : countPerValue.entrySet()) {
|
||||
if (mostPopular == null) {
|
||||
mostPopular = entry;
|
||||
} else if (entry.getValue() >= mostPopular.getValue()) {
|
||||
if (mostPopular == null || entry.getValue() >= mostPopular.getValue()) {
|
||||
mostPopular = entry;
|
||||
}
|
||||
}
|
||||
@@ -46,6 +50,7 @@ public class FloatFrequencyCounter {
|
||||
|
||||
|
||||
public List<Float> getHighterThanMostPopular() {
|
||||
|
||||
Float mostPopular = getMostPopular();
|
||||
List<Float> higher = new ArrayList<>();
|
||||
for (Float value : countPerValue.keySet()) {
|
||||
@@ -59,11 +64,10 @@ public class FloatFrequencyCounter {
|
||||
|
||||
|
||||
public Float getHighest() {
|
||||
|
||||
Float highest = null;
|
||||
for (Float value : countPerValue.keySet()) {
|
||||
if (highest == null) {
|
||||
highest = value;
|
||||
} else if (value > highest) {
|
||||
if (highest == null || value > highest) {
|
||||
highest = value;
|
||||
}
|
||||
}
|
||||
|
||||
+4
@@ -1,7 +1,9 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
|
||||
@@ -13,7 +15,9 @@ public class Footer {
|
||||
|
||||
private List<TextBlock> textBlocks;
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public SearchableText getSearchableText() {
|
||||
|
||||
SearchableText searchableText = new SearchableText();
|
||||
|
||||
+4
@@ -1,7 +1,9 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
|
||||
@@ -13,7 +15,9 @@ public class Header {
|
||||
|
||||
private List<TextBlock> textBlocks;
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public SearchableText getSearchableText() {
|
||||
|
||||
SearchableText searchableText = new SearchableText();
|
||||
|
||||
+3
-1
@@ -2,5 +2,7 @@ package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
public enum Orientation {
|
||||
|
||||
NONE, LEFT, RIGHT
|
||||
NONE,
|
||||
LEFT,
|
||||
RIGHT
|
||||
}
|
||||
|
||||
+12
-8
@@ -1,15 +1,18 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
|
||||
import lombok.Data;
|
||||
import lombok.NonNull;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.NonNull;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Data
|
||||
@RequiredArgsConstructor
|
||||
public class Page {
|
||||
@@ -31,7 +34,8 @@ public class Page {
|
||||
private StringFrequencyCounter fontCounter = new StringFrequencyCounter();
|
||||
private StringFrequencyCounter fontStyleCounter = new StringFrequencyCounter();
|
||||
|
||||
private double cropBoxArea;
|
||||
private float pageWidth;
|
||||
private float pageHeight;
|
||||
|
||||
|
||||
public boolean isRotated() {
|
||||
|
||||
+14
-1
@@ -1,18 +1,26 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.AnnotationStatus;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.IdRemoval;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.entitymapped.ManualImageRecategorization;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entities;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.IdBuilder;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
@Data
|
||||
@NoArgsConstructor
|
||||
public class Paragraph implements Comparable {
|
||||
public class Section implements Comparable {
|
||||
|
||||
private List<AbstractTextContainer> pageBlocks = new ArrayList<>();
|
||||
private List<PdfImage> images = new ArrayList<>();
|
||||
@@ -61,4 +69,9 @@ public class Paragraph implements Comparable {
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
}
|
||||
+21
-3
@@ -1,19 +1,29 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.CellValue;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Image;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Paragraph;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.*;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class SectionText {
|
||||
@@ -23,21 +33,29 @@ public class SectionText {
|
||||
|
||||
private boolean isTable;
|
||||
private String headline;
|
||||
List<Paragraph> paragraphs;
|
||||
|
||||
@Builder.Default
|
||||
private List<SectionArea> sectionAreas = new ArrayList<>();
|
||||
@Builder.Default
|
||||
private Set<Image> images = new HashSet<>();
|
||||
|
||||
@Builder.Default
|
||||
private List<TextBlock> textBlocks = new ArrayList<>();
|
||||
@Builder.Default
|
||||
private Map<String, CellValue> tabularData = new HashMap<>();
|
||||
@Builder.Default
|
||||
private List<Integer> cellStarts = new ArrayList<>();
|
||||
|
||||
|
||||
public void setTabularData(Map<String, CellValue> tabularData) {
|
||||
|
||||
tabularData.remove(null);
|
||||
this.tabularData = tabularData;
|
||||
}
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public SearchableText getSearchableText() {
|
||||
|
||||
SearchableText searchableText = new SearchableText();
|
||||
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class SimplifiedSectionText {
|
||||
|
||||
private int sectionNumber;
|
||||
private String text;
|
||||
|
||||
}
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class SimplifiedText {
|
||||
|
||||
private int numberOfPages;
|
||||
private List<SimplifiedSectionText> sectionTexts = new ArrayList<>();
|
||||
|
||||
}
|
||||
+1
-3
@@ -37,9 +37,7 @@ public class StringFrequencyCounter {
|
||||
|
||||
Map.Entry<String, Integer> mostPopular = null;
|
||||
for (Map.Entry<String, Integer> entry : countPerValue.entrySet()) {
|
||||
if (mostPopular == null) {
|
||||
mostPopular = entry;
|
||||
} else if (entry.getValue() > mostPopular.getValue()) {
|
||||
if (mostPopular == null || entry.getValue() > mostPopular.getValue()) {
|
||||
mostPopular = entry;
|
||||
}
|
||||
}
|
||||
|
||||
+3
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
@@ -8,6 +10,7 @@ import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class Text {
|
||||
|
||||
+184
-14
@@ -1,39 +1,193 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextDirection;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.TextNormalizationUtilities;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@AllArgsConstructor
|
||||
@Builder
|
||||
@Data
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
public class TextBlock extends AbstractTextContainer {
|
||||
|
||||
@Builder.Default
|
||||
private List<TextPositionSequence> sequences = new ArrayList<>();
|
||||
|
||||
@JsonIgnore
|
||||
private int rotation;
|
||||
|
||||
private int indexOnPage;
|
||||
|
||||
@JsonIgnore
|
||||
private String mostPopularWordFont;
|
||||
|
||||
@JsonIgnore
|
||||
private String mostPopularWordStyle;
|
||||
|
||||
@JsonIgnore
|
||||
private float mostPopularWordFontSize;
|
||||
|
||||
@JsonIgnore
|
||||
private float mostPopularWordHeight;
|
||||
|
||||
@JsonIgnore
|
||||
private float mostPopularWordSpaceWidth;
|
||||
|
||||
@JsonIgnore
|
||||
private float highestFontSize;
|
||||
|
||||
@JsonIgnore
|
||||
private String classification;
|
||||
|
||||
|
||||
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation) {
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public TextDirection getDir() {
|
||||
|
||||
return sequences.get(0).getDir();
|
||||
}
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
private float getPageHeight() {
|
||||
|
||||
return sequences.get(0).getPageHeight();
|
||||
}
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
private float getPageWidth() {
|
||||
|
||||
return sequences.get(0).getPageWidth();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the minX value in pdf coordinate system.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
*
|
||||
* @return the minX value in pdf coordinate system
|
||||
*/
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public float getPdfMinX() {
|
||||
|
||||
if (getDir().getDegrees() == 90) {
|
||||
return minY;
|
||||
} else if (getDir().getDegrees() == 180) {
|
||||
return getPageWidth() - maxX;
|
||||
|
||||
} else if (getDir().getDegrees() == 270) {
|
||||
|
||||
return getPageWidth() - maxY;
|
||||
} else {
|
||||
return minX;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the maxX value in pdf coordinate system.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
*
|
||||
* @return the maxX value in pdf coordinate system
|
||||
*/
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public float getPdfMaxX() {
|
||||
|
||||
if (getDir().getDegrees() == 90) {
|
||||
return maxY;
|
||||
} else if (getDir().getDegrees() == 180) {
|
||||
return getPageWidth() - minX;
|
||||
} else if (getDir().getDegrees() == 270) {
|
||||
return getPageWidth() - minY;
|
||||
|
||||
} else {
|
||||
return maxX;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the minY value in pdf coordinate system.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
*
|
||||
* @return the minY value in pdf coordinate system
|
||||
*/
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public float getPdfMinY() {
|
||||
|
||||
if (getDir().getDegrees() == 90) {
|
||||
return minX;
|
||||
} else if (getDir().getDegrees() == 180) {
|
||||
return maxY;
|
||||
|
||||
} else if (getDir().getDegrees() == 270) {
|
||||
return getPageHeight() - maxX;
|
||||
|
||||
} else {
|
||||
return getPageHeight() - maxY;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the maxY value in pdf coordinate system.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
*
|
||||
* @return the maxY value in pdf coordinate system
|
||||
*/
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public float getPdfMaxY() {
|
||||
|
||||
if (getDir().getDegrees() == 90) {
|
||||
return maxX;
|
||||
} else if (getDir().getDegrees() == 180) {
|
||||
|
||||
return minY;
|
||||
} else if (getDir().getDegrees() == 270) {
|
||||
return getPageHeight() - minX;
|
||||
} else {
|
||||
return getPageHeight() - minY;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public TextBlock(float minX, float maxX, float minY, float maxY, List<TextPositionSequence> sequences, int rotation, int indexOnPage) {
|
||||
this.indexOnPage = indexOnPage;
|
||||
this.minX = minX;
|
||||
this.maxX = maxX;
|
||||
this.minY = minY;
|
||||
@@ -42,19 +196,25 @@ public class TextBlock extends AbstractTextContainer {
|
||||
this.rotation = rotation;
|
||||
}
|
||||
|
||||
|
||||
public TextBlock union(TextPositionSequence r) {
|
||||
|
||||
TextBlock union = this.copy();
|
||||
union.add(r);
|
||||
return union;
|
||||
}
|
||||
|
||||
|
||||
public TextBlock union(TextBlock r) {
|
||||
|
||||
TextBlock union = this.copy();
|
||||
union.add(r);
|
||||
return union;
|
||||
}
|
||||
|
||||
|
||||
public void add(TextBlock r) {
|
||||
|
||||
if (r.getMinX() < minX) {
|
||||
minX = r.getMinX();
|
||||
}
|
||||
@@ -70,30 +230,38 @@ public class TextBlock extends AbstractTextContainer {
|
||||
sequences.addAll(r.getSequences());
|
||||
}
|
||||
|
||||
|
||||
public void add(TextPositionSequence r) {
|
||||
if (r.getX1() < minX) {
|
||||
minX = r.getX1();
|
||||
|
||||
if (r.getMinXDirAdj() < minX) {
|
||||
minX = r.getMinXDirAdj();
|
||||
}
|
||||
if (r.getX2() > maxX) {
|
||||
maxX = r.getX2();
|
||||
if (r.getMaxXDirAdj() > maxX) {
|
||||
maxX = r.getMaxXDirAdj();
|
||||
}
|
||||
if (r.getY1() < minY) {
|
||||
minY = r.getY1();
|
||||
if (r.getMinYDirAdj() < minY) {
|
||||
minY = r.getMinYDirAdj();
|
||||
}
|
||||
if (r.getY2() > maxY) {
|
||||
maxY = r.getY2();
|
||||
if (r.getMaxYDirAdj() > maxY) {
|
||||
maxY = r.getMaxYDirAdj();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public TextBlock copy() {
|
||||
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation);
|
||||
|
||||
return new TextBlock(minX, maxX, minY, maxY, sequences, rotation, indexOnPage);
|
||||
}
|
||||
|
||||
|
||||
public void resize(float x1, float y1, float width, float height) {
|
||||
|
||||
set(x1, y1, x1 + width, y1 + height);
|
||||
}
|
||||
|
||||
|
||||
public void set(float x1, float y1, float x2, float y2) {
|
||||
|
||||
this.minX = Math.min(x1, x2);
|
||||
this.maxX = Math.max(x1, x2);
|
||||
this.minY = Math.min(y1, y2);
|
||||
@@ -119,8 +287,10 @@ public class TextBlock extends AbstractTextContainer {
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public String getText() {
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
@@ -128,7 +298,7 @@ public class TextBlock extends AbstractTextContainer {
|
||||
TextPositionSequence previous = null;
|
||||
for (TextPositionSequence word : sequences) {
|
||||
if (previous != null) {
|
||||
if (Math.abs(previous.getRotationAdjustedY() - word.getRotationAdjustedY()) > word.getTextHeight()) {
|
||||
if (Math.abs(previous.getMaxYDirAdj() - word.getMaxYDirAdj()) > word.getTextHeight()) {
|
||||
sb.append('\n');
|
||||
} else {
|
||||
sb.append(' ');
|
||||
|
||||
+4
@@ -1,7 +1,9 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.model;
|
||||
|
||||
import com.dslplatform.json.JsonAttribute;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnore;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchableText;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Data;
|
||||
|
||||
@@ -13,7 +15,9 @@ public class UnclassifiedText {
|
||||
|
||||
private List<TextBlock> textBlocks;
|
||||
|
||||
|
||||
@JsonIgnore
|
||||
@JsonAttribute(ignore = true)
|
||||
public SearchableText getSearchableText() {
|
||||
|
||||
SearchableText searchableText = new SearchableText();
|
||||
|
||||
+105
-158
@@ -2,25 +2,22 @@ package com.iqser.red.service.redaction.v1.server.classification.service;
|
||||
|
||||
import static java.util.stream.Collectors.toSet;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.Iterator;
|
||||
import java.util.List;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Orientation;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.StringFrequencyCounter;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.utils.RulingTextDirAdjustUtil;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.Iterator;
|
||||
import java.util.List;
|
||||
|
||||
@Service
|
||||
@SuppressWarnings("all")
|
||||
@@ -29,11 +26,19 @@ public class BlockificationService {
|
||||
static final float THRESHOLD = 1f;
|
||||
|
||||
|
||||
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines,
|
||||
List<Ruling> verticalRulingLines) {
|
||||
|
||||
sortRotatedSequences(textPositions);
|
||||
/**
|
||||
* This method is building blocks by expanding the minX/maxX and minY/maxY value on each word that is not split by the conditions.
|
||||
* This method must use text direction adjusted postions (DirAdj). Where {0,0} is on the upper left. Never try to change this!
|
||||
* Rulings (Table lines) must be adjusted to the text directions as well, when checking if a block is split by a ruling.
|
||||
*
|
||||
* @param textPositions The words of a page.
|
||||
* @param horizontalRulingLines Horizontal table lines.
|
||||
* @param verticalRulingLines Vertical table lines.
|
||||
* @return Page object that contains the Textblock and text statistics.
|
||||
*/
|
||||
public Page blockify(List<TextPositionSequence> textPositions, List<Ruling> horizontalRulingLines, List<Ruling> verticalRulingLines) {
|
||||
|
||||
int indexOnPage = 0;
|
||||
List<TextPositionSequence> chunkWords = new ArrayList<>();
|
||||
List<AbstractTextContainer> chunkBlockList1 = new ArrayList<>();
|
||||
|
||||
@@ -44,35 +49,36 @@ public class BlockificationService {
|
||||
Float splitX1 = null;
|
||||
for (TextPositionSequence word : textPositions) {
|
||||
|
||||
boolean lineSeparation = minY - word.getY2() > word.getHeight() * 1.25;
|
||||
boolean startFromTop = word.getY1() > maxY + word.getHeight();
|
||||
boolean splitByX = prev != null && maxX + 50 < word.getX1() && prev.getY1() == word.getY1();
|
||||
boolean newLineAfterSplit = prev != null && word.getY1() != prev.getY1() && wasSplitted && splitX1 != word.getX1();
|
||||
boolean splittedByRuling = word.getRotation() == 0 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), verticalRulingLines) || word
|
||||
.getRotation() == 0 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), horizontalRulingLines) || word
|
||||
.getRotation() == 90 && isSplittedByRuling(maxX, minY, word.getX1(), word.getY1(), horizontalRulingLines) || word
|
||||
.getRotation() == 90 && isSplittedByRuling(minX, minY, word.getX1(), word.getY2(), verticalRulingLines);
|
||||
boolean lineSeparation = word.getMinYDirAdj() - maxY > word.getHeight() * 1.25;
|
||||
boolean startFromTop = prev != null && word.getMinYDirAdj() < prev.getMinYDirAdj() - prev.getTextHeight();
|
||||
boolean splitByX = prev != null && maxX + 50 < word.getMinXDirAdj() && prev.getMinYDirAdj() == word.getMinYDirAdj();
|
||||
boolean xIsBeforeFirstX = prev != null && word.getMinXDirAdj() < minX;
|
||||
boolean newLineAfterSplit = prev != null && word.getMinYDirAdj() != prev.getMinYDirAdj() && wasSplitted && splitX1 != word.getMinXDirAdj();
|
||||
boolean isSplitByRuling = isSplitByRuling(minX, minY, maxX, maxY, word, horizontalRulingLines, verticalRulingLines);
|
||||
boolean splitByDir = prev != null && !prev.getDir().equals(word.getDir());
|
||||
|
||||
if (prev != null && (lineSeparation || startFromTop || splitByX || newLineAfterSplit || splittedByRuling)) {
|
||||
if (prev != null && (lineSeparation || startFromTop || splitByX || splitByDir || isSplitByRuling)) {
|
||||
|
||||
Orientation prevOrientation = null;
|
||||
if (!chunkBlockList1.isEmpty()) {
|
||||
prevOrientation = chunkBlockList1.get(chunkBlockList1.size() - 1).getOrientation();
|
||||
}
|
||||
|
||||
TextBlock cb1 = buildTextBlock(chunkWords);
|
||||
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
|
||||
indexOnPage++;
|
||||
|
||||
chunkBlockList1.add(cb1);
|
||||
chunkWords = new ArrayList<>();
|
||||
|
||||
if (splitByX && !splittedByRuling) {
|
||||
if (splitByX && !isSplitByRuling) {
|
||||
wasSplitted = true;
|
||||
cb1.setOrientation(Orientation.LEFT);
|
||||
splitX1 = word.getX1();
|
||||
} else if (newLineAfterSplit && !splittedByRuling) {
|
||||
splitX1 = word.getMinXDirAdj();
|
||||
} else if (newLineAfterSplit && !isSplitByRuling) {
|
||||
wasSplitted = false;
|
||||
cb1.setOrientation(Orientation.RIGHT);
|
||||
splitX1 = null;
|
||||
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !splittedByRuling)) {
|
||||
} else if (prevOrientation != null && prevOrientation.equals(Orientation.RIGHT) && (lineSeparation || !startFromTop || !splitByX || !newLineAfterSplit || !isSplitByRuling)) {
|
||||
cb1.setOrientation(Orientation.LEFT);
|
||||
}
|
||||
|
||||
@@ -86,21 +92,21 @@ public class BlockificationService {
|
||||
chunkWords.add(word);
|
||||
|
||||
prev = word;
|
||||
if (word.getX1() < minX) {
|
||||
minX = word.getX1();
|
||||
if (word.getMinXDirAdj() < minX) {
|
||||
minX = word.getMinXDirAdj();
|
||||
}
|
||||
if (word.getX2() > maxX) {
|
||||
maxX = word.getX2();
|
||||
if (word.getMaxXDirAdj() > maxX) {
|
||||
maxX = word.getMaxXDirAdj();
|
||||
}
|
||||
if (word.getY1() < minY) {
|
||||
minY = word.getY1();
|
||||
if (word.getMinYDirAdj() < minY) {
|
||||
minY = word.getMinYDirAdj();
|
||||
}
|
||||
if (word.getY2() > maxY) {
|
||||
maxY = word.getY2();
|
||||
if (word.getMaxYDirAdj() > maxY) {
|
||||
maxY = word.getMaxYDirAdj();
|
||||
}
|
||||
}
|
||||
|
||||
TextBlock cb1 = buildTextBlock(chunkWords);
|
||||
TextBlock cb1 = buildTextBlock(chunkWords, indexOnPage);
|
||||
if (cb1 != null) {
|
||||
chunkBlockList1.add(cb1);
|
||||
}
|
||||
@@ -113,8 +119,7 @@ public class BlockificationService {
|
||||
TextBlock block = (TextBlock) itty.next();
|
||||
|
||||
if (previousLeft != null && block.getOrientation().equals(Orientation.LEFT)) {
|
||||
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft
|
||||
.getMinY()) {
|
||||
if (previousLeft.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousLeft.getMinY()) {
|
||||
previousLeft.add(block);
|
||||
itty.remove();
|
||||
continue;
|
||||
@@ -122,8 +127,7 @@ public class BlockificationService {
|
||||
}
|
||||
|
||||
if (previousRight != null && block.getOrientation().equals(Orientation.RIGHT)) {
|
||||
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight
|
||||
.getMinY()) {
|
||||
if (previousRight.getMinY() > block.getMinY() && block.getMaxY() + block.getMostPopularWordHeight() > previousRight.getMinY()) {
|
||||
previousRight.add(block);
|
||||
itty.remove();
|
||||
continue;
|
||||
@@ -142,10 +146,8 @@ public class BlockificationService {
|
||||
while (itty.hasNext()) {
|
||||
TextBlock block = (TextBlock) itty.next();
|
||||
|
||||
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
|
||||
.equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY()) || previous != null && previous
|
||||
.getOrientation()
|
||||
.equals(Orientation.LEFT) && block.getOrientation()
|
||||
if (previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation().equals(Orientation.LEFT) && equalsWithThreshold(block.getMaxY(),
|
||||
previous.getMaxY()) || previous != null && previous.getOrientation().equals(Orientation.LEFT) && block.getOrientation()
|
||||
.equals(Orientation.RIGHT) && equalsWithThreshold(block.getMaxY(), previous.getMaxY())) {
|
||||
previous.add(block);
|
||||
itty.remove();
|
||||
@@ -165,7 +167,7 @@ public class BlockificationService {
|
||||
}
|
||||
|
||||
|
||||
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList) {
|
||||
private TextBlock buildTextBlock(List<TextPositionSequence> wordBlockList, int indexOnPage) {
|
||||
|
||||
TextBlock textBlock = null;
|
||||
|
||||
@@ -184,12 +186,16 @@ public class BlockificationService {
|
||||
styleFrequencyCounter.add(wordBlock.getFontStyle());
|
||||
|
||||
if (textBlock == null) {
|
||||
textBlock = new TextBlock(wordBlock.getX1(), wordBlock.getX2(), wordBlock.getY1(), wordBlock.getY2(), wordBlockList, wordBlock
|
||||
.getRotation());
|
||||
textBlock = new TextBlock(wordBlock.getMinXDirAdj(),
|
||||
wordBlock.getMaxXDirAdj(),
|
||||
wordBlock.getMinYDirAdj(),
|
||||
wordBlock.getMaxYDirAdj(),
|
||||
wordBlockList,
|
||||
wordBlock.getRotation(),
|
||||
indexOnPage);
|
||||
} else {
|
||||
TextBlock spatialEntity = textBlock.union(wordBlock);
|
||||
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity
|
||||
.getHeight());
|
||||
textBlock.resize(spatialEntity.getMinX(), spatialEntity.getMinY(), spatialEntity.getWidth(), spatialEntity.getHeight());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -202,22 +208,61 @@ public class BlockificationService {
|
||||
textBlock.setHighestFontSize(fontSizeFrequencyCounter.getHighest());
|
||||
}
|
||||
|
||||
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences()
|
||||
.stream()
|
||||
.map(t -> round(t.getY1(), 3))
|
||||
.collect(toSet())
|
||||
.size() == 1) {
|
||||
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getX1));
|
||||
if (textBlock != null && textBlock.getSequences() != null && textBlock.getSequences().stream().map(t -> round(t.getMinYDirAdj(), 3)).collect(toSet()).size() == 1) {
|
||||
textBlock.getSequences().sort(Comparator.comparing(TextPositionSequence::getMinXDirAdj));
|
||||
}
|
||||
return textBlock;
|
||||
}
|
||||
|
||||
|
||||
private boolean isSplittedByRuling(float previousX2, float previousY1, float currentX1, float currentY1,
|
||||
List<Ruling> rulingLines) {
|
||||
private boolean isSplitByRuling(float minX,
|
||||
float minY,
|
||||
float maxX,
|
||||
float maxY,
|
||||
TextPositionSequence word,
|
||||
List<Ruling> horizontalRulingLines,
|
||||
List<Ruling> verticalRulingLines) {
|
||||
|
||||
return isSplitByRuling(maxX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMinYDirAdj(),
|
||||
verticalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMaxYDirAdj(),
|
||||
horizontalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(maxX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMinYDirAdj(),
|
||||
horizontalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()) //
|
||||
|| isSplitByRuling(minX,
|
||||
minY,
|
||||
word.getMinXDirAdj(),
|
||||
word.getMaxYDirAdj(),
|
||||
verticalRulingLines,
|
||||
word.getDir().getDegrees(),
|
||||
word.getPageWidth(),
|
||||
word.getPageHeight()); //
|
||||
}
|
||||
|
||||
|
||||
private boolean isSplitByRuling(float previousX2, float previousY1, float currentX1, float currentY1, List<Ruling> rulingLines, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
for (Ruling ruling : rulingLines) {
|
||||
if (ruling.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
|
||||
var line = RulingTextDirAdjustUtil.convertToDirAdj(ruling, dir, pageWidth, pageHeight);
|
||||
if (line.intersectsLine(previousX2, previousY1, currentX1, currentY1)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -225,104 +270,6 @@ public class BlockificationService {
|
||||
}
|
||||
|
||||
|
||||
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter,
|
||||
boolean landscape) {
|
||||
|
||||
float minX = 10000;
|
||||
float maxX = -100;
|
||||
float minY = 10000;
|
||||
float maxY = -100;
|
||||
|
||||
for (Page page : pages) {
|
||||
|
||||
if (page.getTextBlocks().isEmpty() || landscape != page.isLandscape()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (AbstractTextContainer container : page.getTextBlocks()) {
|
||||
|
||||
if (container instanceof TextBlock) {
|
||||
TextBlock textBlock = (TextBlock) container;
|
||||
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
|
||||
if (approxLineCount < 2.9f) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (documentFontSizeCounter.getMostPopular() != null) {
|
||||
if (textBlock.getMostPopularWordFontSize() >= documentFontSizeCounter.getMostPopular()) {
|
||||
|
||||
if (textBlock.getMinX() < minX) {
|
||||
minX = textBlock.getMinX();
|
||||
}
|
||||
if (textBlock.getMaxX() > maxX) {
|
||||
maxX = textBlock.getMaxX();
|
||||
}
|
||||
if (textBlock.getMinY() < minY) {
|
||||
minY = textBlock.getMinY();
|
||||
}
|
||||
if (textBlock.getMaxY() > maxY) {
|
||||
maxY = textBlock.getMaxY();
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (container instanceof Table) {
|
||||
Table table = (Table) container;
|
||||
for (List<Cell> row : table.getRows()) {
|
||||
for (Cell cell : row) {
|
||||
|
||||
if (cell == null || cell.getTextBlocks() == null) {
|
||||
continue;
|
||||
}
|
||||
for (TextBlock textBlock : cell.getTextBlocks()) {
|
||||
if (textBlock.getMinX() < minX) {
|
||||
minX = textBlock.getMinX();
|
||||
}
|
||||
if (textBlock.getMaxX() > maxX) {
|
||||
maxX = textBlock.getMaxX();
|
||||
}
|
||||
if (textBlock.getMinY() < minY) {
|
||||
minY = textBlock.getMinY();
|
||||
}
|
||||
if (textBlock.getMaxY() > maxY) {
|
||||
maxY = textBlock.getMaxY();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
return new Rectangle(minY, minX, maxX - minX, maxY - minY);
|
||||
}
|
||||
|
||||
|
||||
private void sortRotatedSequences(List<TextPositionSequence> sequences) {
|
||||
|
||||
List<TextPositionSequence> rotatedWords = new ArrayList<>();
|
||||
Iterator<TextPositionSequence> itty = sequences.iterator();
|
||||
while (itty.hasNext()) {
|
||||
var pos = itty.next();
|
||||
if (pos.getTextPositions().get(0).getDir() == 270) {
|
||||
rotatedWords.add(pos);
|
||||
itty.remove();
|
||||
}
|
||||
}
|
||||
|
||||
if (!rotatedWords.isEmpty() && !sequences.isEmpty()) {
|
||||
rotatedWords.sort(Comparator.comparing(TextPositionSequence::getX1));
|
||||
}
|
||||
sequences.addAll(rotatedWords);
|
||||
}
|
||||
|
||||
|
||||
private double round(float value, int decimalPoints) {
|
||||
|
||||
var d = Math.pow(10, decimalPoints);
|
||||
|
||||
+161
@@ -0,0 +1,161 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.service;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Point;
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.FloatFrequencyCounter;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Cell;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
@Service
|
||||
public class BodyTextFrameService {
|
||||
|
||||
/**
|
||||
* Adjusts and sets the body text frame to a page.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
* The aspect ratio of the page is also regarded.
|
||||
*
|
||||
* @param page The page
|
||||
* @param bodyTextFrame frame that contains the main text on portrait pages
|
||||
* @param landscapeBodyTextFrame frame that contains the main text on landscape pages
|
||||
*/
|
||||
public void setBodyTextFrameAdjustedToPage(Page page, Rectangle bodyTextFrame, Rectangle landscapeBodyTextFrame) {
|
||||
|
||||
Rectangle textFrame = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
|
||||
|
||||
if (page.getPageWidth() > page.getPageHeight() && page.getRotation() == 270) {
|
||||
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getY(), page.getPageHeight() - textFrame.getTopLeft().getX() - textFrame.getWidth()),
|
||||
textFrame.getHeight(),
|
||||
textFrame.getWidth(),
|
||||
0);
|
||||
} else if (page.getPageWidth() > page.getPageHeight() && page.getRotation() != 0) {
|
||||
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getY(), textFrame.getTopLeft().getX()), textFrame.getHeight(), textFrame.getWidth(), page.getPageNumber());
|
||||
} else if (page.getRotation() == 180) {
|
||||
textFrame = new Rectangle(new Point(textFrame.getTopLeft().getX(), page.getPageHeight() - textFrame.getTopLeft().getY() - textFrame.getHeight()),
|
||||
textFrame.getWidth(),
|
||||
textFrame.getHeight(),
|
||||
0);
|
||||
}
|
||||
page.setBodyTextFrame(textFrame);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Calculates the frame that contains the main text, text outside the frame will be e.g. headers or footers.
|
||||
* Note: This needs to use Pdf Coordinate System where {0,0} rotated with the page rotation.
|
||||
* 0 -> LowerLeft
|
||||
* 90 -> UpperLeft
|
||||
* 180 -> UpperRight
|
||||
* 270 -> LowerRight
|
||||
* The aspect ratio of the page is also regarded.
|
||||
*
|
||||
* @param pages List of all pages
|
||||
* @param documentFontSizeCounter Statistics of the document
|
||||
* @param landscape Calculate for landscape or portrait
|
||||
* @return Rectangle of the text frame
|
||||
*/
|
||||
public Rectangle calculateBodyTextFrame(List<Page> pages, FloatFrequencyCounter documentFontSizeCounter, boolean landscape) {
|
||||
|
||||
BodyTextFrameExpansionsRectangle expansionsRectangle = new BodyTextFrameExpansionsRectangle();
|
||||
|
||||
for (Page page : pages) {
|
||||
|
||||
if (page.getTextBlocks().isEmpty() || landscape != page.isLandscape()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (AbstractTextContainer container : page.getTextBlocks()) {
|
||||
|
||||
if (container instanceof TextBlock) {
|
||||
TextBlock textBlock = (TextBlock) container;
|
||||
if (textBlock.getMostPopularWordFont() == null || textBlock.getMostPopularWordStyle() == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
float approxLineCount = PositionUtils.getApproxLineCount(textBlock);
|
||||
if (approxLineCount < 2.9f) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (documentFontSizeCounter.getMostPopular() != null && textBlock.getMostPopularWordFontSize() >= documentFontSizeCounter.getMostPopular()) {
|
||||
|
||||
expandRectangle(textBlock, page, expansionsRectangle);
|
||||
}
|
||||
}
|
||||
|
||||
if (container instanceof Table) {
|
||||
Table table = (Table) container;
|
||||
for (List<Cell> row : table.getRows()) {
|
||||
for (Cell cell : row) {
|
||||
|
||||
if (cell == null || cell.getTextBlocks() == null) {
|
||||
continue;
|
||||
}
|
||||
for (TextBlock textBlock : cell.getTextBlocks()) {
|
||||
expandRectangle(textBlock, page, expansionsRectangle);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return new Rectangle(new Point(expansionsRectangle.minX, expansionsRectangle.minY),
|
||||
expansionsRectangle.maxX - expansionsRectangle.minX,
|
||||
expansionsRectangle.maxY - expansionsRectangle.minY,
|
||||
0);
|
||||
}
|
||||
|
||||
|
||||
private void expandRectangle(TextBlock textBlock, Page page, BodyTextFrameExpansionsRectangle expansionsRectangle) {
|
||||
|
||||
if (page.getPageWidth() > page.getPageHeight() && page.getRotation() != 0) {
|
||||
if (textBlock.getPdfMinY() < expansionsRectangle.minX) {
|
||||
expansionsRectangle.minX = textBlock.getPdfMinY();
|
||||
}
|
||||
if (textBlock.getPdfMaxY() > expansionsRectangle.maxX) {
|
||||
expansionsRectangle.maxX = textBlock.getPdfMaxY();
|
||||
}
|
||||
if (textBlock.getPdfMinX() < expansionsRectangle.minY) {
|
||||
expansionsRectangle.minY = textBlock.getPdfMinX();
|
||||
}
|
||||
if (textBlock.getPdfMaxX() > expansionsRectangle.maxY) {
|
||||
expansionsRectangle.maxY = textBlock.getPdfMaxX();
|
||||
}
|
||||
} else {
|
||||
if (textBlock.getPdfMinX() < expansionsRectangle.minX) {
|
||||
expansionsRectangle.minX = textBlock.getPdfMinX();
|
||||
}
|
||||
if (textBlock.getPdfMaxX() > expansionsRectangle.maxX) {
|
||||
expansionsRectangle.maxX = textBlock.getPdfMaxX();
|
||||
}
|
||||
if (textBlock.getPdfMinY() < expansionsRectangle.minY) {
|
||||
expansionsRectangle.minY = textBlock.getPdfMinY();
|
||||
}
|
||||
if (textBlock.getPdfMaxY() > expansionsRectangle.maxY) {
|
||||
expansionsRectangle.maxY = textBlock.getPdfMaxY();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private class BodyTextFrameExpansionsRectangle {
|
||||
|
||||
float minX = 10000;
|
||||
float maxX = -100;
|
||||
float minY = 10000;
|
||||
float maxY = -100;
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+45
-48
@@ -1,82 +1,82 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.service;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.utils.PositionUtils;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class ClassificationService {
|
||||
|
||||
private final BlockificationService blockificationService;
|
||||
private final BodyTextFrameService bodyTextFrameService;
|
||||
|
||||
|
||||
public void classifyDocument(Document document) {
|
||||
|
||||
Rectangle bodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false);
|
||||
Rectangle landscapeBodyTextFrame = blockificationService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true);
|
||||
|
||||
Rectangle bodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), false);
|
||||
Rectangle landscapeBodyTextFrame = bodyTextFrameService.calculateBodyTextFrame(document.getPages(), document.getFontSizeCounter(), true);
|
||||
List<Float> headlineFontSizes = document.getFontSizeCounter().getHighterThanMostPopular();
|
||||
|
||||
log.debug("Document FontSize counters are: {}", document.getFontSizeCounter().getCountPerValue());
|
||||
|
||||
for (Page page : document.getPages()) {
|
||||
Rectangle btf = page.isLandscape() ? landscapeBodyTextFrame : bodyTextFrame;
|
||||
page.setBodyTextFrame(btf);
|
||||
classifyPage(btf, page, document, headlineFontSizes);
|
||||
bodyTextFrameService.setBodyTextFrameAdjustedToPage(page, bodyTextFrame, landscapeBodyTextFrame);
|
||||
classifyPage(page, document, headlineFontSizes);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public void classifyPage(Rectangle bodyTextFrame, Page page, Document document, List<Float> headlineFontSizes) {
|
||||
public void classifyPage(Page page, Document document, List<Float> headlineFontSizes) {
|
||||
|
||||
for (AbstractTextContainer textBlock : page.getTextBlocks()) {
|
||||
if (textBlock instanceof TextBlock) {
|
||||
classifyBlock((TextBlock) textBlock, bodyTextFrame, page, document, headlineFontSizes);
|
||||
classifyBlock((TextBlock) textBlock, page, document, headlineFontSizes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public void classifyBlock(TextBlock textBlock, Rectangle bodyTextFrame, Page page, Document document,
|
||||
List<Float> headlineFontSizes) {
|
||||
public void classifyBlock(TextBlock textBlock, Page page, Document document, List<Float> headlineFontSizes) {
|
||||
|
||||
var bodyTextFrame = page.getBodyTextFrame();
|
||||
|
||||
if (document.getFontSizeCounter().getMostPopular() == null) {
|
||||
// TODO Figure out why this happens.
|
||||
textBlock.setClassification("Other");
|
||||
return;
|
||||
}
|
||||
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.isRotated()) && (document.getFontSizeCounter()
|
||||
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
|
||||
.getMostPopular())) {
|
||||
if (PositionUtils.isOverBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
|
||||
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
|
||||
textBlock.setClassification("Header");
|
||||
|
||||
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock) && (document.getFontSizeCounter()
|
||||
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter()
|
||||
.getMostPopular())) {
|
||||
} else if (PositionUtils.isUnderBodyTextFrame(bodyTextFrame, textBlock, page.getRotation()) && (document.getFontSizeCounter()
|
||||
.getMostPopular() == null || textBlock.getHighestFontSize() <= document.getFontSizeCounter().getMostPopular())) {
|
||||
textBlock.setClassification("Footer");
|
||||
} else if (page.getPageNumber() == 1 && (!PositionUtils.isTouchingUnderBodyTextFrame(bodyTextFrame, textBlock) && PositionUtils
|
||||
.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock, document.getTextHeightCounter()
|
||||
.getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter()
|
||||
.getMostPopular() || page.getTextBlocks().size() == 1)) {
|
||||
} else if (page.getPageNumber() == 1 && (PositionUtils.getHeightDifferenceBetweenChunkWordAndDocumentWord(textBlock,
|
||||
document.getTextHeightCounter().getMostPopular()) > 2.5 && textBlock.getHighestFontSize() > document.getFontSizeCounter().getMostPopular() || page.getTextBlocks()
|
||||
.size() == 1)) {
|
||||
if (!Pattern.matches("[0-9]+", textBlock.toString())) {
|
||||
textBlock.setClassification("Title");
|
||||
}
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() > document
|
||||
.getFontSizeCounter()
|
||||
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle()
|
||||
.equals("bold") || !document.getFontStyleCounter().getCountPerValue().containsKey("bold") && textBlock.getMostPopularWordFontSize() > document
|
||||
.getFontSizeCounter()
|
||||
.getMostPopular() + 1) && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
|
||||
} else if (textBlock.getMostPopularWordFontSize() > document.getFontSizeCounter()
|
||||
.getMostPopular() && PositionUtils.getApproxLineCount(textBlock) < 4.9 && (textBlock.getMostPopularWordStyle().equals("bold") || !document.getFontStyleCounter()
|
||||
.getCountPerValue()
|
||||
.containsKey("bold") && textBlock.getMostPopularWordFontSize() > document.getFontSizeCounter().getMostPopular() + 1) && textBlock.getSequences()
|
||||
.get(0)
|
||||
.getTextPositions()
|
||||
.get(0)
|
||||
.getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
|
||||
|
||||
for (int i = 1; i <= headlineFontSizes.size(); i++) {
|
||||
if (textBlock.getMostPopularWordFontSize() == headlineFontSizes.get(i - 1)) {
|
||||
@@ -84,28 +84,25 @@ public class ClassificationService {
|
||||
document.setHeadlines(true);
|
||||
}
|
||||
}
|
||||
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText()
|
||||
.startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordStyle()
|
||||
.equals("bold") && !document.getFontStyleCounter()
|
||||
} else if (!textBlock.getText().startsWith("Table ") && !textBlock.getText().startsWith("Figure ") && PositionUtils.isWithinBodyTextFrame(bodyTextFrame,
|
||||
textBlock) && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter()
|
||||
.getMostPopular()
|
||||
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences().get(0).getTextPositions().get(0).getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
|
||||
.equals("bold") && PositionUtils.getApproxLineCount(textBlock) < 2.9 && textBlock.getSequences()
|
||||
.get(0)
|
||||
.getTextPositions()
|
||||
.get(0)
|
||||
.getFontSizeInPt() >= textBlock.getMostPopularWordFontSize()) {
|
||||
textBlock.setClassification("H " + (headlineFontSizes.size() + 1));
|
||||
document.setHeadlines(true);
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
|
||||
.getFontSizeCounter()
|
||||
.getMostPopular() && textBlock.getMostPopularWordStyle()
|
||||
.equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
|
||||
.getMostPopular() && textBlock.getMostPopularWordStyle().equals("bold") && !document.getFontStyleCounter().getMostPopular().equals("bold")) {
|
||||
textBlock.setClassification("TextBlock Bold");
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFont()
|
||||
.equals(document.getFontCounter().getMostPopular()) && textBlock.getMostPopularWordStyle()
|
||||
.equals(document.getFontStyleCounter()
|
||||
.getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
|
||||
.getMostPopular()) {
|
||||
.equals(document.getFontStyleCounter().getMostPopular()) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter().getMostPopular()) {
|
||||
textBlock.setClassification("TextBlock");
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document
|
||||
.getFontSizeCounter()
|
||||
.getMostPopular() && textBlock.getMostPopularWordStyle()
|
||||
.equals("italic") && !document.getFontStyleCounter()
|
||||
} else if (PositionUtils.isWithinBodyTextFrame(bodyTextFrame, textBlock) && textBlock.getMostPopularWordFontSize() == document.getFontSizeCounter()
|
||||
.getMostPopular() && textBlock.getMostPopularWordStyle().equals("italic") && !document.getFontStyleCounter()
|
||||
.getMostPopular()
|
||||
.equals("italic") && PositionUtils.getApproxLineCount(textBlock) < 2.9) {
|
||||
textBlock.setClassification("TextBlock Italic");
|
||||
|
||||
+45
-20
@@ -1,28 +1,27 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.utils;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.Rectangle;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Rectangle;
|
||||
|
||||
import lombok.experimental.UtilityClass;
|
||||
|
||||
@UtilityClass
|
||||
@SuppressWarnings("all")
|
||||
public class PositionUtils {
|
||||
|
||||
public final class PositionUtils {
|
||||
|
||||
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
|
||||
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
|
||||
public boolean isWithinBodyTextFrame(Rectangle btf, TextBlock textBlock) {
|
||||
|
||||
//TODO Currently this is not working for rotated pages.
|
||||
|
||||
if (btf == null || textBlock == null) {
|
||||
return false;
|
||||
}
|
||||
|
||||
double threshold = textBlock.getMostPopularWordHeight() * 3;
|
||||
|
||||
if (textBlock.getMinX() + threshold > btf.getX() &&
|
||||
textBlock.getMaxX() - threshold < btf.getX() + btf.getWidth() &&
|
||||
textBlock.getMinY() + threshold > btf.getY() &&
|
||||
textBlock.getMaxY() - threshold < btf.getY() + btf.getHeight()) {
|
||||
if (textBlock.getPdfMinX() + threshold > btf.getTopLeft().getX() && textBlock.getPdfMaxX() - threshold < btf.getTopLeft()
|
||||
.getX() + btf.getWidth() && textBlock.getPdfMinY() + threshold > btf.getTopLeft().getY() && textBlock.getPdfMaxY() - threshold < btf.getTopLeft()
|
||||
.getY() + btf.getHeight()) {
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
@@ -31,16 +30,27 @@ public class PositionUtils {
|
||||
}
|
||||
|
||||
|
||||
public boolean isOverBodyTextFrame(Rectangle btf, TextBlock textBlock, boolean rotated) {
|
||||
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
|
||||
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
|
||||
public boolean isOverBodyTextFrame(Rectangle btf, TextBlock textBlock, int rotation) {
|
||||
|
||||
if (btf == null || textBlock == null) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (rotated && textBlock.getMinX() < btf.getX()) {
|
||||
// Its very strange, P{0,0} is on top left in this case, instead of lower left.
|
||||
if (rotation == 90 && textBlock.getPdfMaxX() < btf.getTopLeft().getX()) {
|
||||
return true;
|
||||
} else if (!rotated && textBlock.getMinY() > btf.getY() + btf.getHeight()) {
|
||||
}
|
||||
|
||||
if (rotation == 180 && textBlock.getPdfMaxY() < btf.getTopLeft().getY()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (rotation == 270 && textBlock.getPdfMinX() > btf.getTopLeft().getX() + btf.getWidth()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (rotation == 0 && textBlock.getPdfMinY() > btf.getTopLeft().getY() + btf.getHeight()) {
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
@@ -48,16 +58,27 @@ public class PositionUtils {
|
||||
|
||||
}
|
||||
|
||||
|
||||
public boolean isUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
|
||||
|
||||
//TODO Currently this is not working for rotated pages.
|
||||
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
|
||||
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
|
||||
public boolean isUnderBodyTextFrame(Rectangle btf, TextBlock textBlock, int rotation) {
|
||||
|
||||
if (btf == null || textBlock == null) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (textBlock.getMaxY() < btf.getY()) {
|
||||
if (rotation == 90 && textBlock.getPdfMinX() > btf.getTopLeft().getX() + btf.getWidth()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (rotation == 180 && textBlock.getPdfMinY() > btf.getTopLeft().getY() + btf.getHeight()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (rotation == 270 && textBlock.getPdfMaxX() < btf.getTopLeft().getX()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (rotation == 0 && textBlock.getPdfMaxY() < btf.getTopLeft().getY()) {
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
@@ -65,7 +86,8 @@ public class PositionUtils {
|
||||
|
||||
}
|
||||
|
||||
|
||||
// TODO This currently uses pdf coord system. In the futher this should use java coord system.
|
||||
// Note: DirAdj (TextDirection Adjusted) can not be user for this.
|
||||
public boolean isTouchingUnderBodyTextFrame(Rectangle btf, TextBlock textBlock) {
|
||||
|
||||
//TODO Currently this is not working for rotated pages.
|
||||
@@ -74,7 +96,7 @@ public class PositionUtils {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (textBlock.getMinY() < btf.getY()) {
|
||||
if (textBlock.getMinY() < btf.getTopLeft().getY()) {
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
@@ -84,11 +106,14 @@ public class PositionUtils {
|
||||
|
||||
|
||||
public float getHeightDifferenceBetweenChunkWordAndDocumentWord(TextBlock textBlock, Float documentMostPopularWordHeight) {
|
||||
|
||||
return textBlock.getMostPopularWordHeight() - documentMostPopularWordHeight;
|
||||
}
|
||||
|
||||
|
||||
public Float getApproxLineCount(TextBlock textBlock) {
|
||||
|
||||
return textBlock.getHeight() / textBlock.getMostPopularWordHeight();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+67
@@ -0,0 +1,67 @@
|
||||
package com.iqser.red.service.redaction.v1.server.classification.utils;
|
||||
|
||||
import java.awt.geom.Line2D;
|
||||
import java.awt.geom.Point2D;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
|
||||
|
||||
import lombok.experimental.UtilityClass;
|
||||
|
||||
@UtilityClass
|
||||
public final class RulingTextDirAdjustUtil {
|
||||
|
||||
/**
|
||||
* Converts a ruling (line of a table) the same way TextPositions are converted in PDFBox.
|
||||
* This will get the y position of the text, adjusted so that 0,0 is upper left and it is adjusted based on the text direction.
|
||||
*
|
||||
* See org.apache.pdfbox.text.TextPosition
|
||||
*/
|
||||
public Line2D.Float convertToDirAdj(Ruling ruling, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
return new Line2D.Float(convertPoint(ruling.x1, ruling.y1, dir, pageWidth, pageHeight), convertPoint(ruling.x2, ruling.y2, dir, pageWidth, pageHeight));
|
||||
}
|
||||
|
||||
|
||||
private Point2D convertPoint(float x, float y, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
var xAdj = getXRot(x, y, dir, pageWidth, pageHeight);
|
||||
var yAdj = 0f;
|
||||
if (dir == 0 || dir == 180) {
|
||||
yAdj = pageHeight - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
|
||||
} else {
|
||||
yAdj = pageWidth - getYLowerLeftRot(x, y, dir, pageWidth, pageHeight);
|
||||
}
|
||||
return new Point2D.Float(xAdj, yAdj);
|
||||
}
|
||||
|
||||
|
||||
private float getXRot(float x, float y, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
if (dir == 0) {
|
||||
return x;
|
||||
} else if (dir == 90) {
|
||||
return y;
|
||||
} else if (dir == 180) {
|
||||
return pageWidth - x;
|
||||
} else if (dir == 270) {
|
||||
return pageHeight - y;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
private float getYLowerLeftRot(float x, float y, float dir, float pageWidth, float pageHeight) {
|
||||
|
||||
if (dir == 0) {
|
||||
return y;
|
||||
} else if (dir == 90) {
|
||||
return pageWidth - x;
|
||||
} else if (dir == 180) {
|
||||
return pageHeight - y;
|
||||
} else if (dir == 270) {
|
||||
return x;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
}
|
||||
+1
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.DictionaryReso
|
||||
|
||||
@FeignClient(name = "DictionaryResource", url = "${persistence-service.url}")
|
||||
public interface DictionaryClient extends DictionaryResource {
|
||||
|
||||
}
|
||||
+1
-1
@@ -1,10 +1,10 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
|
||||
import org.springframework.cloud.openfeign.FeignClient;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.resources.FileStatusProcessingUpdateResource;
|
||||
|
||||
@FeignClient(name = "FileStatusProcessingUpdateResource", url = "${persistence-service.url}")
|
||||
public interface FileStatusProcessingUpdateClient extends FileStatusProcessingUpdateResource {
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.LegalBasisMapp
|
||||
|
||||
@FeignClient(name = "LegalBasisMappingResource", url = "${persistence-service.url}")
|
||||
public interface LegalBasisClient extends LegalBasisMappingResource {
|
||||
|
||||
}
|
||||
|
||||
+9
-11
@@ -1,16 +1,16 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
|
||||
import org.springframework.lang.NonNull;
|
||||
import org.springframework.lang.Nullable;
|
||||
import org.springframework.util.Assert;
|
||||
import org.springframework.util.FileCopyUtils;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
|
||||
public class MockMultipartFile implements MultipartFile {
|
||||
|
||||
private final String name;
|
||||
@@ -32,8 +32,7 @@ public class MockMultipartFile implements MultipartFile {
|
||||
}
|
||||
|
||||
|
||||
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
|
||||
@Nullable byte[] content) {
|
||||
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, @Nullable byte[] content) {
|
||||
|
||||
Assert.hasLength(name, "Name must not be empty");
|
||||
this.name = name;
|
||||
@@ -43,8 +42,7 @@ public class MockMultipartFile implements MultipartFile {
|
||||
}
|
||||
|
||||
|
||||
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType,
|
||||
InputStream contentStream) throws IOException {
|
||||
public MockMultipartFile(String name, @Nullable String originalFilename, @Nullable String contentType, InputStream contentStream) throws IOException {
|
||||
|
||||
this(name, originalFilename, contentType, FileCopyUtils.copyToByteArray(contentStream));
|
||||
}
|
||||
@@ -82,13 +80,13 @@ public class MockMultipartFile implements MultipartFile {
|
||||
}
|
||||
|
||||
|
||||
public byte[] getBytes() throws IOException {
|
||||
public byte[] getBytes() {
|
||||
|
||||
return this.content;
|
||||
}
|
||||
|
||||
|
||||
public InputStream getInputStream() throws IOException {
|
||||
public InputStream getInputStream() {
|
||||
|
||||
return new ByteArrayInputStream(this.content);
|
||||
}
|
||||
|
||||
+1
@@ -6,4 +6,5 @@ import com.iqser.red.service.persistence.service.v1.api.resources.RulesResource;
|
||||
|
||||
@FeignClient(name = "RulesResource", url = "${persistence-service.url}")
|
||||
public interface RulesClient extends RulesResource {
|
||||
|
||||
}
|
||||
|
||||
+7
-4
@@ -1,5 +1,7 @@
|
||||
package com.iqser.red.service.redaction.v1.server.client.model;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
@@ -7,13 +9,14 @@ import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@CompiledJson
|
||||
@AllArgsConstructor
|
||||
@NoArgsConstructor
|
||||
public class EntityRecogintionEntity {
|
||||
|
||||
private String value;
|
||||
private int startOffset;
|
||||
private int endOffset;
|
||||
private String type;
|
||||
private String value;
|
||||
private int startOffset;
|
||||
private int endOffset;
|
||||
private String type;
|
||||
|
||||
}
|
||||
|
||||
+1
-1
@@ -13,6 +13,6 @@ import lombok.NoArgsConstructor;
|
||||
@NoArgsConstructor
|
||||
public class EntityRecognitionRequest {
|
||||
|
||||
private List<EntityRecognitionSection> data;
|
||||
private List<EntityRecognitionSection> data;
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -17,4 +17,5 @@ public class EntityRecognitionResult {
|
||||
|
||||
@Builder.Default
|
||||
private Map<Integer, List<EntityRecogintionEntity>> entities = new HashMap<>();
|
||||
|
||||
}
|
||||
|
||||
+1
@@ -13,4 +13,5 @@ public class EntityRecognitionSection {
|
||||
|
||||
private int sectionNumber;
|
||||
private String text;
|
||||
|
||||
}
|
||||
|
||||
+5
-5
@@ -4,18 +4,18 @@ import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import com.dslplatform.json.CompiledJson;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@CompiledJson
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class NerEntities {
|
||||
|
||||
@Builder.Default
|
||||
private Map<Integer, List<EntityRecogintionEntity>> result = new HashMap<>();
|
||||
private Map<Integer, List<EntityRecogintionEntity>> data = new HashMap<>();
|
||||
|
||||
}
|
||||
|
||||
+8
@@ -3,7 +3,9 @@ package com.iqser.red.service.redaction.v1.server.controller;
|
||||
import com.iqser.red.commons.spring.ErrorMessage;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.web.bind.annotation.ExceptionHandler;
|
||||
import org.springframework.web.bind.annotation.ResponseBody;
|
||||
@@ -18,10 +20,12 @@ public class ControllerAdvice {
|
||||
|
||||
/* error handling */
|
||||
|
||||
|
||||
@ResponseBody
|
||||
@ResponseStatus(value = HttpStatus.INTERNAL_SERVER_ERROR)
|
||||
@ExceptionHandler(value = NullPointerException.class)
|
||||
public ErrorMessage handleContentNotFoundException(NullPointerException e) {
|
||||
|
||||
if (e != null) {
|
||||
log.error(e.getMessage(), e);
|
||||
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
|
||||
@@ -30,17 +34,21 @@ public class ControllerAdvice {
|
||||
return new ErrorMessage(OffsetDateTime.now(), "Nullpointer exception");
|
||||
}
|
||||
|
||||
|
||||
@ResponseBody
|
||||
@ResponseStatus(value = HttpStatus.BAD_REQUEST)
|
||||
@ExceptionHandler(value = RulesValidationException.class)
|
||||
public ErrorMessage handleRulesValidationException(RulesValidationException e) {
|
||||
|
||||
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
|
||||
}
|
||||
|
||||
|
||||
@ResponseBody
|
||||
@ResponseStatus(value = HttpStatus.NOT_FOUND)
|
||||
@ExceptionHandler(value = NotFoundException.class)
|
||||
public ErrorMessage handleFileNotFoundException(NotFoundException e) {
|
||||
|
||||
return new ErrorMessage(OffsetDateTime.now(), e.getMessage());
|
||||
}
|
||||
|
||||
|
||||
+27
-85
@@ -2,9 +2,7 @@ package com.iqser.red.service.redaction.v1.server.controller;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.apache.pdfbox.io.MemoryUsageSetting;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.web.bind.annotation.PathVariable;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
@@ -12,20 +10,14 @@ import org.springframework.web.bind.annotation.RestController;
|
||||
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.annotations.ManualRedactions;
|
||||
import com.iqser.red.service.persistence.service.v1.api.model.dossiertemplate.dossier.file.FileType;
|
||||
import com.iqser.red.service.redaction.v1.model.AnnotateRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.AnnotateResponse;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionLog;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionRequest;
|
||||
import com.iqser.red.service.redaction.v1.model.RedactionResult;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionArea;
|
||||
import com.iqser.red.service.redaction.v1.model.SectionGrid;
|
||||
import com.iqser.red.service.redaction.v1.resources.RedactionResource;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RedactionException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.AnnotationService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DictionaryService;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.RulesValidationException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.DroolsExecutionService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.ManualRedactionSurroundingTextService;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.RedactionLogMergeService;
|
||||
@@ -45,50 +37,24 @@ public class RedactionController implements RedactionResource {
|
||||
|
||||
private final PdfVisualisationService pdfVisualisationService;
|
||||
private final DroolsExecutionService droolsExecutionService;
|
||||
private final DictionaryService dictionaryService;
|
||||
private final AnnotationService annotationService;
|
||||
private final PdfSegmentationService pdfSegmentationService;
|
||||
private final RedactionStorageService redactionStorageService;
|
||||
private final RedactionLogMergeService redactionLogMergeService;
|
||||
private final ManualRedactionSurroundingTextService manualRedactionSurroundingTextService;
|
||||
|
||||
|
||||
public AnnotateResponse annotate(@RequestBody AnnotateRequest annotateRequest) {
|
||||
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(annotateRequest.getDossierId(), annotateRequest.getFileId(), FileType.ORIGIN));
|
||||
var mergedRedactionLog = getRedactionLog(RedactionRequest.builder()
|
||||
.fileId(annotateRequest.getFileId())
|
||||
.manualRedactions(annotateRequest.getManualRedactions())
|
||||
.dossierId(annotateRequest.getDossierId())
|
||||
.dossierTemplateId(annotateRequest.getDossierTemplateId())
|
||||
.build());
|
||||
var sectionsGrid = redactionStorageService.getSectionGrid(annotateRequest.getDossierId(), annotateRequest.getFileId());
|
||||
|
||||
try (PDDocument pdDocument = PDDocument.load(storedObjectStream, MemoryUsageSetting.setupTempFileOnly())) {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
dictionaryService.updateDictionary(annotateRequest.getDossierTemplateId(), annotateRequest.getDossierId());
|
||||
annotationService.annotate(pdDocument, mergedRedactionLog, sectionsGrid);
|
||||
|
||||
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
|
||||
pdDocument.save(byteArrayOutputStream);
|
||||
return AnnotateResponse.builder().document(byteArrayOutputStream.toByteArray()).build();
|
||||
}
|
||||
|
||||
} catch (Exception e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionResult classify(@RequestBody RedactionRequest redactionRequest) {
|
||||
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
@@ -110,11 +76,15 @@ public class RedactionController implements RedactionResource {
|
||||
@Override
|
||||
public RedactionResult sections(@RequestBody RedactionRequest redactionRequest) {
|
||||
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try {
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
|
||||
Document classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
|
||||
storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
try (PDDocument pdDocument = PDDocument.load(storedObjectStream)) {
|
||||
pdDocument.setAllSecurityToBeRemoved(true);
|
||||
|
||||
@@ -138,8 +108,10 @@ public class RedactionController implements RedactionResource {
|
||||
Document classifiedDoc;
|
||||
|
||||
try {
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.ORIGIN));
|
||||
classifiedDoc = pdfSegmentationService.parseDocument(storedObjectStream, null);
|
||||
var storedObjectStream = redactionStorageService.getStoredObject(RedactionStorageService.StorageIdUtils.getStorageId(redactionRequest.getDossierId(),
|
||||
redactionRequest.getFileId(),
|
||||
FileType.ORIGIN));
|
||||
classifiedDoc = pdfSegmentationService.parseDocument(redactionRequest.getDossierId(), redactionRequest.getFileId(), storedObjectStream, null);
|
||||
} catch (Exception e) {
|
||||
throw new RedactionException(e);
|
||||
}
|
||||
@@ -162,43 +134,18 @@ public class RedactionController implements RedactionResource {
|
||||
@Override
|
||||
public void testRules(@RequestBody String rules) {
|
||||
|
||||
droolsExecutionService.testRules(rules);
|
||||
try {
|
||||
droolsExecutionService.testRules(rules);
|
||||
} catch (Exception e) {
|
||||
throw new RulesValidationException("Could not test rules: " + e.getMessage(), e);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public RedactionLog getRedactionLog(RedactionRequest redactionRequest) {
|
||||
|
||||
log.debug("Requested preview for: {}", redactionRequest);
|
||||
dictionaryService.updateDictionary(redactionRequest.getDossierTemplateId(), redactionRequest.getDossierId());
|
||||
|
||||
var redactionLog = redactionStorageService.getRedactionLog(redactionRequest.getDossierId(), redactionRequest.getFileId());
|
||||
|
||||
if (redactionLog == null) {
|
||||
throw new NotFoundException("RedactionLog not present");
|
||||
}
|
||||
|
||||
log.info("Loaded redaction log with computationalVersion: {}", redactionLog.getAnalysisVersion());
|
||||
|
||||
SectionGrid sectionGrid = redactionStorageService.getSectionGrid(redactionRequest.getDossierId(), redactionRequest.getFileId());
|
||||
if (sectionGrid.getSections().isEmpty()) {
|
||||
|
||||
log.info("SectionGrid does not have headlines set. Computing headlines now!");
|
||||
var text = redactionStorageService.getText(redactionRequest.getDossierId(), redactionRequest.getFileId());
|
||||
|
||||
// enhance section grid with headline data
|
||||
for (var sectionText : text.getSectionTexts()) {
|
||||
sectionGrid.getSections()
|
||||
.add(new SectionGrid.SectionGridSection(sectionText.getSectionNumber(), sectionText.getHeadline(), sectionText.getSectionAreas()
|
||||
.stream()
|
||||
.map(SectionArea::getPage)
|
||||
.collect(Collectors.toSet()), sectionText.getSectionAreas()));
|
||||
}
|
||||
redactionStorageService.storeObject(redactionRequest.getDossierId(), redactionRequest.getFileId(), FileType.SECTION_GRID, sectionGrid);
|
||||
}
|
||||
|
||||
log.info("Loaded redaction log with computationalVersion: {}", redactionLog.getAnalysisVersion());
|
||||
return redactionLogMergeService.mergeRedactionLogData(redactionLog, sectionGrid, redactionRequest.getDossierTemplateId(), redactionRequest.getManualRedactions(), redactionRequest.getExcludedPages());
|
||||
return redactionLogMergeService.provideRedactionLog(redactionRequest);
|
||||
}
|
||||
|
||||
|
||||
@@ -206,19 +153,14 @@ public class RedactionController implements RedactionResource {
|
||||
|
||||
try (ByteArrayOutputStream byteArrayOutputStream = new ByteArrayOutputStream()) {
|
||||
document.save(byteArrayOutputStream);
|
||||
return RedactionResult.builder()
|
||||
.document(byteArrayOutputStream.toByteArray())
|
||||
.numberOfPages(numberOfPages)
|
||||
.build();
|
||||
return RedactionResult.builder().document(byteArrayOutputStream.toByteArray()).numberOfPages(numberOfPages).build();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId,
|
||||
@PathVariable("fileId") String fileId,
|
||||
@RequestBody ManualRedactions manualRedactions) {
|
||||
public ManualRedactions addSurroundingText(@PathVariable("dossierId") String dossierId, @PathVariable("fileId") String fileId, @RequestBody ManualRedactions manualRedactions) {
|
||||
|
||||
var result = manualRedactionSurroundingTextService.addSurroundingText(dossierId, fileId, manualRedactions);
|
||||
log.info("Added surrounding text for manual redaction in dossierId {} and fileId {} took: {}", dossierId, fileId, result.getDuration());
|
||||
|
||||
+4
-1
@@ -1,10 +1,11 @@
|
||||
package com.iqser.red.service.redaction.v1.server.controller;
|
||||
|
||||
|
||||
import com.iqser.red.service.redaction.v1.model.RuleBuilderModel;
|
||||
import com.iqser.red.service.redaction.v1.resources.RuleBuilderResource;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.rulebuilder.RuleBuilderModelService;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
|
||||
@RestController
|
||||
@@ -13,8 +14,10 @@ public class RuleBuilderController implements RuleBuilderResource {
|
||||
|
||||
private final RuleBuilderModelService ruleBuilderModelService;
|
||||
|
||||
|
||||
@Override
|
||||
public RuleBuilderModel getRuleBuilderModel() {
|
||||
|
||||
return ruleBuilderModelService.getRuleBuilderModel();
|
||||
}
|
||||
|
||||
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class AtomicTextBlockData {
|
||||
Long id;
|
||||
String searchText;
|
||||
int start;
|
||||
int end;
|
||||
int[] lineBreaks;
|
||||
int[] stringIdxToPositionIdx;
|
||||
float[][] positions;
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class DocumentData {
|
||||
List<PageData> pages;
|
||||
List<AtomicTextBlockData> atomicTextBlocks;
|
||||
TableOfContentsData tableOfContents;
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class PageData {
|
||||
int number;
|
||||
int height;
|
||||
int width;
|
||||
|
||||
Long header;
|
||||
Long footer;
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.data;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.List;
|
||||
|
||||
import javax.management.openmbean.InvalidKeyException;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(makeFinal = true, level = AccessLevel.PRIVATE)
|
||||
public class TableOfContentsData {
|
||||
|
||||
List<EntryData> entries;
|
||||
|
||||
|
||||
public EntryData get(String tocId) {
|
||||
|
||||
List<Integer> ids = getIds(tocId);
|
||||
if (ids.size() < 1) {
|
||||
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", tocId));
|
||||
}
|
||||
EntryData entry = entries.get(ids.get(0));
|
||||
for (int id : ids.subList(1, ids.size())) {
|
||||
entry = entry.subEntries().get(id);
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
|
||||
private static List<Integer> getIds(String idsAsString) {
|
||||
|
||||
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
|
||||
}
|
||||
|
||||
|
||||
@Builder
|
||||
public record EntryData(String tocId, List<EntryData> subEntries, NodeType type, Long atomicTextBlock, Long page, int numberOnPage) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import lombok.Setter;
|
||||
|
||||
@Setter
|
||||
public class Boundary {
|
||||
|
||||
private int start;
|
||||
private int end;
|
||||
|
||||
|
||||
public Boundary(int start, int end) {
|
||||
|
||||
assert start <= end;
|
||||
this.start = start;
|
||||
this.end = end;
|
||||
}
|
||||
|
||||
|
||||
public int length() {
|
||||
|
||||
return end - start;
|
||||
}
|
||||
|
||||
|
||||
public int start() {
|
||||
|
||||
return start;
|
||||
}
|
||||
|
||||
|
||||
public int end() {
|
||||
|
||||
return end;
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(Boundary boundary) {
|
||||
|
||||
return start <= boundary.start() && boundary.end() <= end;
|
||||
}
|
||||
|
||||
|
||||
public boolean containedBy(Boundary boundary) {
|
||||
|
||||
return boundary.start() <= start && end <= boundary.end();
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
throw new UnsupportedOperationException("start > end");
|
||||
}
|
||||
return this.start <= start && end <= this.end;
|
||||
}
|
||||
|
||||
|
||||
public boolean containedBy(int start, int end) {
|
||||
|
||||
if (start > end) {
|
||||
throw new UnsupportedOperationException("start > end");
|
||||
}
|
||||
return start <= this.start && this.end <= end;
|
||||
}
|
||||
|
||||
|
||||
public boolean contains(int index) {
|
||||
|
||||
return start <= index && index < end;
|
||||
}
|
||||
|
||||
|
||||
public boolean intersects(Boundary boundary) {
|
||||
|
||||
return contains(boundary.start()) || contains(boundary.end());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return format("Boundary [%d|%d)", start, end);
|
||||
}
|
||||
|
||||
}
|
||||
+96
@@ -0,0 +1,96 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class DocumentGraph {
|
||||
|
||||
List<SectionNode> sections;
|
||||
List<PageNode> pages;
|
||||
TableOfContents tableOfContents;
|
||||
Integer numberOfPages;
|
||||
TextBlock text;
|
||||
|
||||
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamAtomicTextBlocksInOrder().collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
public Stream<AtomicTextBlock> streamAtomicTextBlocksInOrder() {
|
||||
|
||||
return Stream.concat(//
|
||||
streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock),//
|
||||
Stream.concat(//
|
||||
pages.stream().map(PageNode::getHeader),//
|
||||
pages.stream().map(PageNode::getFooter)));
|
||||
}
|
||||
|
||||
|
||||
public EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
|
||||
addEntityToGraphAndSetFields(entity);
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
public void addEntityToGraphAndSetFields(EntityNode entity) {
|
||||
|
||||
try {
|
||||
boolean inserted = streamAllNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
|
||||
} catch (NotFoundException e) {
|
||||
enrichEntity(entity, text);
|
||||
log.warn("Entity \"{}\" with {} is in between two main sections and will be removed!", entity.getValue(), entity.getBoundary());
|
||||
entity.removeFromGraph();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
public Set<EntityNode> getEntities() {
|
||||
|
||||
return streamAllNodes().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getEntities).flatMap(List::stream).collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
|
||||
private Stream<DocumentGraphNode> streamAllNodes() {
|
||||
|
||||
return tableOfContents.streamEntriesInOrder().map(TableOfContents.Entry::node);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return text.toString();
|
||||
}
|
||||
|
||||
}
|
||||
+113
@@ -0,0 +1,113 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import javax.management.openmbean.InvalidKeyException;
|
||||
|
||||
import com.google.common.hash.Hashing;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class TableOfContents {
|
||||
|
||||
List<Entry> entries;
|
||||
|
||||
|
||||
public TableOfContents() {
|
||||
|
||||
entries = new LinkedList<>();
|
||||
}
|
||||
|
||||
|
||||
public String createNewEntryAndReturnId(NodeType nodeType, String summary, DocumentGraphNode node) {
|
||||
|
||||
String id = String.format("%d", entries.size());
|
||||
entries.add(new Entry(nodeType, id, summary, new LinkedList<>(), node));
|
||||
return id;
|
||||
}
|
||||
|
||||
|
||||
public String createNewChildEntryAndReturnId(String parentId, NodeType nodeType, String summary, DocumentGraphNode node) {
|
||||
|
||||
Entry parent = getEntryById(parentId);
|
||||
String childId = parentId + String.format(".%d", parent.children().size());
|
||||
parent.children().add(new Entry(nodeType, childId, summary, new LinkedList<>(), node));
|
||||
return childId;
|
||||
}
|
||||
|
||||
|
||||
public Entry getEntryById(String parentId) {
|
||||
|
||||
List<Integer> ids = getIds(parentId);
|
||||
if (ids.size() < 1) {
|
||||
throw new InvalidKeyException(format("Section Identifier: \"%s\" is not valid.", parentId));
|
||||
}
|
||||
Entry entry = entries.get(ids.get(0));
|
||||
for (int id : ids.subList(1, ids.size())) {
|
||||
entry = entry.children().get(id);
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return String.join("\n", streamEntriesInOrder().map(Entry::toString).toList());
|
||||
}
|
||||
|
||||
|
||||
public String toString(String id) {
|
||||
|
||||
return String.join("\n", streamSubEntriesInOrder(id).map(Entry::toString).toList());
|
||||
}
|
||||
|
||||
|
||||
public Stream<Entry> streamEntriesInOrder() {
|
||||
|
||||
return entries.stream().flatMap(TableOfContents::flatten);
|
||||
}
|
||||
|
||||
|
||||
public Stream<Entry> streamSubEntriesInOrder(String parentId) {
|
||||
|
||||
return Stream.of(getEntryById(parentId)).flatMap(TableOfContents::flatten);
|
||||
}
|
||||
|
||||
|
||||
private static List<Integer> getIds(String idsAsString) {
|
||||
|
||||
return Arrays.stream(idsAsString.split("\\.")).map(Integer::valueOf).toList();
|
||||
}
|
||||
|
||||
|
||||
private static Stream<Entry> flatten(Entry entry) {
|
||||
|
||||
return Stream.concat(Stream.of(entry), entry.children().stream().flatMap(TableOfContents::flatten));
|
||||
}
|
||||
|
||||
|
||||
public record Entry(NodeType type, String id, String summary, List<Entry> children, DocumentGraphNode node) {
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return id + ": " + type + ".: " + summary;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
return Hashing.murmur3_32_fixed().hashString(type + id + summary + children.hashCode(), StandardCharsets.UTF_8).hashCode();
|
||||
}
|
||||
}
|
||||
}
|
||||
+219
@@ -0,0 +1,219 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import static com.iqser.red.service.redaction.v1.server.document.services.EntityEnrichmentUtility.enrichEntity;
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
public interface DocumentGraphNode {
|
||||
|
||||
/**
|
||||
* Searches all Nodes located underneath this Node in the TableOfContents and concatenates their AtomicTextBlocks into a single TextBlockEntity.
|
||||
* So, for a Section all AtomicTextBlocks of Subsections, Paragraphs, and Tables are concatenated into a single TextBlockEntity
|
||||
*
|
||||
* @return TextBlock containing all AtomicTextBlocks that are located under this Node.
|
||||
*/
|
||||
TextBlock buildTextBlock();
|
||||
|
||||
|
||||
/**
|
||||
* Any Node maintains its own Set of Entities.
|
||||
* This Set contains all Entities, whose first index is located in any of the AtomicTextBlocks underneath this Node.
|
||||
*
|
||||
* @return Set of all Entities associated with this Node
|
||||
*/
|
||||
Set<EntityNode> getEntities();
|
||||
|
||||
|
||||
/**
|
||||
* Returns the PageNode associated with this Node.
|
||||
* If the node has more than one PageNode associated, it returns the PageNode with the lowest number.
|
||||
* For example a section might span multiple pages, it then returns the page where the section starts.
|
||||
*
|
||||
* @return PageNode representing the first page on which the Node is located in the document
|
||||
*/
|
||||
PageNode getPage();
|
||||
|
||||
|
||||
/**
|
||||
* Any Node except the First level of Sections, Header, Footer, and Pages have a direct Parent.
|
||||
* For example a Paragraph has a parent Section, a Table Cell has a parent Table, etc...
|
||||
* hasParent() may be used to check whether a parent is present.
|
||||
*
|
||||
* @return Node that represents the Parent or null, if no parent is present.
|
||||
*/
|
||||
DocumentGraphNode getParent();
|
||||
|
||||
|
||||
Stream<DocumentGraphNode> streamAllSubNodes();
|
||||
|
||||
|
||||
/**
|
||||
* Each AtomicTextBlock has a number assigned per page, this returns the number of the first AtomicTextBlock underneath this node
|
||||
*
|
||||
* @return Integer representing the number on the page
|
||||
*/
|
||||
Integer getNumberOnPage();
|
||||
|
||||
|
||||
/**
|
||||
*
|
||||
* @return the fist headline whent traversing the tree upwards
|
||||
*/
|
||||
default CharSequence getHeadline() {
|
||||
return getParent().getHeadline();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* By default, a parent is always present, this needs to be overwritten for Headers, Footers, Pages, and Sections.
|
||||
*
|
||||
* @return boolean, indicating whether a parent is present.
|
||||
*/
|
||||
default boolean hasParent() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* by default a Node does not have direct access to an AtomicTextBlock
|
||||
*
|
||||
* @return boolean, indicating if a Node has direct access to an AtomicTextBlock
|
||||
*/
|
||||
default boolean isTerminal() {
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* by default a Node does not have direct access to an AtomicTextBlock, this method throws a UnsupportedOperationException if not overridden.
|
||||
*
|
||||
* @return AtomicTextBlock
|
||||
*/
|
||||
default AtomicTextBlock getAtomicTextBlock() {
|
||||
|
||||
throw new UnsupportedOperationException("Only terminal Nodes have access to AtomicTextBlocks!");
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* creates an EntityNode with only initial values set, inserts it into the subgraph contained by this node and sets the inferrable fields.
|
||||
* Throws NotFoundException and removes the entity if the provided boundary could not be found in the subgraph.
|
||||
*
|
||||
* @param boundary start and end indices in String coordinates of the entity to be created
|
||||
* @param type type of the entity to be created
|
||||
* @param entityType entityType of the entity to be created
|
||||
* @return the newly created and inserted EntityNode with all fields set.
|
||||
*/
|
||||
default EntityNode createAndAddEntity(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
EntityNode entity = EntityNode.initialEntityNode(boundary, type, entityType);
|
||||
addEntityToNodeAndSetFields(entity);
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* searches for the first terminal Node containing the start index of the entity to be inserted.
|
||||
* Catches NotFoundException to remove the EntityNode from the graph, then rethrows it
|
||||
*
|
||||
* @param entity newly created EntityNode with only initial values set
|
||||
*/
|
||||
default void addEntityToNodeAndSetFields(EntityNode entity) {
|
||||
|
||||
try {
|
||||
streamAllSubNodes().anyMatch(node -> node.addEntityAndSetFieldsIfStartIndexContained(entity));
|
||||
} catch (NotFoundException e) {
|
||||
entity.removeFromGraph();
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* If this Node's AtomicTextBlock contains the start index of the entity, the entity's position is read from the AtomicTextBlock.
|
||||
* If the position can not be fully read from the AtomicTextBlock, it recursively looks in the TextBlocks of the parents until all positions are found.
|
||||
* Further, the function throws NotFoundException if no parent contains all positions.
|
||||
* This occurs, when the Entity is in between Nodes that do not share a parent, e.g. main sections.
|
||||
* Finally, the function adds the Entity to its own list of Entities and to every parents' list recursively.
|
||||
*
|
||||
* @param entity The entity to be added to the graph
|
||||
* @return true, if the entity has been added successfully.
|
||||
* false, if the entity's start index is not contained or the Node doesn't have an AtomicTextBlock
|
||||
*/
|
||||
default boolean addEntityAndSetFieldsIfStartIndexContained(EntityNode entity) {
|
||||
|
||||
if (!isTerminal()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
AtomicTextBlock atomicTextBlock = getAtomicTextBlock();
|
||||
if (atomicTextBlock.containsIndex(entity.getBoundary().start())) {
|
||||
|
||||
entity.addContainingNode(this);
|
||||
|
||||
getEntities().add(entity);
|
||||
|
||||
addEntityToPage(entity);
|
||||
addEntityToParents(entity);
|
||||
setFields(entity, atomicTextBlock);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
private void addEntityToPage(EntityNode entity) {
|
||||
|
||||
getPage().getEntities().add(entity);
|
||||
entity.setPage(getPage());
|
||||
}
|
||||
|
||||
|
||||
private void setFields(EntityNode entity, AtomicTextBlock atomicTextBlock) {
|
||||
|
||||
if (atomicTextBlock.containsBoundary(entity.getBoundary())) {
|
||||
enrichEntity(entity, atomicTextBlock);
|
||||
} else {
|
||||
this.setFieldsFromParents(this, entity);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addEntityToParents(EntityNode entity) {
|
||||
|
||||
DocumentGraphNode node = this;
|
||||
while (node.hasParent()) {
|
||||
node = node.getParent();
|
||||
node.getEntities().add(entity);
|
||||
entity.addContainingNode(node);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void setFieldsFromParents(DocumentGraphNode node, EntityNode entity) {
|
||||
|
||||
if (node.hasParent()) {
|
||||
DocumentGraphNode parent = node.getParent();
|
||||
TextBlock textBlock = parent.buildTextBlock();
|
||||
if (textBlock.containsBoundary(entity.getBoundary())) {
|
||||
enrichEntity(entity, textBlock);
|
||||
return;
|
||||
} else {
|
||||
setFieldsFromParents(parent, entity);
|
||||
}
|
||||
}
|
||||
throw new NotFoundException(format("Position could not be found for Entity %s", entity.toString()));
|
||||
}
|
||||
|
||||
}
|
||||
+103
@@ -0,0 +1,103 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import com.google.common.hash.Hashing;
|
||||
import com.iqser.red.service.redaction.v1.model.Engine;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.Entity;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.EntityType;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class EntityNode {
|
||||
|
||||
public static EntityNode initialEntityNode(Boundary boundary, String type, EntityType entityType) {
|
||||
|
||||
return EntityNode.builder().type(type).entityType(entityType).boundary(boundary).build();
|
||||
}
|
||||
|
||||
|
||||
// initial values
|
||||
Boundary boundary;
|
||||
String type;
|
||||
EntityType entityType;
|
||||
|
||||
@Builder.Default
|
||||
boolean redaction = false;
|
||||
@Builder.Default
|
||||
boolean falsePositive = false;
|
||||
@Builder.Default
|
||||
boolean removed = false;
|
||||
@Builder.Default
|
||||
boolean ignored = false;
|
||||
@Builder.Default
|
||||
boolean resized = false;
|
||||
@Builder.Default
|
||||
boolean skipRemoveEntitiesContainedInLarger = false;
|
||||
@Builder.Default
|
||||
boolean isDictionaryEntry = false;
|
||||
@Builder.Default
|
||||
Set<Engine> engines = new HashSet<>();
|
||||
@Builder.Default
|
||||
Set<Entity> references = new HashSet<>();
|
||||
@Builder.Default
|
||||
int matchedRule = -1;
|
||||
@Builder.Default
|
||||
String redactionReason = "";
|
||||
@Builder.Default
|
||||
String legalBasis = "";
|
||||
|
||||
// inferrable from graph
|
||||
String value;
|
||||
CharSequence textBefore;
|
||||
CharSequence textAfter;
|
||||
PageNode page;
|
||||
List<Rectangle2D> positions;
|
||||
@Builder.Default
|
||||
Set<DocumentGraphNode> containingNodes = new HashSet<>();
|
||||
|
||||
|
||||
public void addContainingNode(DocumentGraphNode containingNode) {
|
||||
|
||||
containingNodes.add(containingNode);
|
||||
}
|
||||
|
||||
|
||||
public void removeFromGraph() {
|
||||
|
||||
getContainingNodes().forEach(node -> node.getEntities().remove(this));
|
||||
getPage().getEntities().remove(this);
|
||||
setRemoved(true);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
|
||||
var sb = new StringBuilder();
|
||||
sb.append(value);
|
||||
sb.append(boundary.start());
|
||||
sb.append(page.getNumber());
|
||||
positions.forEach(r -> {
|
||||
sb.append(r.getMinX());
|
||||
sb.append(r.getMinY());
|
||||
sb.append(r.getWidth());
|
||||
sb.append(r.getHeight());
|
||||
});
|
||||
return Hashing.murmur3_128().hashString(sb.toString(), StandardCharsets.UTF_8).hashCode();
|
||||
}
|
||||
|
||||
}
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
public enum NodeType {
|
||||
SECTION {
|
||||
public String toString() {
|
||||
|
||||
return "Section";
|
||||
}
|
||||
},
|
||||
PARAGRAPH {
|
||||
public String toString() {
|
||||
|
||||
return "Paragraph";
|
||||
}
|
||||
},
|
||||
TABLE {
|
||||
public String toString() {
|
||||
|
||||
return "Table";
|
||||
}
|
||||
},
|
||||
TABLE_CELL {
|
||||
public String toString() {
|
||||
|
||||
return "Cell";
|
||||
}
|
||||
}
|
||||
}
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class PageNode implements DocumentGraphNode{
|
||||
|
||||
Integer number;
|
||||
Integer height;
|
||||
Integer width;
|
||||
List<DocumentGraphNode> mainBody;
|
||||
AtomicTextBlock header;
|
||||
AtomicTextBlock footer;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return mainBody.stream().filter(DocumentGraphNode::isTerminal).map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean hasParent() {
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return mainBody.stream();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Integer getNumberOnPage() {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return header.getSearchText() + buildTextBlock().toString() + footer.getSearchText();
|
||||
}
|
||||
|
||||
}
|
||||
+70
@@ -0,0 +1,70 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class ParagraphNode implements DocumentGraphNode {
|
||||
|
||||
String tocId;
|
||||
Integer numberOnPage;
|
||||
Integer numberInSection;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
SectionNode parentSection;
|
||||
@EqualsAndHashCode.Exclude
|
||||
PageNode page;
|
||||
AtomicTextBlock atomicTextBlock;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock buildTextBlock() {
|
||||
|
||||
return atomicTextBlock;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentSection;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return tocId + ": " + atomicTextBlock.toString();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return Stream.of(this);
|
||||
}
|
||||
|
||||
}
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
@Slf4j
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class SectionNode implements DocumentGraphNode {
|
||||
|
||||
String tocId;
|
||||
Integer numberOnPage;
|
||||
@EqualsAndHashCode.Exclude
|
||||
TableOfContents tableOfContents;
|
||||
@EqualsAndHashCode.Exclude
|
||||
DocumentGraphNode parentSection;
|
||||
|
||||
|
||||
AtomicTextBlock headline;
|
||||
|
||||
List<SectionNode> subSections;
|
||||
List<ParagraphNode> paragraphs;
|
||||
List<TableNode> tables;
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<PageNode> pages;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
Set<EntityNode> entities = new HashSet<>();
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return tocId + ": " + headline.toString();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamAllSubNodes().map(DocumentGraphNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
if (hasParent()) {
|
||||
return parentSection;
|
||||
} else {
|
||||
throw new UnsupportedOperationException("This section has no parent Section!");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return tableOfContents.streamSubEntriesInOrder(tocId).map(TableOfContents.Entry::node);
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean hasParent() {
|
||||
|
||||
return parentSection != null;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock getAtomicTextBlock() {
|
||||
|
||||
return headline;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return pages.get(0);
|
||||
}
|
||||
|
||||
}
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class TableCellNode implements DocumentGraphNode {
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
TableNode parentTable;
|
||||
Integer numberOnPage;
|
||||
AtomicTextBlock atomicTextBlock;
|
||||
PageNode page;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<EntityNode> entities = new LinkedList<>();
|
||||
|
||||
|
||||
@Override
|
||||
public AtomicTextBlock buildTextBlock() {
|
||||
|
||||
return atomicTextBlock;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentTable;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public boolean isTerminal() {
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return Stream.of(this);
|
||||
}
|
||||
|
||||
}
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.nodes;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlockCollector;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.ConcatenatedTextBlock;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class TableNode implements DocumentGraphNode {
|
||||
|
||||
Integer id;
|
||||
String tocId;
|
||||
Integer numberOfRows;
|
||||
Integer numberOfCols;
|
||||
Integer numberOnPage;
|
||||
List<TableCellNode> tableHeaders;
|
||||
List<List<TableCellNode>> tableCells;
|
||||
TableOfContents tableOfContents;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
SectionNode parentSection;
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<PageNode> pages;
|
||||
|
||||
@Builder.Default
|
||||
@EqualsAndHashCode.Exclude
|
||||
List<EntityNode> entities = new LinkedList<>();
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableCells() {
|
||||
|
||||
return tableCells.stream().flatMap(List::stream);
|
||||
}
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableRow(int row) {
|
||||
|
||||
return tableCells.get(row).stream();
|
||||
}
|
||||
|
||||
|
||||
private Stream<TableCellNode> streamTableCol(int col) {
|
||||
|
||||
return tableCells.stream().map(row -> row.get(col));
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock buildTextBlock() {
|
||||
|
||||
return streamTableCells().map(TableCellNode::getAtomicTextBlock).collect(new TextBlockCollector());
|
||||
}
|
||||
|
||||
@Override
|
||||
public Stream<DocumentGraphNode> streamAllSubNodes() {
|
||||
|
||||
return streamTableCells().map(Function.identity());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public DocumentGraphNode getParent() {
|
||||
|
||||
return parentSection;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public PageNode getPage() {
|
||||
|
||||
return pages.get(0);
|
||||
}
|
||||
|
||||
}
|
||||
+126
@@ -0,0 +1,126 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
@AllArgsConstructor
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class AtomicTextBlock implements TextBlock {
|
||||
|
||||
Long id;
|
||||
|
||||
//string coordinates
|
||||
Boundary boundary;
|
||||
String searchText;
|
||||
List<Integer> lineBreaks;
|
||||
|
||||
//position coordinates
|
||||
List<Integer> stringIdxToPositionIdx;
|
||||
List<Rectangle2D> positions;
|
||||
|
||||
@EqualsAndHashCode.Exclude
|
||||
DocumentGraphNode parent;
|
||||
|
||||
|
||||
public int indexOf(String searchTerm) {
|
||||
|
||||
int pos = searchText.indexOf(searchTerm);
|
||||
return pos == -1 ? -1 : pos + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int numberOfLines() {
|
||||
|
||||
return lineBreaks.size();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public List<AtomicTextBlock> getAtomicTextBlocks() {
|
||||
|
||||
return List.of(this);
|
||||
}
|
||||
|
||||
|
||||
public int getNextLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak > fromIndex) //
|
||||
.findFirst() //
|
||||
.orElse(searchText.length()) + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int getPreviousLinebreak(int fromIndex) {
|
||||
|
||||
return lineBreaks.stream()//
|
||||
.filter(linebreak -> linebreak <= fromIndex)//
|
||||
.reduce((a, b) -> b)//
|
||||
.orElse(0) + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public Rectangle2D getPosition(int stringIdx) {
|
||||
|
||||
return positions.get(stringIdxToPositionIdx.get(stringIdx - boundary.start()));
|
||||
}
|
||||
|
||||
|
||||
public List<Rectangle2D> getPositions(Boundary boundary) {
|
||||
|
||||
if (!containsBoundary(boundary)) {
|
||||
throw new IndexOutOfBoundsException(format("%s is out of bounds for %s",
|
||||
boundary,
|
||||
this.boundary));
|
||||
}
|
||||
|
||||
if (boundary.end() == this.boundary.end()) {
|
||||
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), positions.size());
|
||||
}
|
||||
|
||||
return positions.subList(stringIdxToPositionIdx.get(boundary.start() - this.boundary.start()), stringIdxToPositionIdx.get(boundary.end() - this.boundary.start()));
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int length() {
|
||||
|
||||
return searchText.length();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public char charAt(int index) {
|
||||
|
||||
return searchText.charAt(index - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public CharSequence subSequence(int start, int end) {
|
||||
|
||||
return searchText.substring(start - boundary.start(), end - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
|
||||
return searchText;
|
||||
}
|
||||
|
||||
}
|
||||
+155
@@ -0,0 +1,155 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
import lombok.AccessLevel;
|
||||
import lombok.Data;
|
||||
import lombok.experimental.FieldDefaults;
|
||||
|
||||
@Data
|
||||
@FieldDefaults(level = AccessLevel.PRIVATE)
|
||||
public class ConcatenatedTextBlock implements TextBlock, Supplier<ConcatenatedTextBlock> {
|
||||
|
||||
List<AtomicTextBlock> atomicTextBlocks;
|
||||
StringBuilder searchText;
|
||||
Boundary boundary;
|
||||
|
||||
|
||||
public ConcatenatedTextBlock(List<AtomicTextBlock> atomicTextBlocks) {
|
||||
|
||||
this.atomicTextBlocks = new LinkedList<>();
|
||||
this.searchText = new StringBuilder();
|
||||
if (atomicTextBlocks.isEmpty()) {
|
||||
boundary = new Boundary(-1, -1);
|
||||
return;
|
||||
}
|
||||
var firstTextBlock = atomicTextBlocks.get(0);
|
||||
this.atomicTextBlocks.add(firstTextBlock);
|
||||
this.searchText.append(firstTextBlock.getSearchText());
|
||||
boundary = new Boundary(firstTextBlock.getBoundary().start(), firstTextBlock.getBoundary().end());
|
||||
|
||||
atomicTextBlocks.subList(1, atomicTextBlocks.size()).forEach(this::concat);
|
||||
}
|
||||
|
||||
|
||||
public ConcatenatedTextBlock(AtomicTextBlock atomicTextBlocks) {
|
||||
|
||||
new ConcatenatedTextBlock(List.of(atomicTextBlocks));
|
||||
}
|
||||
|
||||
|
||||
public ConcatenatedTextBlock concat(TextBlock textBlock) {
|
||||
|
||||
if (this.atomicTextBlocks.isEmpty()) {
|
||||
boundary.setStart(textBlock.getBoundary().start());
|
||||
boundary.setEnd(textBlock.getBoundary().end());
|
||||
} else if (boundary.end() != textBlock.getBoundary().start()) {
|
||||
throw new UnsupportedOperationException(format("Can only concat consecutive TextBlocks, trying to concat %s to %s", textBlock.getBoundary(), boundary));
|
||||
}
|
||||
this.searchText.append(textBlock.getSearchText());
|
||||
this.atomicTextBlocks.addAll(textBlock.getAtomicTextBlocks());
|
||||
boundary.setEnd(textBlock.getBoundary().end());
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
public int indexOf(String searchTerm) {
|
||||
|
||||
int pos = this.searchText.indexOf(searchTerm);
|
||||
return pos == -1 ? -1 : pos + boundary.start();
|
||||
}
|
||||
|
||||
|
||||
public int numberOfLines() {
|
||||
|
||||
return atomicTextBlocks.stream().map(AtomicTextBlock::getLineBreaks).mapToInt(List::size).sum();
|
||||
}
|
||||
|
||||
|
||||
public int getNextLinebreak(int fromIndex) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(fromIndex).getNextLinebreak(fromIndex);
|
||||
}
|
||||
|
||||
|
||||
public int getPreviousLinebreak(int fromIndex) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(fromIndex).getPreviousLinebreak(fromIndex);
|
||||
}
|
||||
|
||||
|
||||
public Rectangle2D getPosition(int stringIdx) {
|
||||
|
||||
return getAtomicTextBlockByStringIndex(stringIdx).getPosition(stringIdx);
|
||||
}
|
||||
|
||||
|
||||
public List<Rectangle2D> getPositions(Boundary boundary) {
|
||||
|
||||
List<AtomicTextBlock> textBlocks = getAllAtomicTextBlocksPartiallyInStringIdxRange(boundary);
|
||||
|
||||
if (textBlocks.size() == 1) {
|
||||
return textBlocks.get(0).getPositions(boundary);
|
||||
}
|
||||
|
||||
AtomicTextBlock firstTextBlock = textBlocks.get(0);
|
||||
List<Rectangle2D> positions = new LinkedList<>(firstTextBlock.getPositions(new Boundary(boundary.start(), firstTextBlock.getBoundary().end())));
|
||||
|
||||
for (AtomicTextBlock textBlock : textBlocks.subList(1, textBlocks.size() - 1)) {
|
||||
positions.addAll(textBlock.getPositions());
|
||||
}
|
||||
|
||||
var lastTextBlock = textBlocks.get(textBlocks.size() - 1);
|
||||
positions.addAll(lastTextBlock.getPositions(new Boundary(lastTextBlock.getBoundary().start(), boundary.end())));
|
||||
|
||||
return positions;
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock getAtomicTextBlockByStringIndex(int stringIdx) {
|
||||
|
||||
return atomicTextBlocks.stream().filter(textBlock -> (textBlock.getBoundary().end()) > stringIdx).findFirst().orElseThrow(IndexOutOfBoundsException::new);
|
||||
}
|
||||
|
||||
|
||||
private List<AtomicTextBlock> getAllAtomicTextBlocksPartiallyInStringIdxRange(Boundary boundary) {
|
||||
|
||||
return atomicTextBlocks.stream().filter(tb -> tb.getBoundary().intersects(boundary)).toList();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public int length() {
|
||||
|
||||
return this.searchText.length();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public char charAt(int index) {
|
||||
|
||||
return searchText.charAt(index - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public CharSequence subSequence(int start, int end) {
|
||||
|
||||
return searchText.subSequence(start - boundary.start(), end - boundary.start());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public ConcatenatedTextBlock get() {
|
||||
|
||||
return this;
|
||||
}
|
||||
|
||||
}
|
||||
+65
@@ -0,0 +1,65 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
public interface TextBlock extends CharSequence {
|
||||
|
||||
CharSequence getSearchText();
|
||||
|
||||
|
||||
List<AtomicTextBlock> getAtomicTextBlocks();
|
||||
|
||||
|
||||
Boundary getBoundary();
|
||||
|
||||
|
||||
int getNextLinebreak(int fromIndex);
|
||||
|
||||
|
||||
int getPreviousLinebreak(int fromIndex);
|
||||
|
||||
|
||||
Rectangle2D getPosition(int stringIdx);
|
||||
|
||||
|
||||
List<Rectangle2D> getPositions(Boundary range);
|
||||
|
||||
|
||||
int numberOfLines();
|
||||
|
||||
|
||||
int indexOf(String searchTerm);
|
||||
|
||||
|
||||
default CharSequence getFirstLine() {
|
||||
|
||||
return subSequence(getBoundary().start(), getNextLinebreak(getBoundary().start()));
|
||||
}
|
||||
|
||||
|
||||
default boolean containsBoundary(Boundary boundary) {
|
||||
|
||||
if (boundary.end() < boundary.start()) {
|
||||
throw new IllegalArgumentException(format("Invalid %s, StartIndex must be smaller than EndIndex", boundary));
|
||||
}
|
||||
return getBoundary().contains(boundary);
|
||||
}
|
||||
|
||||
|
||||
default boolean containsIndex(int stringIndex) {
|
||||
|
||||
return getBoundary().contains(stringIndex);
|
||||
}
|
||||
|
||||
|
||||
default CharSequence subSequence(Boundary boundary) {
|
||||
|
||||
return subSequence(boundary.start(), boundary.end());
|
||||
}
|
||||
|
||||
}
|
||||
+51
@@ -0,0 +1,51 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.graph.textblock;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.Set;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.BinaryOperator;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collector;
|
||||
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
@NoArgsConstructor
|
||||
public class TextBlockCollector implements Collector<AtomicTextBlock, ConcatenatedTextBlock, ConcatenatedTextBlock> {
|
||||
|
||||
|
||||
@Override
|
||||
public Supplier<ConcatenatedTextBlock> supplier() {
|
||||
|
||||
return new ConcatenatedTextBlock(Collections.emptyList());
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public BiConsumer<ConcatenatedTextBlock, AtomicTextBlock> accumulator() {
|
||||
|
||||
return ConcatenatedTextBlock::concat;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public BinaryOperator<ConcatenatedTextBlock> combiner() {
|
||||
|
||||
return ConcatenatedTextBlock::concat;
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Function<ConcatenatedTextBlock, ConcatenatedTextBlock> finisher() {
|
||||
|
||||
return Function.identity();
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public Set<Characteristics> characteristics() {
|
||||
|
||||
return Set.of(Characteristics.IDENTITY_FINISH, Characteristics.CONCURRENT);
|
||||
}
|
||||
|
||||
}
|
||||
+98
@@ -0,0 +1,98 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.List;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
|
||||
@Service
|
||||
public class DocumentDataMapper {
|
||||
|
||||
public DocumentData toDocumentData(DocumentGraph documentGraph) {
|
||||
|
||||
List<AtomicTextBlockData> atomicTextBlockData = documentGraph.streamAtomicTextBlocksInOrder().map(this::toAtomicTextBlockData).toList();
|
||||
List<PageData> pageData = documentGraph.getPages().stream().map(this::toPageData).toList();
|
||||
TableOfContentsData tableOfContentsData = toTableOfContentsData(documentGraph.getTableOfContents());
|
||||
return DocumentData.builder().atomicTextBlocks(atomicTextBlockData).pages(pageData).tableOfContents(tableOfContentsData).build();
|
||||
}
|
||||
|
||||
|
||||
private TableOfContentsData toTableOfContentsData(TableOfContents tableOfContents) {
|
||||
|
||||
return new TableOfContentsData(tableOfContents.getEntries().stream().map(this::toEntryData).toList());
|
||||
}
|
||||
|
||||
|
||||
private TableOfContentsData.EntryData toEntryData(TableOfContents.Entry entry) {
|
||||
|
||||
return TableOfContentsData.EntryData.builder()
|
||||
.tocId(entry.id())
|
||||
.subEntries(entry.children().stream().map(this::toEntryData).toList())
|
||||
.type(entry.type())
|
||||
.atomicTextBlock(entry.node().isTerminal() ? entry.node().getAtomicTextBlock().getId() : -1L)
|
||||
.page(Long.valueOf(entry.node().getPage().getNumber()))
|
||||
.numberOnPage(entry.node().getNumberOnPage())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private PageData toPageData(PageNode pageNode) {
|
||||
|
||||
return PageData.builder()
|
||||
.height(pageNode.getHeight())
|
||||
.width(pageNode.getWidth())
|
||||
.number(pageNode.getNumber())
|
||||
.footer(pageNode.getFooter().getId())
|
||||
.header(pageNode.getHeader().getId())
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlockData toAtomicTextBlockData(AtomicTextBlock atomicTextBlock) {
|
||||
|
||||
return AtomicTextBlockData.builder()
|
||||
.id(atomicTextBlock.getId())
|
||||
.searchText(atomicTextBlock.getSearchText())
|
||||
.start(atomicTextBlock.getBoundary().start())
|
||||
.end(atomicTextBlock.getBoundary().end())
|
||||
.lineBreaks(toPrimitiveIntArray(atomicTextBlock.getLineBreaks()))
|
||||
.stringIdxToPositionIdx(toPrimitiveIntArray(atomicTextBlock.getStringIdxToPositionIdx()))
|
||||
.positions(toPrimitiveFloatMatrix(atomicTextBlock.getPositions()))
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private float[][] toPrimitiveFloatMatrix(List<Rectangle2D> positions) {
|
||||
|
||||
float[][] positionMatrix = new float[positions.size()][];
|
||||
for (int i = 0; i < positions.size(); i++) {
|
||||
float[] singlePositions = new float[4];
|
||||
singlePositions[0] = (float) positions.get(i).getMinX();
|
||||
singlePositions[1] = (float) positions.get(i).getMinY();
|
||||
singlePositions[2] = (float) positions.get(i).getWidth();
|
||||
singlePositions[3] = (float) positions.get(i).getHeight();
|
||||
positionMatrix[i] = singlePositions;
|
||||
}
|
||||
return positionMatrix;
|
||||
}
|
||||
|
||||
|
||||
private int[] toPrimitiveIntArray(List<Integer> list) {
|
||||
|
||||
int[] array = new int[list.size()];
|
||||
for (int i = 0; i < list.size(); i++) {
|
||||
array[i] = list.get(i);
|
||||
}
|
||||
return array;
|
||||
}
|
||||
|
||||
}
|
||||
+304
@@ -0,0 +1,304 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Document;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Footer;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Header;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Page;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.Section;
|
||||
import com.iqser.red.service.redaction.v1.server.classification.model.TextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.TableNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.RedRectangle2D;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.SearchTextWithTextPositionModel;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.service.SearchTextWithTextPositionFactory;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.AbstractTextContainer;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Table;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class DocumentGraphFactory {
|
||||
|
||||
private final SearchTextWithTextPositionFactory searchTextWithTextPositionFactory;
|
||||
|
||||
|
||||
public DocumentGraph buildDocumentGraph(Document document) {
|
||||
|
||||
Context context = new Context(new TableOfContents(), new LinkedList<>(), new LinkedList<>(), new AtomicInteger(0), new AtomicLong(0));
|
||||
|
||||
context.pages.addAll(document.getPages().stream().map(this::buildPage).toList());
|
||||
|
||||
// is tracked by Table of Contents
|
||||
addSections(document, context);
|
||||
|
||||
// not tracked by Table of Contents
|
||||
addHeaderAndFooterToEachPage(document, context);
|
||||
DocumentGraph documentGraph = DocumentGraph.builder()
|
||||
.numberOfPages(context.pages.size())
|
||||
.pages(context.pages)
|
||||
.sections(context.sections)
|
||||
.tableOfContents(context.tableOfContents)
|
||||
.build();
|
||||
documentGraph.setText(documentGraph.buildTextBlock());
|
||||
return documentGraph;
|
||||
}
|
||||
|
||||
|
||||
private void addSections(Document document, Context context) {
|
||||
|
||||
for (var section : document.getSections()) {
|
||||
addSection(section, context);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addSection(Section section, Context context) {
|
||||
|
||||
SectionNode sectionEntity = SectionNode.builder()
|
||||
.entities(new LinkedList<>())
|
||||
.pages(new LinkedList<>())
|
||||
.paragraphs(new LinkedList<>())
|
||||
.tables(new LinkedList<>())
|
||||
.subSections(new LinkedList<>())
|
||||
.tableOfContents(context.tableOfContents())
|
||||
.build();
|
||||
|
||||
context.sections().add(sectionEntity);
|
||||
List<AbstractTextContainer> pageBlocks = new ArrayList<>(section.getPageBlocks());
|
||||
PageNode page = getPage(section.getPageBlocks().get(0).getPage(), context);
|
||||
sectionEntity.getPages().add(page);
|
||||
page.getMainBody().add(sectionEntity);
|
||||
if (pageBlocks.get(0) instanceof TextBlock) {
|
||||
sectionEntity.setHeadline(buildAtomicTextBlock(((TextBlock) pageBlocks.get(0)).getSequences(), sectionEntity, context));
|
||||
sectionEntity.setNumberOnPage(((TextBlock) pageBlocks.get(0)).getIndexOnPage());
|
||||
pageBlocks.remove(0);
|
||||
} else {
|
||||
sectionEntity.setNumberOnPage(1);
|
||||
sectionEntity.setHeadline(emptyTextBlock(sectionEntity, context));
|
||||
}
|
||||
|
||||
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(sectionEntity.getHeadline()), sectionEntity);
|
||||
sectionEntity.setTocId(sectionId);
|
||||
|
||||
int paragraphIdx = 0;
|
||||
int tableIdx = 0;
|
||||
for (AbstractTextContainer abstractTextContainer : pageBlocks) {
|
||||
if (abstractTextContainer instanceof TextBlock) {
|
||||
addParagraph(sectionEntity, (TextBlock) abstractTextContainer, paragraphIdx, context);
|
||||
paragraphIdx++;
|
||||
} else if (abstractTextContainer instanceof Table) {
|
||||
//addTable(sectionEntity, (Table) abstractTextContainer, tableIdx, context);
|
||||
tableIdx++;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addTable(SectionNode sectionEntity, Table table, int tableIdx, Context context) {
|
||||
|
||||
PageNode page = getPage(table.getPage(), context);
|
||||
TableNode tableEntity = TableNode.builder().id(tableIdx).tableOfContents(context.tableOfContents()).pages(new LinkedList<>()).parentSection(sectionEntity).build();
|
||||
sectionEntity.getTables().add(tableEntity);
|
||||
|
||||
if (!page.getMainBody().contains(sectionEntity)) {
|
||||
sectionEntity.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(tableEntity);
|
||||
|
||||
}
|
||||
|
||||
|
||||
private void addParagraph(SectionNode sectionEntity, TextBlock originalTextBlock, int paragraphIdx, Context context) {
|
||||
|
||||
PageNode page = getPage(originalTextBlock.getPage(), context);
|
||||
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(originalTextBlock.getIndexOnPage()).page(page).parentSection(sectionEntity).build();
|
||||
sectionEntity.getParagraphs().add(paragraph);
|
||||
|
||||
if (!page.getMainBody().contains(sectionEntity)) {
|
||||
sectionEntity.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(paragraph);
|
||||
|
||||
var textBlock = buildAtomicTextBlock(originalTextBlock.getSequences(), paragraph, context);
|
||||
paragraph.setAtomicTextBlock(textBlock);
|
||||
|
||||
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(sectionEntity.getTocId(), NodeType.PARAGRAPH, buildSummary(textBlock), paragraph);
|
||||
paragraph.setTocId(tocId);
|
||||
}
|
||||
|
||||
|
||||
private void addHeaderAndFooterToEachPage(Document document, Context context) {
|
||||
|
||||
Map<Integer, List<TextBlock>> headers = document.getHeaders()
|
||||
.stream()
|
||||
.map(Header::getTextBlocks)
|
||||
.flatMap(List::stream)
|
||||
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
|
||||
|
||||
Map<Integer, List<TextBlock>> footers = document.getFooters()
|
||||
.stream()
|
||||
.map(Footer::getTextBlocks)
|
||||
.flatMap(List::stream)
|
||||
.collect(Collectors.groupingBy(AbstractTextContainer::getPage, Collectors.toList()));
|
||||
|
||||
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
|
||||
if (headers.containsKey(pageIndex)) {
|
||||
addHeader(headers.get(pageIndex), context);
|
||||
} else {
|
||||
addEmptyHeader(pageIndex, context);
|
||||
}
|
||||
}
|
||||
|
||||
for (int pageIndex = 1; pageIndex <= document.getPages().size(); pageIndex++) {
|
||||
if (footers.containsKey(pageIndex)) {
|
||||
addFooter(footers.get(pageIndex), context);
|
||||
} else {
|
||||
addEmptyFooter(pageIndex, context);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void addFooter(List<TextBlock> textBlocks, Context context) {
|
||||
|
||||
PageNode page = getPage(textBlocks.get(0).getPage(), context);
|
||||
AtomicTextBlock footer = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
|
||||
page.setFooter(footer);
|
||||
}
|
||||
|
||||
|
||||
public void addHeader(List<TextBlock> textBlocks, Context context) {
|
||||
|
||||
PageNode page = getPage(textBlocks.get(0).getPage(), context);
|
||||
AtomicTextBlock header = buildAtomicTextBlock(mergeAndSortTextPositionSequences(textBlocks), page, context);
|
||||
page.setHeader(header);
|
||||
}
|
||||
|
||||
|
||||
private void addEmptyFooter(int pageIndex, Context context) {
|
||||
|
||||
PageNode page = getPage(pageIndex, context);
|
||||
page.setFooter(emptyTextBlock(page, context));
|
||||
}
|
||||
|
||||
|
||||
private void addEmptyHeader(int pageIndex, Context context) {
|
||||
|
||||
PageNode page = getPage(pageIndex, context);
|
||||
page.setHeader(emptyTextBlock(page, context));
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock emptyTextBlock(DocumentGraphNode parent, Context context) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(context.textBlockIdx.getAndIncrement())
|
||||
.boundary(new Boundary(context.stringOffset.get(), context.stringOffset.get()))
|
||||
.searchText("")
|
||||
.lineBreaks(Collections.emptyList())
|
||||
.stringIdxToPositionIdx(Collections.emptyList())
|
||||
.positions(Collections.emptyList())
|
||||
.parent(parent)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private static String buildSummary(AtomicTextBlock textBlock) {
|
||||
|
||||
if (textBlock == null) {
|
||||
return " probably a table";
|
||||
}
|
||||
|
||||
String[] words = textBlock.getFirstLine().toString().split(" ");
|
||||
int bound = Math.min(words.length, 4);
|
||||
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
|
||||
|
||||
return String.join(" ", list);
|
||||
}
|
||||
|
||||
|
||||
private PageNode buildPage(Page p) {
|
||||
|
||||
return PageNode.builder().height((int) p.getPageHeight()).width((int) p.getPageWidth()).number(p.getPageNumber()).mainBody(new LinkedList<>()).build();
|
||||
}
|
||||
|
||||
|
||||
private List<TextPositionSequence> mergeAndSortTextPositionSequences(List<TextBlock> textBlocks) {
|
||||
|
||||
Comparator<TextPositionSequence> sortByX = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[0] - sequence2.getTextPositions()
|
||||
.get(0)
|
||||
.getPosition()[0]);
|
||||
Comparator<TextPositionSequence> sortByY = (sequence1, sequence2) -> (int) (sequence1.getTextPositions().get(0).getPosition()[1] - sequence2.getTextPositions()
|
||||
.get(0)
|
||||
.getPosition()[1]);
|
||||
|
||||
return textBlocks.stream().map(TextBlock::getSequences).flatMap(List::stream).sorted(sortByX.thenComparing(sortByY)).toList();
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock buildAtomicTextBlock(List<TextPositionSequence> sequences, DocumentGraphNode parent, Context context) {
|
||||
|
||||
SearchTextWithTextPositionModel searchTextWithTextPositionModel = searchTextWithTextPositionFactory.buildSearchTextToTextPositionModel(sequences);
|
||||
int offset = context.stringOffset().getAndAdd(searchTextWithTextPositionModel.getSearchText().length());
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(context.textBlockIdx.getAndIncrement())
|
||||
.parent(parent)
|
||||
.searchText(searchTextWithTextPositionModel.getSearchText())
|
||||
.lineBreaks(searchTextWithTextPositionModel.getLineBreaks())
|
||||
.positions(toRectangle2D(searchTextWithTextPositionModel.getPositions()))
|
||||
.stringIdxToPositionIdx(searchTextWithTextPositionModel.getStringCoordsToPositionCoords())
|
||||
.boundary(new Boundary(offset, offset + searchTextWithTextPositionModel.getSearchText().length()))
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private List<Rectangle2D> toRectangle2D(List<RedRectangle2D> positions) {
|
||||
|
||||
return positions.stream().map(r -> (Rectangle2D) new Rectangle2D.Double(r.getX(), r.getY(), r.getWidth(), r.getHeight())).toList();
|
||||
}
|
||||
|
||||
|
||||
private PageNode getPage(int pageIndex, Context context) {
|
||||
|
||||
return context.pages.stream()
|
||||
.filter(page -> page.getNumber() == pageIndex)
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
|
||||
}
|
||||
|
||||
|
||||
record Context(
|
||||
TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, AtomicInteger stringOffset, AtomicLong textBlockIdx) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+171
@@ -0,0 +1,171 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
import static java.lang.Math.toIntExact;
|
||||
import static java.lang.String.format;
|
||||
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.commons.lang3.NotImplementedException;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import com.google.common.primitives.Ints;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.AtomicTextBlockData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.DocumentData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.PageData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.data.TableOfContentsData;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.DocumentGraph;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.TableOfContents;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.DocumentGraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.NodeType;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.PageNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.ParagraphNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.SectionNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.AtomicTextBlock;
|
||||
import com.iqser.red.service.redaction.v1.server.exception.NotFoundException;
|
||||
|
||||
@Service
|
||||
public class DocumentGraphMapper {
|
||||
|
||||
public DocumentGraph toDocumentGraph(DocumentData documentData) {
|
||||
|
||||
Context context = new Context(documentData, new TableOfContents(), new LinkedList<>(), new LinkedList<>(), documentData.getAtomicTextBlocks());
|
||||
|
||||
context.pages.addAll(documentData.getPages().stream().map(pageData -> buildPage(pageData, context)).toList());
|
||||
buildNodesFromTableOfContents("", context);
|
||||
DocumentGraph documentGraph= DocumentGraph.builder()
|
||||
.numberOfPages(documentData.getPages().size())
|
||||
.pages(context.pages)
|
||||
.sections(context.sections)
|
||||
.tableOfContents(context.tableOfContents)
|
||||
.build();
|
||||
documentGraph.setText(documentGraph.buildTextBlock());
|
||||
return documentGraph;
|
||||
}
|
||||
|
||||
|
||||
private void buildNodesFromTableOfContents(String currentTocId, Context context) {
|
||||
List <TableOfContentsData.EntryData> entries;
|
||||
if(currentTocId.equals("")) {
|
||||
entries = context.documentData().getTableOfContents().getEntries();
|
||||
} else {
|
||||
entries = context.documentData().getTableOfContents().get(currentTocId).subEntries();
|
||||
}
|
||||
for (TableOfContentsData.EntryData entryData : entries) {
|
||||
|
||||
switch (entryData.type()) {
|
||||
case SECTION -> buildSection(entryData, currentTocId, context);
|
||||
case PARAGRAPH -> buildParagraph(entryData, currentTocId, context);
|
||||
default -> throw new NotImplementedException("Not yet implemented for type " + entryData.type());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private void buildSection(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
|
||||
|
||||
SectionNode section = SectionNode.builder()
|
||||
.entities(new LinkedList<>())
|
||||
.pages(new LinkedList<>())
|
||||
.paragraphs(new LinkedList<>())
|
||||
.tables(new LinkedList<>())
|
||||
.subSections(new LinkedList<>())
|
||||
.tableOfContents(context.tableOfContents())
|
||||
.numberOnPage(entryData.numberOnPage())
|
||||
.build();
|
||||
|
||||
context.sections().add(section);
|
||||
section.setHeadline(toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(entryData.atomicTextBlock())), section));
|
||||
|
||||
if (!currentTocId.equals("")) {
|
||||
SectionNode parent = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
|
||||
section.setParentSection(parent);
|
||||
parent.getSubSections().add(section);
|
||||
}
|
||||
|
||||
PageNode page = getPage(entryData.page(), context);
|
||||
page.getMainBody().add(section);
|
||||
section.getPages().add(page);
|
||||
|
||||
String sectionId = context.tableOfContents.createNewEntryAndReturnId(NodeType.SECTION, buildSummary(section.getHeadline()), section);
|
||||
section.setTocId(sectionId);
|
||||
buildNodesFromTableOfContents(sectionId, context);
|
||||
}
|
||||
|
||||
|
||||
private void buildParagraph(TableOfContentsData.EntryData entryData, String currentTocId, Context context) {
|
||||
|
||||
PageNode page = getPage(entryData.page(), context);
|
||||
SectionNode parentSection = (SectionNode) context.tableOfContents().getEntryById(currentTocId).node();
|
||||
ParagraphNode paragraph = ParagraphNode.builder().numberOnPage(entryData.numberOnPage()).page(page).parentSection(parentSection).build();
|
||||
AtomicTextBlock atomicTextBlock = toAtomicTextBlock(context.atomicTextBlockData.get(toIntExact(entryData.atomicTextBlock())), paragraph);
|
||||
paragraph.setAtomicTextBlock(atomicTextBlock);
|
||||
|
||||
if (!page.getMainBody().contains(parentSection)) {
|
||||
parentSection.getPages().add(page);
|
||||
}
|
||||
page.getMainBody().add(paragraph);
|
||||
|
||||
String tocId = context.tableOfContents.createNewChildEntryAndReturnId(currentTocId, NodeType.PARAGRAPH, buildSummary(atomicTextBlock), paragraph);
|
||||
paragraph.setTocId(tocId);
|
||||
}
|
||||
|
||||
|
||||
private PageNode buildPage(PageData p, Context context) {
|
||||
|
||||
PageNode page = PageNode.builder().height(p.getHeight()).width(p.getWidth()).number(p.getNumber()).mainBody(new LinkedList<>()).build();
|
||||
AtomicTextBlock header = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getHeader())), page);
|
||||
AtomicTextBlock footer = toAtomicTextBlock(context.atomicTextBlockData().get(toIntExact(p.getFooter())), page);
|
||||
page.setHeader(header);
|
||||
page.setFooter(footer);
|
||||
return page;
|
||||
}
|
||||
|
||||
|
||||
private static String buildSummary(com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock textBlock) {
|
||||
|
||||
if (textBlock == null) {
|
||||
return " probably a table";
|
||||
}
|
||||
|
||||
String[] words = textBlock.getFirstLine().toString().split(" ");
|
||||
int bound = Math.min(words.length, 4);
|
||||
List<String> list = new ArrayList<>(Arrays.asList(words).subList(0, bound));
|
||||
|
||||
return String.join(" ", list);
|
||||
}
|
||||
|
||||
|
||||
private AtomicTextBlock toAtomicTextBlock(AtomicTextBlockData atomicTextBlockData, DocumentGraphNode parent) {
|
||||
|
||||
return AtomicTextBlock.builder()
|
||||
.id(atomicTextBlockData.getId())
|
||||
.searchText(atomicTextBlockData.getSearchText())
|
||||
.boundary(new Boundary(atomicTextBlockData.getStart(), atomicTextBlockData.getEnd()))
|
||||
.lineBreaks(Ints.asList(atomicTextBlockData.getLineBreaks()))
|
||||
.positions(Arrays.stream(atomicTextBlockData.getPositions())
|
||||
.map(floatArr -> (Rectangle2D) new Rectangle2D.Float(floatArr[0], floatArr[1], floatArr[2], floatArr[3]))
|
||||
.toList())
|
||||
.stringIdxToPositionIdx(Ints.asList(atomicTextBlockData.getStringIdxToPositionIdx()))
|
||||
.parent(parent)
|
||||
.build();
|
||||
}
|
||||
|
||||
|
||||
private PageNode getPage(Long pageIndex, Context context) {
|
||||
|
||||
return context.pages.stream()
|
||||
.filter(page -> page.getNumber() == toIntExact(pageIndex))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new NotFoundException(format("Page with number %d not found", pageIndex)));
|
||||
}
|
||||
|
||||
|
||||
record Context(DocumentData documentData, TableOfContents tableOfContents, List<PageNode> pages, List<SectionNode> sections, List<AtomicTextBlockData> atomicTextBlockData) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.nodes.EntityNode;
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.textblock.TextBlock;
|
||||
|
||||
public class EntityEnrichmentUtility {
|
||||
|
||||
public static EntityNode enrichEntity(EntityNode entity, TextBlock textBlock) {
|
||||
|
||||
entity.setPositions(textBlock.getPositions(entity.getBoundary()));
|
||||
entity.setTextAfter(findTextAfter(entity.getBoundary().end(), textBlock));
|
||||
entity.setTextBefore(findTextBefore(entity.getBoundary().start(), textBlock));
|
||||
entity.setValue(textBlock.subSequence(entity.getBoundary()).toString());
|
||||
return entity;
|
||||
}
|
||||
|
||||
|
||||
private static CharSequence findTextAfter(int index, TextBlock textBlock) {
|
||||
|
||||
int nextLineBreak = textBlock.getNextLinebreak(index);
|
||||
return textBlock.subSequence(index, nextLineBreak);
|
||||
}
|
||||
|
||||
|
||||
private static CharSequence findTextBefore(int index, TextBlock textBlock) {
|
||||
|
||||
int previousLinebreak = textBlock.getPreviousLinebreak(index);
|
||||
return textBlock.subSequence(previousLinebreak, index);
|
||||
}
|
||||
|
||||
}
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.util.Comparator;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
|
||||
public class RangeComparators {
|
||||
|
||||
public static Comparator<Boundary> contained() {
|
||||
|
||||
return (range1, range2) -> {
|
||||
if (contained(range1, range2)) {
|
||||
return -1;
|
||||
} else if (contained(range2, range1)) {
|
||||
return 1;
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* @param range1 A Range
|
||||
* @param range2 Also Range
|
||||
* @return true, if range1 contains range2
|
||||
* false, otherwise
|
||||
*/
|
||||
public static boolean contained(Boundary range1, Boundary range2) {
|
||||
|
||||
return range1.start() <= range2.start() && range2.end() <= range1.end();
|
||||
}
|
||||
|
||||
public static boolean contained(Boundary range, int index) {
|
||||
|
||||
return range.start() <= index && index < range.end();
|
||||
}
|
||||
}
|
||||
+37
@@ -0,0 +1,37 @@
|
||||
package com.iqser.red.service.redaction.v1.server.document.services;
|
||||
|
||||
import java.util.LinkedList;
|
||||
import java.util.List;
|
||||
import java.util.regex.Matcher;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.document.graph.Boundary;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.utils.Patterns;
|
||||
|
||||
public class RegexMatcher {
|
||||
|
||||
public static boolean anyMatch(CharSequence searchText, String regexPattern) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
return pattern.matcher(searchText).find();
|
||||
}
|
||||
|
||||
|
||||
public static Boundary findFirstBoundary(String regexPattern, CharSequence searchText) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
Matcher matcher = pattern.matcher(searchText);
|
||||
return new Boundary(matcher.start(), matcher.end());
|
||||
}
|
||||
|
||||
public static List<Boundary> findBoundaries(String regexPattern, CharSequence searchText) {
|
||||
|
||||
var pattern = Patterns.getCompiledPattern(regexPattern, false);
|
||||
Matcher matcher = pattern.matcher(searchText);
|
||||
List<Boundary> boundaries = new LinkedList<>();
|
||||
while (matcher.find()) {
|
||||
boundaries.add(new Boundary(matcher.start(), matcher.end()));
|
||||
}
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
}
|
||||
+1
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.exception;
|
||||
public class NotFoundException extends RuntimeException {
|
||||
|
||||
public NotFoundException(String message) {
|
||||
|
||||
super(message);
|
||||
}
|
||||
|
||||
|
||||
+3
@@ -3,10 +3,13 @@ package com.iqser.red.service.redaction.v1.server.exception;
|
||||
public class RedactionException extends RuntimeException {
|
||||
|
||||
public RedactionException(Throwable cause) {
|
||||
|
||||
super("Could not parse document", cause);
|
||||
}
|
||||
|
||||
|
||||
public RedactionException() {
|
||||
|
||||
super("Could not parse document");
|
||||
}
|
||||
|
||||
|
||||
+1
@@ -3,6 +3,7 @@ package com.iqser.red.service.redaction.v1.server.exception;
|
||||
public class RulesValidationException extends RuntimeException {
|
||||
|
||||
public RulesValidationException(String message, Throwable t) {
|
||||
|
||||
super(message, t);
|
||||
}
|
||||
|
||||
|
||||
-52
@@ -1,52 +0,0 @@
|
||||
package com.iqser.red.service.redaction.v1.server.memory;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import java.text.CharacterIterator;
|
||||
import java.text.StringCharacterIterator;
|
||||
|
||||
@Slf4j
|
||||
public class MemoryStats {
|
||||
|
||||
|
||||
public static void printMemoryStats() {
|
||||
log.info("\n\n ------------------------------ \n" +
|
||||
" Used Memory: " + humanReadableByteCountBin(getUsedMemory()) + "\n" +
|
||||
" Free Memory: " + humanReadableByteCountBin(getFreeMemory()) + "\n" +
|
||||
" Total Memory: " + humanReadableByteCountBin(getTotalMemory()) + "\n" +
|
||||
" Max Memory: " + humanReadableByteCountBin(getMaxMemory()) + "\n" +
|
||||
"\n ------------------------------ \n");
|
||||
}
|
||||
|
||||
|
||||
public static String humanReadableByteCountBin(long bytes) {
|
||||
long absB = bytes == Long.MIN_VALUE ? Long.MAX_VALUE : Math.abs(bytes);
|
||||
if (absB < 1024) {
|
||||
return bytes + " B";
|
||||
}
|
||||
long value = absB;
|
||||
CharacterIterator ci = new StringCharacterIterator("KMGTPE");
|
||||
for (int i = 40; i >= 0 && absB > 0xfffccccccccccccL >> i; i -= 10) {
|
||||
value >>= 10;
|
||||
ci.next();
|
||||
}
|
||||
value *= Long.signum(bytes);
|
||||
return String.format("%.1f %ciB", value / 1024.0, ci.current());
|
||||
}
|
||||
|
||||
private static long getMaxMemory() {
|
||||
return Runtime.getRuntime().maxMemory();
|
||||
}
|
||||
|
||||
private static long getUsedMemory() {
|
||||
return getMaxMemory() - getFreeMemory();
|
||||
}
|
||||
|
||||
private static long getTotalMemory() {
|
||||
return Runtime.getRuntime().totalMemory();
|
||||
}
|
||||
|
||||
private static long getFreeMemory() {
|
||||
return Runtime.getRuntime().freeMemory();
|
||||
}
|
||||
}
|
||||
+95
-86
@@ -69,17 +69,17 @@ import org.apache.pdfbox.pdmodel.font.PDFontDescriptor;
|
||||
|
||||
/**
|
||||
* LEGACY text calculations which are known to be incorrect but are depended on by PDFTextStripper.
|
||||
*
|
||||
* <p>
|
||||
* This class exists only so that we don't break the code of users who have their own subclasses of
|
||||
* PDFTextStripper. It replaces the mostly empty implementation of showGlyph() in PDFStreamEngine
|
||||
* with a heuristic implementation which is backwards compatible.
|
||||
*
|
||||
* <p>
|
||||
* DO NOT USE THIS CODE UNLESS YOU ARE WORKING WITH PDFTextStripper.
|
||||
* THIS CODE IS DELIBERATELY INCORRECT, USE PDFStreamEngine INSTEAD.
|
||||
*/
|
||||
@SuppressWarnings({"PMD", "checkstyle:all"})
|
||||
class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
{
|
||||
class LegacyPDFStreamEngine extends PDFStreamEngine {
|
||||
|
||||
private static final Log LOG = LogFactory.getLog(LegacyPDFStreamEngine.class);
|
||||
|
||||
private int pageRotation;
|
||||
@@ -88,11 +88,12 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
private final GlyphList glyphList;
|
||||
private final Map<COSDictionary, Float> fontHeightMap = new WeakHashMap<COSDictionary, Float>();
|
||||
|
||||
|
||||
/**
|
||||
* Constructor.
|
||||
*/
|
||||
LegacyPDFStreamEngine() throws IOException
|
||||
{
|
||||
LegacyPDFStreamEngine() throws IOException {
|
||||
|
||||
addOperator(new BeginText());
|
||||
addOperator(new Concatenate());
|
||||
addOperator(new DrawObject()); // special text version
|
||||
@@ -122,6 +123,7 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
glyphList = new GlyphList(GlyphList.getAdobeGlyphList(), input);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* This will initialize and process the contents of the stream.
|
||||
*
|
||||
@@ -129,33 +131,27 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
* @throws java.io.IOException if there is an error accessing the stream.
|
||||
*/
|
||||
@Override
|
||||
public void processPage(PDPage page) throws IOException
|
||||
{
|
||||
public void processPage(PDPage page) throws IOException {
|
||||
|
||||
this.pageRotation = page.getRotation();
|
||||
this.pageSize = page.getCropBox();
|
||||
|
||||
if (pageSize.getLowerLeftX() == 0 && pageSize.getLowerLeftY() == 0)
|
||||
{
|
||||
if (pageSize.getLowerLeftX() == 0 && pageSize.getLowerLeftY() == 0) {
|
||||
translateMatrix = null;
|
||||
}
|
||||
else
|
||||
{
|
||||
} else {
|
||||
// translation matrix for cropbox
|
||||
translateMatrix = Matrix.getTranslateInstance(-pageSize.getLowerLeftX(), -pageSize.getLowerLeftY());
|
||||
}
|
||||
super.processPage(page);
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Called when a glyph is to be processed. The heuristic calculations here were originally
|
||||
* written by Ben Litchfield for PDFStreamEngine.
|
||||
*/
|
||||
@Override
|
||||
protected void showGlyph(Matrix textRenderingMatrix, PDFont font, int code,
|
||||
String unicode,
|
||||
Vector displacement)
|
||||
throws IOException
|
||||
{
|
||||
protected void showGlyph(Matrix textRenderingMatrix, PDFont font, int code, String unicode, Vector displacement) throws IOException {
|
||||
//
|
||||
// legacy calculations which were previously in PDFStreamEngine
|
||||
//
|
||||
@@ -173,25 +169,19 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
// the sorting algorithm is based on the width of the character. As the displacement
|
||||
// for vertical characters doesn't provide any suitable value for it, we have to
|
||||
// calculate our own
|
||||
if (font.isVertical())
|
||||
{
|
||||
if (font.isVertical()) {
|
||||
displacementX = font.getWidth(code) / 1000;
|
||||
// there may be an additional scaling factor for true type fonts
|
||||
TrueTypeFont ttf = null;
|
||||
if (font instanceof PDTrueTypeFont)
|
||||
{
|
||||
ttf = ((PDTrueTypeFont)font).getTrueTypeFont();
|
||||
}
|
||||
else if (font instanceof PDType0Font)
|
||||
{
|
||||
PDCIDFont cidFont = ((PDType0Font)font).getDescendantFont();
|
||||
if (cidFont instanceof PDCIDFontType2)
|
||||
{
|
||||
ttf = ((PDCIDFontType2)cidFont).getTrueTypeFont();
|
||||
if (font instanceof PDTrueTypeFont) {
|
||||
ttf = ((PDTrueTypeFont) font).getTrueTypeFont();
|
||||
} else if (font instanceof PDType0Font) {
|
||||
PDCIDFont cidFont = ((PDType0Font) font).getDescendantFont();
|
||||
if (cidFont instanceof PDCIDFontType2) {
|
||||
ttf = ((PDCIDFontType2) cidFont).getTrueTypeFont();
|
||||
}
|
||||
}
|
||||
if (ttf != null && ttf.getUnitsPerEm() != 1000)
|
||||
{
|
||||
if (ttf != null && ttf.getUnitsPerEm() != 1000) {
|
||||
displacementX *= 1000f / ttf.getUnitsPerEm();
|
||||
}
|
||||
}
|
||||
@@ -219,8 +209,7 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
// (modified) width and height calculations
|
||||
float dxDisplay = nextX - textRenderingMatrix.getTranslateX();
|
||||
Float fontHeight = fontHeightMap.get(font.getCOSObject());
|
||||
if (fontHeight == null)
|
||||
{
|
||||
if (fontHeight == null) {
|
||||
fontHeight = computeFontHeight(font);
|
||||
fontHeightMap.put(font.getCOSObject(), fontHeight);
|
||||
}
|
||||
@@ -237,30 +226,24 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
// saved).
|
||||
|
||||
float glyphSpaceToTextSpaceFactor = 1 / 1000f;
|
||||
if (font instanceof PDType3Font)
|
||||
{
|
||||
if (font instanceof PDType3Font) {
|
||||
glyphSpaceToTextSpaceFactor = font.getFontMatrix().getScaleX();
|
||||
}
|
||||
|
||||
float spaceWidthText = 0;
|
||||
try
|
||||
{
|
||||
try {
|
||||
// to avoid crash as described in PDFBOX-614, see what the space displacement should be
|
||||
spaceWidthText = font.getSpaceWidth() * glyphSpaceToTextSpaceFactor;
|
||||
}
|
||||
catch (Throwable exception)
|
||||
{
|
||||
} catch (Throwable exception) {
|
||||
LOG.warn(exception, exception);
|
||||
}
|
||||
|
||||
if (spaceWidthText == 0)
|
||||
{
|
||||
if (spaceWidthText == 0) {
|
||||
spaceWidthText = font.getAverageFontWidth() * glyphSpaceToTextSpaceFactor;
|
||||
// the average space width appears to be higher than necessary so make it smaller
|
||||
spaceWidthText *= .80f;
|
||||
}
|
||||
if (spaceWidthText == 0)
|
||||
{
|
||||
if (spaceWidthText == 0) {
|
||||
spaceWidthText = 1.0f; // if could not find font, use a generic value
|
||||
}
|
||||
|
||||
@@ -273,15 +256,11 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
// when there is no Unicode mapping available, Acrobat simply coerces the character code
|
||||
// into Unicode, so we do the same. Subclasses of PDFStreamEngine don't necessarily want
|
||||
// this, which is why we leave it until this point in PDFTextStreamEngine.
|
||||
if (unicodeMapping == null)
|
||||
{
|
||||
if (font instanceof PDSimpleFont)
|
||||
{
|
||||
if (unicodeMapping == null) {
|
||||
if (font instanceof PDSimpleFont) {
|
||||
char c = (char) code;
|
||||
unicodeMapping = new String(new char[] { c });
|
||||
}
|
||||
else
|
||||
{
|
||||
unicodeMapping = new String(new char[]{c});
|
||||
} else {
|
||||
// Acrobat doesn't seem to coerce composite font's character codes, instead it
|
||||
// skips them. See the "allah2.pdf" TestTextStripper file.
|
||||
return;
|
||||
@@ -290,88 +269,118 @@ class LegacyPDFStreamEngine extends PDFStreamEngine
|
||||
|
||||
// adjust for cropbox if needed
|
||||
Matrix translatedTextRenderingMatrix;
|
||||
if (translateMatrix == null)
|
||||
{
|
||||
if (translateMatrix == null) {
|
||||
translatedTextRenderingMatrix = textRenderingMatrix;
|
||||
}
|
||||
else
|
||||
{
|
||||
} else {
|
||||
translatedTextRenderingMatrix = Matrix.concatenate(translateMatrix, textRenderingMatrix);
|
||||
nextX -= pageSize.getLowerLeftX();
|
||||
nextY -= pageSize.getLowerLeftY();
|
||||
}
|
||||
|
||||
processTextPosition(new TextPosition(pageRotation, pageSize.getWidth(),
|
||||
pageSize.getHeight(), translatedTextRenderingMatrix, nextX, nextY,
|
||||
Math.abs(dyDisplay), dxDisplay,
|
||||
Math.abs(spaceWidthDisplay), unicodeMapping, new int[] { code }, font,
|
||||
fontSize,
|
||||
(int)(fontSize * textMatrix.getScalingFactorX())));
|
||||
// This is a hack for unicode letter with 2 chars e.g. RA see unicodeProblem.pdf
|
||||
if (unicodeMapping.length() == 2) {
|
||||
processTextPosition(new TextPosition(pageRotation,
|
||||
pageSize.getWidth(),
|
||||
pageSize.getHeight(),
|
||||
translatedTextRenderingMatrix,
|
||||
nextX,
|
||||
nextY,
|
||||
Math.abs(dyDisplay),
|
||||
dxDisplay,
|
||||
Math.abs(spaceWidthDisplay),
|
||||
Character.toString(unicodeMapping.charAt(0)),
|
||||
new int[]{code},
|
||||
font,
|
||||
fontSize,
|
||||
(int) (fontSize * textMatrix.getScalingFactorX())));
|
||||
processTextPosition(new TextPosition(pageRotation,
|
||||
pageSize.getWidth(),
|
||||
pageSize.getHeight(),
|
||||
translatedTextRenderingMatrix,
|
||||
nextX,
|
||||
nextY,
|
||||
Math.abs(dyDisplay),
|
||||
dxDisplay,
|
||||
Math.abs(spaceWidthDisplay),
|
||||
Character.toString(unicodeMapping.charAt(1)),
|
||||
new int[]{code},
|
||||
font,
|
||||
fontSize,
|
||||
(int) (fontSize * textMatrix.getScalingFactorX())));
|
||||
} else {
|
||||
|
||||
processTextPosition(new TextPosition(pageRotation,
|
||||
pageSize.getWidth(),
|
||||
pageSize.getHeight(),
|
||||
translatedTextRenderingMatrix,
|
||||
nextX,
|
||||
nextY,
|
||||
Math.abs(dyDisplay),
|
||||
dxDisplay,
|
||||
Math.abs(spaceWidthDisplay),
|
||||
unicodeMapping,
|
||||
new int[]{code},
|
||||
font,
|
||||
fontSize,
|
||||
(int) (fontSize * textMatrix.getScalingFactorX())));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Compute the font height. Override this if you want to use own calculations.
|
||||
*
|
||||
*
|
||||
* @param font the font.
|
||||
* @return the font height.
|
||||
*
|
||||
* @throws IOException if there is an error while getting the font bounding box.
|
||||
*/
|
||||
protected float computeFontHeight(PDFont font) throws IOException
|
||||
{
|
||||
protected float computeFontHeight(PDFont font) throws IOException {
|
||||
|
||||
BoundingBox bbox = font.getBoundingBox();
|
||||
if (bbox.getLowerLeftY() < Short.MIN_VALUE)
|
||||
{
|
||||
if (bbox.getLowerLeftY() < Short.MIN_VALUE) {
|
||||
// PDFBOX-2158 and PDFBOX-3130
|
||||
// files by Salmat eSolutions / ClibPDF Library
|
||||
bbox.setLowerLeftY(- (bbox.getLowerLeftY() + 65536));
|
||||
bbox.setLowerLeftY(-(bbox.getLowerLeftY() + 65536));
|
||||
}
|
||||
// 1/2 the bbox is used as the height todo: why?
|
||||
float glyphHeight = bbox.getHeight() / 2;
|
||||
|
||||
// sometimes the bbox has very high values, but CapHeight is OK
|
||||
PDFontDescriptor fontDescriptor = font.getFontDescriptor();
|
||||
if (fontDescriptor != null)
|
||||
{
|
||||
if (fontDescriptor != null) {
|
||||
float capHeight = fontDescriptor.getCapHeight();
|
||||
if (Float.compare(capHeight, 0) != 0 &&
|
||||
(capHeight < glyphHeight || Float.compare(glyphHeight, 0) == 0))
|
||||
{
|
||||
if (Float.compare(capHeight, 0) != 0 && (capHeight < glyphHeight || Float.compare(glyphHeight, 0) == 0)) {
|
||||
glyphHeight = capHeight;
|
||||
}
|
||||
// PDFBOX-3464, PDFBOX-4480, PDFBOX-4553:
|
||||
// sometimes even CapHeight has very high value, but Ascent and Descent are ok
|
||||
float ascent = fontDescriptor.getAscent();
|
||||
float descent = fontDescriptor.getDescent();
|
||||
if (capHeight > ascent && ascent > 0 && descent < 0 &&
|
||||
((ascent - descent) / 2 < glyphHeight || Float.compare(glyphHeight, 0) == 0))
|
||||
{
|
||||
if (capHeight > ascent && ascent > 0 && descent < 0 && ((ascent - descent) / 2 < glyphHeight || Float.compare(glyphHeight, 0) == 0)) {
|
||||
glyphHeight = (ascent - descent) / 2;
|
||||
}
|
||||
}
|
||||
|
||||
// transformPoint from glyph space -> text space
|
||||
float height;
|
||||
if (font instanceof PDType3Font)
|
||||
{
|
||||
if (font instanceof PDType3Font) {
|
||||
height = font.getFontMatrix().transformPoint(0, glyphHeight).y;
|
||||
}
|
||||
else
|
||||
{
|
||||
} else {
|
||||
height = glyphHeight / 1000;
|
||||
}
|
||||
|
||||
return height;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* A method provided as an event interface to allow a subclass to perform some specific
|
||||
* functionality when text needs to be processed.
|
||||
*
|
||||
* @param text The text to be processed.
|
||||
*/
|
||||
protected void processTextPosition(TextPosition text)
|
||||
{
|
||||
protected void processTextPosition(TextPosition text) {
|
||||
// subclasses can override to provide specific functionality
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
+12
-22
@@ -1,8 +1,10 @@
|
||||
package com.iqser.red.service.redaction.v1.server.parsing;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
|
||||
import org.apache.pdfbox.text.PDFTextStripperByArea;
|
||||
import org.apache.pdfbox.text.TextPosition;
|
||||
|
||||
@@ -18,19 +20,19 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
|
||||
@Setter
|
||||
private int pageNumber;
|
||||
|
||||
|
||||
public PDFAreaTextStripper() throws IOException {
|
||||
|
||||
}
|
||||
|
||||
|
||||
@Override
|
||||
public void writeString(String text, List<TextPosition> textPositions) throws IOException {
|
||||
|
||||
int startIndex = 0;
|
||||
for (int i = 0; i <= textPositions.size() - 1; i++) {
|
||||
|
||||
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
|
||||
.getUnicode()
|
||||
.equals("\u00A0"))) {
|
||||
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0"))) {
|
||||
startIndex++;
|
||||
continue;
|
||||
}
|
||||
@@ -38,32 +40,23 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
|
||||
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
|
||||
if (i > 0 && textPositions.get(i).getX() < textPositions.get(i - 1).getX()) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
}
|
||||
startIndex = i;
|
||||
}
|
||||
|
||||
|
||||
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i).getX() > textPositions.get(i - 1).getEndX() + 1) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
}
|
||||
startIndex = i;
|
||||
}
|
||||
|
||||
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
|
||||
.getUnicode()
|
||||
.equals("\u00A0")) && i <= textPositions.size() - 2) {
|
||||
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0")) && i <= textPositions.size() - 2) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
}
|
||||
startIndex = i + 1;
|
||||
@@ -71,14 +64,10 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
|
||||
}
|
||||
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
|
||||
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
|
||||
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1).getUnicode().equals("\u00A0"))) {
|
||||
sublist = sublist.subList(0, sublist.size() - 1);
|
||||
}
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals("\u00A0")))) {
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
}
|
||||
super.writeString(text);
|
||||
@@ -86,6 +75,7 @@ public class PDFAreaTextStripper extends PDFTextStripperByArea {
|
||||
|
||||
|
||||
public void clearPositions() {
|
||||
|
||||
textPositionSequences = new ArrayList<>();
|
||||
}
|
||||
|
||||
|
||||
+30
-100
@@ -1,53 +1,30 @@
|
||||
package com.iqser.red.service.redaction.v1.server.parsing;
|
||||
|
||||
import java.awt.geom.Point2D;
|
||||
import java.awt.geom.Rectangle2D;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.commons.lang3.reflect.FieldUtils;
|
||||
import org.apache.pdfbox.contentstream.operator.Operator;
|
||||
import org.apache.pdfbox.contentstream.operator.OperatorName;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColorN;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingColorSpace;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceCMYKColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceGrayColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetNonStrokingDeviceRGBColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColorN;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingColorSpace;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceCMYKColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceGrayColor;
|
||||
import org.apache.pdfbox.contentstream.operator.color.SetStrokingDeviceRGBColor;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetFlatness;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetLineCapStyle;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetLineDashPattern;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetLineJoinStyle;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetLineMiterLimit;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetLineWidth;
|
||||
import org.apache.pdfbox.contentstream.operator.state.SetRenderingIntent;
|
||||
import org.apache.pdfbox.contentstream.operator.text.SetFontAndSize;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.cos.COSNumber;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
|
||||
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
|
||||
import org.apache.pdfbox.text.TextPosition;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.RedTextPosition;
|
||||
import com.iqser.red.service.redaction.v1.server.parsing.model.TextPositionSequence;
|
||||
import com.iqser.red.service.redaction.v1.server.redaction.model.PdfImage;
|
||||
import com.iqser.red.service.redaction.v1.server.tableextraction.model.Ruling;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import org.apache.pdfbox.contentstream.operator.Operator;
|
||||
import org.apache.pdfbox.contentstream.operator.OperatorName;
|
||||
import org.apache.pdfbox.contentstream.operator.color.*;
|
||||
import org.apache.pdfbox.contentstream.operator.state.*;
|
||||
import org.apache.pdfbox.contentstream.operator.text.SetFontAndSize;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSNumber;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.text.TextPosition;
|
||||
|
||||
import java.awt.geom.Point2D;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
|
||||
@Slf4j
|
||||
public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
|
||||
@@ -66,8 +43,6 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
private int minCharHeight;
|
||||
@Getter
|
||||
private int maxCharHeight;
|
||||
@Getter
|
||||
private List<PdfImage> images = new ArrayList<>();
|
||||
|
||||
private float path_x;
|
||||
private float path_y;
|
||||
@@ -183,42 +158,12 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
graphicsPath.clear();
|
||||
break;
|
||||
|
||||
// case OperatorName.DRAW_OBJECT:
|
||||
// processImageOperation(arguments);
|
||||
// break;
|
||||
|
||||
}
|
||||
|
||||
super.processOperator(operator, arguments);
|
||||
}
|
||||
|
||||
|
||||
protected void processImageOperation(List<COSBase> arguments) {
|
||||
|
||||
try {
|
||||
COSName objectName = (COSName) arguments.get(0);
|
||||
PDXObject xobject = getResources().getXObject(objectName);
|
||||
if (xobject instanceof PDImageXObject) {
|
||||
PDImageXObject image = (PDImageXObject) xobject;
|
||||
Matrix ctmNew = getGraphicsState().getCurrentTransformationMatrix();
|
||||
|
||||
Rectangle2D rect = new Rectangle2D.Float(ctmNew.getTranslateX(), ctmNew.getTranslateY(), ctmNew.getScaleX(), ctmNew.getScaleY());
|
||||
|
||||
// Memory Hack - sofReference kills me
|
||||
FieldUtils.writeField(image, "cachedImageSubsampling", -1, true);
|
||||
|
||||
if (rect.getHeight() > 2 && rect.getWidth() > 2) {
|
||||
this.images.add(new PdfImage(image.getImage(), rect, pageNumber, image.getImage()
|
||||
.getColorModel()
|
||||
.hasAlpha()));
|
||||
}
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.warn("Problem during image extraction: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private float floatValue(COSBase value) {
|
||||
|
||||
if (value instanceof COSNumber) {
|
||||
@@ -239,14 +184,11 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
|
||||
try {
|
||||
if (stroke && !getGraphicsState().getStrokingColor().isPattern() && getGraphicsState().getStrokingColor()
|
||||
.toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor()
|
||||
.isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
|
||||
.toRGB() == 0 || !stroke && !getGraphicsState().getNonStrokingColor().isPattern() && getGraphicsState().getNonStrokingColor().toRGB() == 0) {
|
||||
rulings.addAll(path);
|
||||
}
|
||||
} catch (UnsupportedOperationException e) {
|
||||
log.error("UnsupportedOperationException: " + getGraphicsState().getStrokingColor()
|
||||
.getColorSpace()
|
||||
.getName() + " or " + getGraphicsState().getNonStrokingColor()
|
||||
log.debug("UnsupportedOperationException: " + getGraphicsState().getStrokingColor().getColorSpace().getName() + " or " + getGraphicsState().getNonStrokingColor()
|
||||
.getColorSpace()
|
||||
.getName() + " does not support toRGB");
|
||||
}
|
||||
@@ -259,6 +201,8 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
int startIndex = 0;
|
||||
RedTextPosition previous = null;
|
||||
|
||||
textPositions.sort(Comparator.comparing(TextPosition::getXDirAdj));
|
||||
|
||||
for (int i = 0; i <= textPositions.size() - 1; i++) {
|
||||
|
||||
if (!textPositionSequences.isEmpty()) {
|
||||
@@ -283,9 +227,7 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
maxCharHeight = charHeight;
|
||||
}
|
||||
|
||||
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
|
||||
.getUnicode()
|
||||
.equals("\u00A0") || textPositions.get(i).getUnicode().equals("\t"))) {
|
||||
if (i == 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0") || textPositions.get(i).getUnicode().equals("\t"))) {
|
||||
startIndex++;
|
||||
continue;
|
||||
}
|
||||
@@ -293,9 +235,7 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
// Strange but sometimes this is happening, for example: Metolachlor2.pdf
|
||||
if (i > 0 && textPositions.get(i).getXDirAdj() < textPositions.get(i - 1).getXDirAdj()) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals("\t")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
@@ -303,12 +243,9 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
startIndex = i;
|
||||
}
|
||||
|
||||
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i)
|
||||
.getX() > textPositions.get(i - 1).getEndX() + 1) {
|
||||
if (textPositions.get(i).getRotation() == 0 && i > 0 && textPositions.get(i).getX() > textPositions.get(i - 1).getEndX() + 1) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals("\t")))) {
|
||||
textPositionSequences.add(new TextPositionSequence(sublist, pageNumber));
|
||||
@@ -316,15 +253,11 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
startIndex = i;
|
||||
}
|
||||
|
||||
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i)
|
||||
.getUnicode()
|
||||
.equals("\u00A0") || textPositions.get(i)
|
||||
if (i > 0 && (textPositions.get(i).getUnicode().equals(" ") || textPositions.get(i).getUnicode().equals("\u00A0") || textPositions.get(i)
|
||||
.getUnicode()
|
||||
.equals("\t")) && i <= textPositions.size() - 2) {
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, i);
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals("\t")))) {
|
||||
|
||||
@@ -343,17 +276,15 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
}
|
||||
|
||||
List<TextPosition> sublist = textPositions.subList(startIndex, textPositions.size());
|
||||
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1)
|
||||
.getUnicode()
|
||||
.equals(" ") || sublist.get(sublist.size() - 1)
|
||||
if (!sublist.isEmpty() && (sublist.get(sublist.size() - 1).getUnicode().equals(" ") || sublist.get(sublist.size() - 1)
|
||||
.getUnicode()
|
||||
.equals("\u00A0") || sublist.get(sublist.size() - 1).getUnicode().equals("\t"))) {
|
||||
sublist = sublist.subList(0, sublist.size() - 1);
|
||||
}
|
||||
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0)
|
||||
if (!(sublist.isEmpty() || sublist.size() == 1 && (sublist.get(0).getUnicode().equals(" ") || sublist.get(0).getUnicode().equals("\u00A0") || sublist.get(0)
|
||||
.getUnicode()
|
||||
.equals("\u00A0") || sublist.get(0).getUnicode().equals("\t")))) {
|
||||
.equals("\t")))) {
|
||||
if (previous != null && sublist.get(0).getYDirAdj() == previous.getYDirAdj() && sublist.get(0)
|
||||
.getXDirAdj() - (previous.getXDirAdj() + previous.getWidthDirAdj()) < 0.01) {
|
||||
for (TextPosition t : sublist) {
|
||||
@@ -375,7 +306,6 @@ public class PDFLinesTextStripper extends PDFTextStripper {
|
||||
minCharHeight = Integer.MAX_VALUE;
|
||||
maxCharHeight = 0;
|
||||
textPositionSequences.clear();
|
||||
images = new ArrayList<>();
|
||||
rulings.clear();
|
||||
graphicsPath.clear();
|
||||
path_x = 0.0f;
|
||||
|
||||
+569
-683
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user