diff --git a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/OffsetString.java b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/OffsetString.java index 4c12fe9a..b209787e 100644 --- a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/OffsetString.java +++ b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/OffsetString.java @@ -23,5 +23,17 @@ public class OffsetString { return new OffsetString(trimmed, newStart, newEnd); } + + public OffsetString replaceAll(String regex, String replacement) { + + String trimmed = this.value.replaceAll(regex, replacement); + int indexInUntrimmed = this.value.indexOf(trimmed); + + int newStart = this.start + indexInUntrimmed; + int newEnd = newStart + trimmed.length(); + + return new OffsetString(trimmed, newStart, newEnd); + } + } diff --git a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java index 5b4d0605..da5d3e88 100644 --- a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java +++ b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/model/Section.java @@ -208,7 +208,7 @@ public class Section { return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && value.equals(attribute.getValue())); } - + @SuppressWarnings("unused") @WhenCondition public boolean fileAttributeContainsAnyOf(@Argument(ArgumentType.FILE_ATTRIBUTE) String label, @Argument(ArgumentType.STRING) String... values) { @@ -218,6 +218,7 @@ public class Section { return fileAttributes != null && fileAttributes.stream().anyMatch(attribute -> label.equals(attribute.getLabel()) && valueSet.contains(attribute.getValue())); } + @SuppressWarnings("unused") @WhenCondition public boolean fileAttributeByIdEqualsIgnoreCase(@Argument(ArgumentType.FILE_ATTRIBUTE) String id, @Argument(ArgumentType.STRING) String value) { @@ -558,6 +559,35 @@ public class Section { } + @ThenAction + @SuppressWarnings("unused") + public void redactByRegExWithNewlines(@Argument(ArgumentType.REGEX) String pattern, + @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, + @Argument(ArgumentType.INTEGER) int group, + @Argument(ArgumentType.TYPE) String asType, + @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, + @Argument(ArgumentType.STRING) String reason, + @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { + + redactByRegExWithNewlines(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false, false); + } + + + @ThenAction + @SuppressWarnings("unused") + public void redactByRegExWithNewlines(@Argument(ArgumentType.REGEX) String pattern, + @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, + @Argument(ArgumentType.INTEGER) int group, + @Argument(ArgumentType.TYPE) String asType, + @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, + @Argument(ArgumentType.BOOLEAN) boolean onlyExactMatch, + @Argument(ArgumentType.STRING) String reason, + @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { + + redactByRegExWithNewlines(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false, onlyExactMatch); + } + + @ThenAction @SuppressWarnings("unused") public void redactByRegEx(@Argument(ArgumentType.REGEX) String pattern, @@ -568,21 +598,7 @@ public class Section { @Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { - redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false); - } - - - @ThenAction - @SuppressWarnings("unused") - public void redactByRegExWithNewlines(@Argument(ArgumentType.REGEX) String pattern, - @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, - @Argument(ArgumentType.INTEGER) int group, - @Argument(ArgumentType.TYPE) String asType, - @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, - @Argument(ArgumentType.STRING) String reason, - @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { - - redactByRegExWithNewlines(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false); + redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, false, false); } @@ -597,7 +613,23 @@ public class Section { @Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { - redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, skipRemoveEntitiesContainedInLarger); + redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, skipRemoveEntitiesContainedInLarger, false); + } + + + @ThenAction + @SuppressWarnings("unused") + public void redactByRegEx(@Argument(ArgumentType.REGEX) String pattern, + @Argument(ArgumentType.BOOLEAN) boolean patternCaseInsensitive, + @Argument(ArgumentType.INTEGER) int group, + @Argument(ArgumentType.TYPE) String asType, + @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, + @Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger, + @Argument(ArgumentType.BOOLEAN) boolean onlyExactMatch, + @Argument(ArgumentType.STRING) String reason, + @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { + + redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, legalBasis, true, skipRemoveEntitiesContainedInLarger, onlyExactMatch); } @@ -610,7 +642,7 @@ public class Section { @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, @Argument(ArgumentType.STRING) String reason) { - redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false, false); + redactByRegEx(pattern, patternCaseInsensitive, group, asType, ruleNumber, reason, null, false, false, false); } @@ -625,7 +657,7 @@ public class Section { @Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { - redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, legalBasis, true, false, false, false); + redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, legalBasis, true, false, false, false, false); } @@ -640,7 +672,7 @@ public class Section { @Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.LEGAL_BASIS) String legalBasis) { - redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, legalBasis, true, false, false, false); + redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, legalBasis, true, false, false, false, false); } @@ -669,7 +701,9 @@ public class Section { legalBasis, true, skipRemoveEntitiesContainedInLarger, - sortedResult, false); + sortedResult, + false, + false); } @@ -701,10 +735,46 @@ public class Section { legalBasis, true, skipRemoveEntitiesContainedInLarger, - sortedResult, ignoreTables); + sortedResult, + ignoreTables, + false); } + @ThenAction + @SuppressWarnings("unused") + public void redactBetween(@Argument(ArgumentType.STRING) String start, + @Argument(ArgumentType.STRING) String stop, + @Argument(ArgumentType.BOOLEAN) boolean includeStart, + @Argument(ArgumentType.BOOLEAN) boolean includeStop, + @Argument(ArgumentType.TYPE) String asType, + @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, + @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere, + @Argument(ArgumentType.BOOLEAN) boolean excludeHeadLine, + @Argument(ArgumentType.STRING) String reason, + @Argument(ArgumentType.LEGAL_BASIS) String legalBasis, + @Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger, + @Argument(ArgumentType.BOOLEAN) boolean sortedResult, + @Argument(ArgumentType.BOOLEAN) boolean ignoreTables, + @Argument(ArgumentType.BOOLEAN) boolean onlyExactMatch) { + + redactBetween(start, + stop, + includeStart, + includeStop, + asType, + ruleNumber, + redactEverywhere, + excludeHeadLine, + reason, + legalBasis, + true, + skipRemoveEntitiesContainedInLarger, + sortedResult, + ignoreTables, + onlyExactMatch); + } + @ThenAction @SuppressWarnings("unused") @@ -733,7 +803,9 @@ public class Section { legalBasis, true, skipRemoveEntitiesContainedInLarger, - sortedResult, false); + sortedResult, + false, + false); } @@ -751,7 +823,8 @@ public class Section { @Argument(ArgumentType.STRING) String reason, @Argument(ArgumentType.STRING) String legalBasis, @Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger, - @Argument(ArgumentType.BOOLEAN) boolean sortedResult) { + @Argument(ArgumentType.BOOLEAN) boolean sortedResult, + @Argument(ArgumentType.BOOLEAN) boolean onlyExactMatch) { String startValue = getFirstRexExMatch(searchText, startPattern, startPatternCaseInsensitive, startGroup); @@ -776,11 +849,46 @@ public class Section { legalBasis, true, skipRemoveEntitiesContainedInLarger, - sortedResult, false); + sortedResult, + false, + onlyExactMatch); } } + @ThenAction + public void redactBetweenRegexes(@Argument(ArgumentType.STRING) String startPattern, + @Argument(ArgumentType.BOOLEAN) boolean startPatternCaseInsensitive, + @Argument(ArgumentType.INTEGER) int startGroup, + @Argument(ArgumentType.BOOLEAN) boolean includeStart, + @Argument(ArgumentType.STRING) String stopPattern, + @Argument(ArgumentType.BOOLEAN) boolean stopPatternCaseInsensitive, + @Argument(ArgumentType.INTEGER) int stopGroup, + @Argument(ArgumentType.BOOLEAN) boolean includeStop, + @Argument(ArgumentType.RULE_NUMBER) int ruleNumber, + @Argument(ArgumentType.TYPE) String type, + @Argument(ArgumentType.STRING) String reason, + @Argument(ArgumentType.STRING) String legalBasis, + @Argument(ArgumentType.BOOLEAN) boolean skipRemoveEntitiesContainedInLarger, + @Argument(ArgumentType.BOOLEAN) boolean sortedResult) { + + redactBetweenRegexes(startPattern, + startPatternCaseInsensitive, + startGroup, + includeStart, + stopPattern, + stopPatternCaseInsensitive, + stopGroup, + includeStop, + ruleNumber, + type, + reason, + legalBasis, + skipRemoveEntitiesContainedInLarger, + sortedResult, false); + } + + @Deprecated @ThenAction @SuppressWarnings("unused") @@ -791,19 +899,7 @@ public class Section { @Argument(ArgumentType.BOOLEAN) boolean redactEverywhere, @Argument(ArgumentType.STRING) String reason) { - redactBetween(start, - stop, - false, - false, - asType, - ruleNumber, - redactEverywhere, - false, - reason, - null, - false, - false, - false, false); + redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, false, reason, null, false, false, false, false, false); } @@ -817,19 +913,7 @@ public class Section { @Argument(ArgumentType.BOOLEAN) boolean excludeHeadLine, @Argument(ArgumentType.STRING) String reason) { - redactBetween(start, - stop, - false, - false, - asType, - ruleNumber, - redactEverywhere, - excludeHeadLine, - reason, - null, - false, - false, - false, false); + redactBetween(start, stop, false, false, asType, ruleNumber, redactEverywhere, excludeHeadLine, reason, null, false, false, false, false, false); } @@ -1449,7 +1533,8 @@ public class Section { if (StringUtils.isNotBlank(stringOffset.getValue())) { var trimmedOffsetString = stringOffset.trim(); Set found = findEntities(trimmedOffsetString.getValue(), asType, false, true, ruleNumber, reason, legalBasis, Engine.RULE, false).stream() - .filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()).collect(Collectors.toSet()); + .filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()) + .collect(Collectors.toSet()); found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger)); EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary); @@ -1463,7 +1548,16 @@ public class Section { } - private void redactByRegExWithNewlines(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis, boolean redaction, boolean skipRemoveEntitiesContainedInLarger) { + private void redactByRegExWithNewlines(String pattern, + boolean patternCaseInsensitive, + int group, + String asType, + int ruleNumber, + String reason, + String legalBasis, + boolean redaction, + boolean skipRemoveEntitiesContainedInLarger, + boolean onlyExactMatch) { Pattern compiledPattern = Patterns.getCompiledMultilinePattern(pattern, patternCaseInsensitive); @@ -1471,8 +1565,16 @@ public class Section { while (matcher.find()) { String match = matcher.group(group); + int start = matcher.start(group); + int end = matcher.end(group); + + OffsetString offsetString = new OffsetString(match, start, end); + var trimmedOffsetString = offsetString.replaceAll("\\n", " ").trim(); + if (StringUtils.isNotBlank(match)) { - Set found = findEntities(match.replaceAll("\\n", " ").trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false); + Set found = findEntities(trimmedOffsetString.getValue(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false).stream() + .filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()) + .collect(Collectors.toSet()); found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger)); EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary); } @@ -1480,7 +1582,16 @@ public class Section { } - private void redactByRegEx(String pattern, boolean patternCaseInsensitive, int group, String asType, int ruleNumber, String reason, String legalBasis, boolean redaction, boolean skipRemoveEntitiesContainedInLarger) { + private void redactByRegEx(String pattern, + boolean patternCaseInsensitive, + int group, + String asType, + int ruleNumber, + String reason, + String legalBasis, + boolean redaction, + boolean skipRemoveEntitiesContainedInLarger, + boolean onlyExactMatch) { Pattern compiledPattern = Patterns.getCompiledPattern(pattern, patternCaseInsensitive); @@ -1488,8 +1599,17 @@ public class Section { while (matcher.find()) { String match = matcher.group(group); + + int start = matcher.start(group); + int end = matcher.end(group); + + OffsetString offsetString = new OffsetString(match, start, end); + var trimmedOffsetString = offsetString.trim(); + if (StringUtils.isNotBlank(match)) { - Set found = findEntities(match.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false); + Set found = findEntities(trimmedOffsetString.getValue(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false).stream() + .filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()) + .collect(Collectors.toSet()); found.forEach(f -> f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger)); EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary); } @@ -1526,60 +1646,65 @@ public class Section { boolean redaction, boolean skipRemoveEntitiesContainedInLarger, boolean sortedResult, - boolean ignoreTables) { + boolean ignoreTables, + boolean onlyExactMatch) { - - if(isInTable && ignoreTables){ + if (isInTable && ignoreTables) { return; } - String[] values = new String[1]; + List values = new ArrayList<>(); if (start.isEmpty() && stop.isEmpty()) { if (excludeHeadLine && searchText.contains(headline)) { - values[0] = StringUtils.substringAfter(searchText, headline); + values.add(OffsetStringUtils.substringAfter(searchText, headline)); } else { - values[0] = searchText; + values.add(new OffsetString(searchText, 0, searchText.length())); } } else if (start.isEmpty() && searchText.contains(stop)) { - values[0] = StringUtils.substringBefore(searchText, stop); + values.add(OffsetStringUtils.substringBefore(searchText, stop)); } else if (stop.isEmpty() && searchText.contains(start)) { - values[0] = StringUtils.substringAfter(searchText, start); + values.add(OffsetStringUtils.substringAfter(searchText, start)); } else { - values = StringUtils.substringsBetween(searchText, start, stop); + var stringsBetween = OffsetStringUtils.substringsBetween(searchText, start, stop); + if(stringsBetween != null) { + values.addAll(stringsBetween); + } } - if (values != null) { - for (String value : values) { - if (StringUtils.isNotBlank(value)) { + for (OffsetString value : values) { + if (StringUtils.isNotBlank(value.getValue())) { - String searchString = value; + OffsetString searchString = value; - if (!start.isEmpty() && includeStart) { - searchString = start + searchString; + if (!start.isEmpty() && includeStart) { + searchString = new OffsetString(start + searchString, searchString.getStart() - start.length(), searchString.getEnd()); + } + + if (!stop.isEmpty() && includeStop) { + searchString = new OffsetString(searchString + stop, searchString.getStart(), searchString.getEnd() + stop.length()); + } + + var trimmedOffsetString = searchString.trim(); + + Set found = findEntities(trimmedOffsetString.getValue(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false).stream() + .filter(f -> !onlyExactMatch || f.getStart() == trimmedOffsetString.getStart() && f.getEnd() == trimmedOffsetString.getEnd()) + .collect(Collectors.toSet()); + found.forEach(f -> { + f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger); + if (sortedResult) { + f.setWord(searchableText.getAsStringWithLinebreaksSorted(f.getPositionSequences() + .stream() + .map(EntityPositionSequence::getSequences) + .flatMap(Collection::stream) + .collect(Collectors.toList()))); } + }); - if (!stop.isEmpty() && includeStop) { - searchString = searchString + stop; - } + EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary); - Set found = findEntities(searchString.trim(), asType, false, redaction, ruleNumber, reason, legalBasis, Engine.RULE, false); - found.forEach(f -> { - f.setSkipRemoveEntitiesContainedInLarger(skipRemoveEntitiesContainedInLarger); - if (sortedResult) { - f.setWord(searchableText.getAsStringWithLinebreaksSorted(f.getPositionSequences() - .stream() - .map(EntityPositionSequence::getSequences) - .flatMap(Collection::stream) - .collect(Collectors.toList()))); - } - }); - - EntitySearchUtils.addEntitiesWithHigherRank(entities, found, dictionary); - - if (redactEverywhere && !isLocal()) { - localDictionaryAdds.computeIfAbsent(asType, x -> new HashSet<>()).add(searchString.trim()); - } + if (redactEverywhere && !isLocal()) { + localDictionaryAdds.computeIfAbsent(asType, x -> new HashSet<>()).add(trimmedOffsetString.getValue()); } } } diff --git a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/utils/OffsetStringUtils.java b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/utils/OffsetStringUtils.java index 3154624b..9585d7f2 100644 --- a/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/utils/OffsetStringUtils.java +++ b/redaction-service-v1/redaction-service-server-v1/src/main/java/com/iqser/red/service/redaction/v1/server/redaction/utils/OffsetStringUtils.java @@ -15,8 +15,8 @@ public class OffsetStringUtils { /** * Same logic as in StringUtils.redactBetween, but returns a list of object with offsets insteadof on the Strings only. * - * @param str – the String containing the substrings, null returns null, empty returns empty - * @param open – the String identifying the start of the substring, empty returns null + * @param str – the String containing the substrings, null returns null, empty returns empty + * @param open – the String identifying the start of the substring, empty returns null * @param close – the String identifying the end of the substring, empty returns null * @return a list of Strings with their offsets */ @@ -52,5 +52,41 @@ public class OffsetStringUtils { return list; } + + public static OffsetString substringAfter(final String str, final String separator) { + + if (StringUtils.isEmpty(str)) { + return new OffsetString(str, 0, 0); + } + if (separator == null) { + return new OffsetString(str, 0, 0); + } + final int pos = str.indexOf(separator); + if (pos == StringUtils.INDEX_NOT_FOUND) { + return new OffsetString(str, 0, 0); + } + + String result = str.substring(pos + separator.length()); + + return new OffsetString(result, pos + separator.length(), str.length()); + } + + + public static OffsetString substringBefore(final String str, final String separator) { + + if (StringUtils.isEmpty(str) || separator == null) { + return new OffsetString(str, 0, 0); + } + if (separator.isEmpty()) { + return new OffsetString(str, 0, 0); + } + final int pos = str.indexOf(separator); + if (pos == StringUtils.INDEX_NOT_FOUND) { + return new OffsetString(str, 0, 0); + } + String result = str.substring(0, pos); + return new OffsetString(result, 0, pos); + } + }