Optimize imports

Reformatted code (Java convention; tab is 4 spaces)
This commit is contained in:
robert-bor 2016-11-30 12:07:03 +01:00
parent 267e895059
commit 255069624b
13 changed files with 200 additions and 185 deletions

View File

@ -1,4 +1,5 @@
<project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd"> <project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
<modelVersion>4.0.0</modelVersion> <modelVersion>4.0.0</modelVersion>
<groupId>org.ahocorasick</groupId> <groupId>org.ahocorasick</groupId>

View File

@ -9,7 +9,7 @@ public class Interval implements Intervalable {
* Constructs an interval with a start and end position. * Constructs an interval with a start and end position.
* *
* @param start The interval's starting text position. * @param start The interval's starting text position.
* @param end The interval's ending text position. * @param end The interval's ending text position.
*/ */
public Interval(final int start, final int end) { public Interval(final int start, final int end) {
this.start = start; this.start = start;
@ -55,7 +55,7 @@ public class Interval implements Intervalable {
*/ */
public boolean overlapsWith(final Interval other) { public boolean overlapsWith(final Interval other) {
return this.start <= other.getEnd() && return this.start <= other.getEnd() &&
this.end >= other.getStart(); this.end >= other.getStart();
} }
public boolean overlapsWith(int point) { public boolean overlapsWith(int point) {
@ -67,9 +67,9 @@ public class Interval implements Intervalable {
if (!(o instanceof Intervalable)) { if (!(o instanceof Intervalable)) {
return false; return false;
} }
Intervalable other = (Intervalable)o; Intervalable other = (Intervalable) o;
return this.start == other.getStart() && return this.start == other.getStart() &&
this.end == other.getEnd(); this.end == other.getEnd();
} }
@Override @Override
@ -82,7 +82,7 @@ public class Interval implements Intervalable {
if (!(o instanceof Intervalable)) { if (!(o instanceof Intervalable)) {
return -1; return -1;
} }
Intervalable other = (Intervalable)o; Intervalable other = (Intervalable) o;
int comparison = this.start - other.getStart(); int comparison = this.start - other.getStart();
return comparison != 0 ? comparison : this.end - other.getEnd(); return comparison != 0 ? comparison : this.end - other.getEnd();
} }

View File

@ -6,7 +6,7 @@ import java.util.List;
public class IntervalNode { public class IntervalNode {
private enum Direction { LEFT, RIGHT } private enum Direction {LEFT, RIGHT}
private IntervalNode left = null; private IntervalNode left = null;
private IntervalNode right = null; private IntervalNode right = null;
@ -93,12 +93,12 @@ public class IntervalNode {
List<Intervalable> overlaps = new ArrayList<Intervalable>(); List<Intervalable> overlaps = new ArrayList<Intervalable>();
for (Intervalable currentInterval : this.intervals) { for (Intervalable currentInterval : this.intervals) {
switch (direction) { switch (direction) {
case LEFT : case LEFT:
if (currentInterval.getStart() <= interval.getEnd()) { if (currentInterval.getStart() <= interval.getEnd()) {
overlaps.add(currentInterval); overlaps.add(currentInterval);
} }
break; break;
case RIGHT : case RIGHT:
if (currentInterval.getEnd() >= interval.getStart()) { if (currentInterval.getEnd() >= interval.getStart()) {
overlaps.add(currentInterval); overlaps.add(currentInterval);
} }

View File

@ -3,7 +3,9 @@ package org.ahocorasick.interval;
public interface Intervalable extends Comparable { public interface Intervalable extends Comparable {
public int getStart(); public int getStart();
public int getEnd(); public int getEnd();
public int size(); public int size();
} }

View File

@ -4,43 +4,51 @@ import java.util.*;
/** /**
* <p> * <p>
* A state has various important tasks it must attend to: * A state has various important tasks it must attend to:
* </p> * </p>
*
* <ul>
* <li>success; when a character points to another state, it must return that state</li>
* <li>failure; when a character has no matching state, the algorithm must be able to fall back on a
* state with less depth</li>
* <li>emits; when this state is passed and keywords have been matched, the matches must be
* 'emitted' so that they can be used later on.</li>
* </ul>
*
* <p> * <p>
* The root state is special in the sense that it has no failure state; it cannot fail. If it 'fails' * <ul>
* it will still parse the next character and start from the root node. This ensures that the algorithm * <li>success; when a character points to another state, it must return that state</li>
* always runs. All other states always have a fail state. * <li>failure; when a character has no matching state, the algorithm must be able to fall back on a
* state with less depth</li>
* <li>emits; when this state is passed and keywords have been matched, the matches must be
* 'emitted' so that they can be used later on.</li>
* </ul>
* <p>
* <p>
* The root state is special in the sense that it has no failure state; it cannot fail. If it 'fails'
* it will still parse the next character and start from the root node. This ensures that the algorithm
* always runs. All other states always have a fail state.
* </p> * </p>
* *
* @author Robert Bor * @author Robert Bor
*/ */
public class State { public class State {
/** effective the size of the keyword */ /**
* effective the size of the keyword
*/
private final int depth; private final int depth;
/** only used for the root state to refer to itself in case no matches have been found */ /**
* only used for the root state to refer to itself in case no matches have been found
*/
private final State rootState; private final State rootState;
/** /**
* referred to in the white paper as the 'goto' structure. From a state it is possible to go * referred to in the white paper as the 'goto' structure. From a state it is possible to go
* to other states, depending on the character passed. * to other states, depending on the character passed.
*/ */
private final Map<Character,State> success = new HashMap<>(); private final Map<Character, State> success = new HashMap<>();
/** if no matching states are found, the failure state will be returned */ /**
* if no matching states are found, the failure state will be returned
*/
private State failure; private State failure;
/** whenever this state is reached, it will emit the matches keywords for future reference */ /**
* whenever this state is reached, it will emit the matches keywords for future reference
*/
private Set<String> emits; private Set<String> emits;
public State() { public State() {
@ -54,11 +62,11 @@ public class State {
private State nextState(final Character character, final boolean ignoreRootState) { private State nextState(final Character character, final boolean ignoreRootState) {
State nextState = this.success.get(character); State nextState = this.success.get(character);
if (!ignoreRootState && nextState == null && this.rootState != null) { if (!ignoreRootState && nextState == null && this.rootState != null) {
nextState = this.rootState; nextState = this.rootState;
} }
return nextState; return nextState;
} }
@ -69,21 +77,21 @@ public class State {
public State nextStateIgnoreRootState(Character character) { public State nextStateIgnoreRootState(Character character) {
return nextState(character, true); return nextState(character, true);
} }
public State addState( String keyword ) { public State addState(String keyword) {
State state = this; State state = this;
for (final Character character : keyword.toCharArray()) { for (final Character character : keyword.toCharArray()) {
state = state.addState(character); state = state.addState(character);
} }
return state; return state;
} }
public State addState(Character character) { public State addState(Character character) {
State nextState = nextStateIgnoreRootState(character); State nextState = nextStateIgnoreRootState(character);
if (nextState == null) { if (nextState == null) {
nextState = new State(this.depth+1); nextState = new State(this.depth + 1);
this.success.put(character, nextState); this.success.put(character, nextState);
} }
return nextState; return nextState;
@ -107,7 +115,7 @@ public class State {
} }
public Collection<String> emit() { public Collection<String> emit() {
return this.emits == null ? Collections.<String> emptyList() : this.emits; return this.emits == null ? Collections.<String>emptyList() : this.emits;
} }
public State failure() { public State failure() {

View File

@ -1,20 +1,22 @@
package org.ahocorasick.trie; package org.ahocorasick.trie;
import static java.lang.Character.isWhitespace;
import java.util.ArrayList;
import java.util.Collection;
import java.util.List;
import java.util.Queue;
import java.util.concurrent.LinkedBlockingDeque;
import org.ahocorasick.interval.IntervalTree; import org.ahocorasick.interval.IntervalTree;
import org.ahocorasick.interval.Intervalable; import org.ahocorasick.interval.Intervalable;
import org.ahocorasick.trie.handler.DefaultEmitHandler; import org.ahocorasick.trie.handler.DefaultEmitHandler;
import org.ahocorasick.trie.handler.EmitHandler; import org.ahocorasick.trie.handler.EmitHandler;
import java.util.ArrayList;
import java.util.Collection;
import java.util.List;
import java.util.Queue;
import java.util.concurrent.LinkedBlockingDeque;
import static java.lang.Character.isWhitespace;
/** /**
* Based on the Aho-Corasick white paper, Bell technologies: * Based on the Aho-Corasick white paper, Bell technologies:
* http://cr.yp.to/bib/1975/aho.pdf * http://cr.yp.to/bib/1975/aho.pdf
* *
* @author Robert Bor * @author Robert Bor
*/ */
public class Trie { public class Trie {
@ -27,21 +29,20 @@ public class Trie {
this.trieConfig = trieConfig; this.trieConfig = trieConfig;
this.rootState = new State(); this.rootState = new State();
} }
/** /**
* Used by the builder to add a text search keyword. * Used by the builder to add a text search keyword.
* *
* @param keyword The search term to add to the list of search terms. * @param keyword The search term to add to the list of search terms.
*
* @throws NullPointerException if the keyword is null. * @throws NullPointerException if the keyword is null.
*/ */
private void addKeyword(String keyword) { private void addKeyword(String keyword) {
if( keyword.isEmpty() ) { if (keyword.isEmpty()) {
return; return;
} }
if( isCaseInsensitive() ) { if (isCaseInsensitive()) {
keyword = keyword.toLowerCase(); keyword = keyword.toLowerCase();
} }
addState(keyword).addEmit(keyword); addState(keyword).addEmit(keyword);
@ -49,44 +50,44 @@ public class Trie {
/** /**
* Delegates to addKeyword. * Delegates to addKeyword.
* *
* @param keywords List of search term to add to the list of search terms. * @param keywords List of search term to add to the list of search terms.
*/ */
private void addKeywords( final String[] keywords ) { private void addKeywords(final String[] keywords) {
for( final String keyword : keywords ) { for (final String keyword : keywords) {
addKeyword( keyword ); addKeyword(keyword);
} }
} }
/** /**
* Delegates to addKeyword. * Delegates to addKeyword.
* *
* @param keywords List of search term to add to the list of search terms. * @param keywords List of search term to add to the list of search terms.
*/ */
private void addKeywords( final Collection<String> keywords ) { private void addKeywords(final Collection<String> keywords) {
for( final String keyword : keywords ) { for (final String keyword : keywords) {
addKeyword( keyword ); addKeyword(keyword);
} }
} }
private State addState(final String keyword) { private State addState(final String keyword) {
return getRootState().addState(keyword); return getRootState().addState(keyword);
} }
public Collection<Token> tokenize(final String text) { public Collection<Token> tokenize(final String text) {
final Collection<Token> tokens = new ArrayList<>(); final Collection<Token> tokens = new ArrayList<>();
final Collection<Emit> collectedEmits = parseText(text); final Collection<Emit> collectedEmits = parseText(text);
int lastCollectedPosition = -1; int lastCollectedPosition = -1;
for (final Emit emit : collectedEmits) { for (final Emit emit : collectedEmits) {
if (emit.getStart() - lastCollectedPosition > 1) { if (emit.getStart() - lastCollectedPosition > 1) {
tokens.add(createFragment(emit, text, lastCollectedPosition)); tokens.add(createFragment(emit, text, lastCollectedPosition));
} }
tokens.add(createMatch(emit, text)); tokens.add(createMatch(emit, text));
lastCollectedPosition = emit.getEnd(); lastCollectedPosition = emit.getEnd();
} }
if (text.length() - lastCollectedPosition > 1) { if (text.length() - lastCollectedPosition > 1) {
tokens.add(createFragment(null, text, lastCollectedPosition)); tokens.add(createFragment(null, text, lastCollectedPosition));
} }
@ -95,11 +96,11 @@ public class Trie {
} }
private Token createFragment(final Emit emit, final String text, final int lastCollectedPosition) { private Token createFragment(final Emit emit, final String text, final int lastCollectedPosition) {
return new FragmentToken(text.substring(lastCollectedPosition+1, emit == null ? text.length() : emit.getStart())); return new FragmentToken(text.substring(lastCollectedPosition + 1, emit == null ? text.length() : emit.getStart()));
} }
private Token createMatch(Emit emit, String text) { private Token createMatch(Emit emit, String text) {
return new MatchToken(text.substring(emit.getStart(), emit.getEnd()+1), emit); return new MatchToken(text.substring(emit.getStart(), emit.getEnd() + 1), emit);
} }
@SuppressWarnings("unchecked") @SuppressWarnings("unchecked")
@ -118,7 +119,7 @@ public class Trie {
} }
if (!trieConfig.isAllowOverlaps()) { if (!trieConfig.isAllowOverlaps()) {
IntervalTree intervalTree = new IntervalTree((List<Intervalable>)(List<?>)collectedEmits); IntervalTree intervalTree = new IntervalTree((List<Intervalable>) (List<?>) collectedEmits);
intervalTree.removeOverlaps((List<Intervalable>) (List<?>) collectedEmits); intervalTree.removeOverlaps((List<Intervalable>) (List<?>) collectedEmits);
} }
@ -131,15 +132,15 @@ public class Trie {
public void parseText(final CharSequence text, final EmitHandler emitHandler) { public void parseText(final CharSequence text, final EmitHandler emitHandler) {
State currentState = getRootState(); State currentState = getRootState();
for (int position = 0; position < text.length(); position++) { for (int position = 0; position < text.length(); position++) {
Character character = text.charAt(position); Character character = text.charAt(position);
// TODO: Maybe lowercase the entire string at once? // TODO: Maybe lowercase the entire string at once?
if (trieConfig.isCaseInsensitive()) { if (trieConfig.isCaseInsensitive()) {
character = Character.toLowerCase(character); character = Character.toLowerCase(character);
} }
currentState = getState(currentState, character); currentState = getState(currentState, character);
if (storeEmits(position, currentState, emitHandler) && trieConfig.isStopOnHit()) { if (storeEmits(position, currentState, emitHandler) && trieConfig.isStopOnHit()) {
return; return;
@ -149,7 +150,7 @@ public class Trie {
/** /**
* The first matching text sequence. * The first matching text sequence.
* *
* @param text The text to search for keywords. * @param text The text to search for keywords.
* @return null if no matches found. * @return null if no matches found.
*/ */
@ -164,18 +165,18 @@ public class Trie {
} else { } else {
// Fast path. Returns first match found. // Fast path. Returns first match found.
State currentState = getRootState(); State currentState = getRootState();
for (int position = 0; position < text.length(); position++) { for (int position = 0; position < text.length(); position++) {
Character character = text.charAt(position); Character character = text.charAt(position);
// TODO: Lowercase the entire string at once? // TODO: Lowercase the entire string at once?
if (trieConfig.isCaseInsensitive()) { if (trieConfig.isCaseInsensitive()) {
character = Character.toLowerCase(character); character = Character.toLowerCase(character);
} }
currentState = getState(currentState, character); currentState = getState(currentState, character);
Collection<String> emitStrs = currentState.emit(); Collection<String> emitStrs = currentState.emit();
if (emitStrs != null && !emitStrs.isEmpty()) { if (emitStrs != null && !emitStrs.isEmpty()) {
for (final String emitStr : emitStrs) { for (final String emitStr : emitStrs) {
final Emit emit = new Emit(position - emitStr.length() + 1, position, emitStr); final Emit emit = new Emit(position - emitStr.length() + 1, position, emitStr);
@ -190,26 +191,26 @@ public class Trie {
} }
} }
} }
return null; return null;
} }
private boolean isPartialMatch(final CharSequence searchText, final Emit emit) { private boolean isPartialMatch(final CharSequence searchText, final Emit emit) {
return (emit.getStart() != 0 && return (emit.getStart() != 0 &&
Character.isAlphabetic(searchText.charAt(emit.getStart() - 1))) || Character.isAlphabetic(searchText.charAt(emit.getStart() - 1))) ||
(emit.getEnd() + 1 != searchText.length() && (emit.getEnd() + 1 != searchText.length() &&
Character.isAlphabetic(searchText.charAt(emit.getEnd() + 1))); Character.isAlphabetic(searchText.charAt(emit.getEnd() + 1)));
} }
private void removePartialMatches(final CharSequence searchText, final List<Emit> collectedEmits) { private void removePartialMatches(final CharSequence searchText, final List<Emit> collectedEmits) {
final List<Emit> removeEmits = new ArrayList<>(); final List<Emit> removeEmits = new ArrayList<>();
for (final Emit emit : collectedEmits) { for (final Emit emit : collectedEmits) {
if (isPartialMatch(searchText, emit)) { if (isPartialMatch(searchText, emit)) {
removeEmits.add(emit); removeEmits.add(emit);
} }
} }
for (final Emit removeEmit : removeEmits) { for (final Emit removeEmit : removeEmits) {
collectedEmits.remove(removeEmit); collectedEmits.remove(removeEmit);
} }
@ -218,15 +219,15 @@ public class Trie {
private void removePartialMatchesWhiteSpaceSeparated(final CharSequence searchText, final List<Emit> collectedEmits) { private void removePartialMatchesWhiteSpaceSeparated(final CharSequence searchText, final List<Emit> collectedEmits) {
final long size = searchText.length(); final long size = searchText.length();
final List<Emit> removeEmits = new ArrayList<>(); final List<Emit> removeEmits = new ArrayList<>();
for (final Emit emit : collectedEmits) { for (final Emit emit : collectedEmits) {
if ((emit.getStart() == 0 || isWhitespace(searchText.charAt(emit.getStart() - 1))) && if ((emit.getStart() == 0 || isWhitespace(searchText.charAt(emit.getStart() - 1))) &&
(emit.getEnd() + 1 == size || isWhitespace(searchText.charAt(emit.getEnd() + 1)))) { (emit.getEnd() + 1 == size || isWhitespace(searchText.charAt(emit.getEnd() + 1)))) {
continue; continue;
} }
removeEmits.add(emit); removeEmits.add(emit);
} }
for (final Emit removeEmit : removeEmits) { for (final Emit removeEmit : removeEmits) {
collectedEmits.remove(removeEmit); collectedEmits.remove(removeEmit);
} }
@ -234,12 +235,12 @@ public class Trie {
private State getState(State currentState, final Character character) { private State getState(State currentState, final Character character) {
State newCurrentState = currentState.nextState(character); State newCurrentState = currentState.nextState(character);
while (newCurrentState == null) { while (newCurrentState == null) {
currentState = currentState.failure(); currentState = currentState.failure();
newCurrentState = currentState.nextState(character); newCurrentState = currentState.nextState(character);
} }
return newCurrentState; return newCurrentState;
} }
@ -276,7 +277,7 @@ public class Trie {
private boolean storeEmits(final int position, final State currentState, final EmitHandler emitHandler) { private boolean storeEmits(final int position, final State currentState, final EmitHandler emitHandler) {
boolean emitted = false; boolean emitted = false;
final Collection<String> emits = currentState.emit(); final Collection<String> emits = currentState.emit();
// TODO: The check for empty might be superfluous. // TODO: The check for empty might be superfluous.
if (emits != null && !emits.isEmpty()) { if (emits != null && !emits.isEmpty()) {
for (final String emit : emits) { for (final String emit : emits) {
@ -284,21 +285,21 @@ public class Trie {
emitted = true; emitted = true;
} }
} }
return emitted; return emitted;
} }
private boolean isCaseInsensitive() { private boolean isCaseInsensitive() {
return trieConfig.isCaseInsensitive(); return trieConfig.isCaseInsensitive();
} }
private State getRootState() { private State getRootState() {
return this.rootState; return this.rootState;
} }
/** /**
* Provides a fluent interface for constructing Trie instances. * Provides a fluent interface for constructing Trie instances.
* *
* @return The builder used to configure its Trie. * @return The builder used to configure its Trie.
*/ */
public static TrieBuilder builder() { public static TrieBuilder builder() {
@ -314,14 +315,15 @@ public class Trie {
/** /**
* Default (empty) constructor. * Default (empty) constructor.
*/ */
private TrieBuilder() {} private TrieBuilder() {
}
/** /**
* Configure the Trie to ignore case when searching for keywords in * Configure the Trie to ignore case when searching for keywords in
* the text. This must be called before calling addKeyword because * the text. This must be called before calling addKeyword because
* the algorithm converts keywords to lowercase as they are added, * the algorithm converts keywords to lowercase as they are added,
* depending on this case sensitivity setting. * depending on this case sensitivity setting.
* *
* @return This builder. * @return This builder.
*/ */
public TrieBuilder ignoreCase() { public TrieBuilder ignoreCase() {
@ -331,7 +333,7 @@ public class Trie {
/** /**
* Configure the Trie to ignore overlapping keywords. * Configure the Trie to ignore overlapping keywords.
* *
* @return This builder. * @return This builder.
*/ */
public TrieBuilder ignoreOverlaps() { public TrieBuilder ignoreOverlaps() {
@ -341,9 +343,8 @@ public class Trie {
/** /**
* Adds a keyword to the Trie's list of text search keywords. * Adds a keyword to the Trie's list of text search keywords.
* *
* @param keyword The keyword to add to the list. * @param keyword The keyword to add to the list.
*
* @return This builder. * @return This builder.
* @throws NullPointerException if the keyword is null. * @throws NullPointerException if the keyword is null.
*/ */
@ -351,34 +352,32 @@ public class Trie {
this.trie.addKeyword(keyword); this.trie.addKeyword(keyword);
return this; return this;
} }
/** /**
* Adds a list of keywords to the Trie's list of text search keywords. * Adds a list of keywords to the Trie's list of text search keywords.
* *
* @param keywords The keywords to add to the list. * @param keywords The keywords to add to the list.
*
* @return This builder. * @return This builder.
*/ */
public TrieBuilder addKeywords(final String... keywords) { public TrieBuilder addKeywords(final String... keywords) {
this.trie.addKeywords(keywords); this.trie.addKeywords(keywords);
return this; return this;
} }
/** /**
* Adds a list of keywords to the Trie's list of text search keywords. * Adds a list of keywords to the Trie's list of text search keywords.
* *
* @param keywords The keywords to add to the list. * @param keywords The keywords to add to the list.
*
* @return This builder. * @return This builder.
*/ */
public TrieBuilder addKeywords(final Collection<String> keywords) { public TrieBuilder addKeywords(final Collection<String> keywords) {
this.trie.addKeywords(keywords); this.trie.addKeywords(keywords);
return this; return this;
} }
/** /**
* Configure the Trie to match whole keywords in the text. * Configure the Trie to match whole keywords in the text.
* *
* @return This builder. * @return This builder.
*/ */
public TrieBuilder onlyWholeWords() { public TrieBuilder onlyWholeWords() {
@ -390,7 +389,7 @@ public class Trie {
* Configure the Trie to match whole keywords that are separated by * Configure the Trie to match whole keywords that are separated by
* whitespace in the text. For example, "this keyword thatkeyword" * whitespace in the text. For example, "this keyword thatkeyword"
* would only match the first occurrence of "keyword". * would only match the first occurrence of "keyword".
* *
* @return This builder. * @return This builder.
*/ */
public TrieBuilder onlyWholeWordsWhiteSpaceSeparated() { public TrieBuilder onlyWholeWordsWhiteSpaceSeparated() {
@ -401,7 +400,7 @@ public class Trie {
/** /**
* Configure the Trie to stop after the first keyword is found in the * Configure the Trie to stop after the first keyword is found in the
* text. * text.
* *
* @return This builder. * @return This builder.
*/ */
public TrieBuilder stopOnHit() { public TrieBuilder stopOnHit() {
@ -411,27 +410,25 @@ public class Trie {
/** /**
* Configure the Trie based on the builder settings. * Configure the Trie based on the builder settings.
* *
* @return The configured Trie. * @return The configured Trie.
*/ */
public Trie build() { public Trie build() {
this.trie.constructFailureStates(); this.trie.constructFailureStates();
return this.trie; return this.trie;
} }
/** /**
* @deprecated Use ignoreCase()
*
* @return This builder. * @return This builder.
* @deprecated Use ignoreCase()
*/ */
public TrieBuilder caseInsensitive() { public TrieBuilder caseInsensitive() {
return ignoreCase(); return ignoreCase();
} }
/** /**
* @deprecated Use ignoreOverlaps()
*
* @return This builder. * @return This builder.
* @deprecated Use ignoreOverlaps()
*/ */
public TrieBuilder removeOverlaps() { public TrieBuilder removeOverlaps() {
return ignoreOverlaps(); return ignoreOverlaps();

View File

@ -12,9 +12,13 @@ public class TrieConfig {
private boolean stopOnHit = false; private boolean stopOnHit = false;
public boolean isStopOnHit() { return stopOnHit; } public boolean isStopOnHit() {
return stopOnHit;
}
public void setStopOnHit(boolean stopOnHit) { this.stopOnHit = stopOnHit; } public void setStopOnHit(boolean stopOnHit) {
this.stopOnHit = stopOnHit;
}
public boolean isAllowOverlaps() { public boolean isAllowOverlaps() {
return allowOverlaps; return allowOverlaps;
@ -32,7 +36,9 @@ public class TrieConfig {
this.onlyWholeWords = onlyWholeWords; this.onlyWholeWords = onlyWholeWords;
} }
public boolean isOnlyWholeWordsWhiteSpaceSeparated() { return onlyWholeWordsWhiteSpaceSeparated; } public boolean isOnlyWholeWordsWhiteSpaceSeparated() {
return onlyWholeWordsWhiteSpaceSeparated;
}
public void setOnlyWholeWordsWhiteSpaceSeparated(boolean onlyWholeWordsWhiteSpaceSeparated) { public void setOnlyWholeWordsWhiteSpaceSeparated(boolean onlyWholeWordsWhiteSpaceSeparated) {
this.onlyWholeWordsWhiteSpaceSeparated = onlyWholeWordsWhiteSpaceSeparated; this.onlyWholeWordsWhiteSpaceSeparated = onlyWholeWordsWhiteSpaceSeparated;

View File

@ -2,29 +2,29 @@ package org.ahocorasick.interval;
import org.junit.Test; import org.junit.Test;
import java.util.*; import java.util.Iterator;
import java.util.Set;
import java.util.TreeSet;
import static junit.framework.Assert.assertEquals; import static junit.framework.Assert.*;
import static junit.framework.Assert.assertFalse;
import static junit.framework.Assert.assertTrue;
public class IntervalTest { public class IntervalTest {
@Test @Test
public void construct() { public void construct() {
Interval i = new Interval(1,3); Interval i = new Interval(1, 3);
assertEquals(1, i.getStart()); assertEquals(1, i.getStart());
assertEquals(3, i.getEnd()); assertEquals(3, i.getEnd());
} }
@Test @Test
public void size() { public void size() {
assertEquals(3, new Interval(0,2).size()); assertEquals(3, new Interval(0, 2).size());
} }
@Test @Test
public void intervaloverlaps() { public void intervaloverlaps() {
assertTrue(new Interval(1,3).overlapsWith(new Interval(2,4))); assertTrue(new Interval(1, 3).overlapsWith(new Interval(2, 4)));
} }
@Test @Test
@ -34,7 +34,7 @@ public class IntervalTest {
@Test @Test
public void pointOverlaps() { public void pointOverlaps() {
assertTrue(new Interval(1,3).overlapsWith(2)); assertTrue(new Interval(1, 3).overlapsWith(2));
} }
@Test @Test

View File

@ -9,7 +9,7 @@ import java.util.List;
import static junit.framework.Assert.assertEquals; import static junit.framework.Assert.assertEquals;
public class IntervalTreeTest { public class IntervalTreeTest {
@Test @Test
public void findOverlaps() { public void findOverlaps() {
List<Intervalable> intervals = new ArrayList<Intervalable>(); List<Intervalable> intervals = new ArrayList<Intervalable>();
@ -20,7 +20,7 @@ public class IntervalTreeTest {
intervals.add(new Interval(4, 6)); intervals.add(new Interval(4, 6));
intervals.add(new Interval(5, 7)); intervals.add(new Interval(5, 7));
IntervalTree intervalTree = new IntervalTree(intervals); IntervalTree intervalTree = new IntervalTree(intervals);
List<Intervalable> overlaps = intervalTree.findOverlaps(new Interval(1,3)); List<Intervalable> overlaps = intervalTree.findOverlaps(new Interval(1, 3));
assertEquals(3, overlaps.size()); assertEquals(3, overlaps.size());
Iterator<Intervalable> overlapsIt = overlaps.iterator(); Iterator<Intervalable> overlapsIt = overlaps.iterator();
assertOverlap(overlapsIt.next(), 2, 4); assertOverlap(overlapsIt.next(), 2, 4);
@ -47,5 +47,5 @@ public class IntervalTreeTest {
assertEquals(expectedStart, interval.getStart()); assertEquals(expectedStart, interval.getStart());
assertEquals(expectedEnd, interval.getEnd()); assertEquals(expectedEnd, interval.getEnd());
} }
} }

View File

@ -13,9 +13,9 @@ public class IntervalableComparatorByPositionTest {
@Test @Test
public void sortOnPosition() { public void sortOnPosition() {
List<Intervalable> intervals = new ArrayList<Intervalable>(); List<Intervalable> intervals = new ArrayList<Intervalable>();
intervals.add(new Interval(4,5)); intervals.add(new Interval(4, 5));
intervals.add(new Interval(1,4)); intervals.add(new Interval(1, 4));
intervals.add(new Interval(3,8)); intervals.add(new Interval(3, 8));
Collections.sort(intervals, new IntervalableComparatorByPosition()); Collections.sort(intervals, new IntervalableComparatorByPosition());
assertEquals(4, intervals.get(0).size()); assertEquals(4, intervals.get(0).size());
assertEquals(6, intervals.get(1).size()); assertEquals(6, intervals.get(1).size());

View File

@ -13,9 +13,9 @@ public class IntervalableComparatorBySizeTest {
@Test @Test
public void sortOnSize() { public void sortOnSize() {
List<Intervalable> intervals = new ArrayList<Intervalable>(); List<Intervalable> intervals = new ArrayList<Intervalable>();
intervals.add(new Interval(4,5)); intervals.add(new Interval(4, 5));
intervals.add(new Interval(1,4)); intervals.add(new Interval(1, 4));
intervals.add(new Interval(3,8)); intervals.add(new Interval(3, 8));
Collections.sort(intervals, new IntervalableComparatorBySize()); Collections.sort(intervals, new IntervalableComparatorBySize());
assertEquals(6, intervals.get(0).size()); assertEquals(6, intervals.get(0).size());
assertEquals(4, intervals.get(1).size()); assertEquals(4, intervals.get(1).size());
@ -25,8 +25,8 @@ public class IntervalableComparatorBySizeTest {
@Test @Test
public void sortOnSizeThenPosition() { public void sortOnSizeThenPosition() {
List<Intervalable> intervals = new ArrayList<Intervalable>(); List<Intervalable> intervals = new ArrayList<Intervalable>();
intervals.add(new Interval(4,7)); intervals.add(new Interval(4, 7));
intervals.add(new Interval(2,5)); intervals.add(new Interval(2, 5));
Collections.sort(intervals, new IntervalableComparatorBySize()); Collections.sort(intervals, new IntervalableComparatorBySize());
assertEquals(2, intervals.get(0).getStart()); assertEquals(2, intervals.get(0).getStart());
assertEquals(4, intervals.get(1).getStart()); assertEquals(4, intervals.get(1).getStart());

View File

@ -1,6 +1,5 @@
package org.ahocorasick.trie; package org.ahocorasick.trie;
import org.ahocorasick.trie.State;
import org.junit.Test; import org.junit.Test;
import static junit.framework.Assert.assertEquals; import static junit.framework.Assert.assertEquals;
@ -11,9 +10,9 @@ public class StateTest {
public void constructSequenceOfCharacters() { public void constructSequenceOfCharacters() {
State rootState = new State(); State rootState = new State();
rootState rootState
.addState('a') .addState('a')
.addState('b') .addState('b')
.addState('c'); .addState('c');
State currentState = rootState.nextState('a'); State currentState = rootState.nextState('a');
assertEquals(1, currentState.getDepth()); assertEquals(1, currentState.getDepth());
currentState = currentState.nextState('b'); currentState = currentState.nextState('b');

View File

@ -1,34 +1,36 @@
package org.ahocorasick.trie; package org.ahocorasick.trie;
import org.ahocorasick.trie.handler.EmitHandler;
import org.junit.Test;
import java.util.ArrayList; import java.util.ArrayList;
import java.util.Collection; import java.util.Collection;
import java.util.Iterator; import java.util.Iterator;
import java.util.List; import java.util.List;
import java.util.concurrent.ThreadLocalRandom; import java.util.concurrent.ThreadLocalRandom;
import static junit.framework.Assert.assertEquals; import static junit.framework.Assert.assertEquals;
import org.ahocorasick.trie.handler.EmitHandler;
import static org.junit.Assert.assertTrue; import static org.junit.Assert.assertTrue;
import org.junit.Test;
public class TrieTest { public class TrieTest {
private final static String[] ALPHABET = new String[]{ private final static String[] ALPHABET = new String[]{
"abc", "bcd", "cde" "abc", "bcd", "cde"
}; };
private final static String[] PRONOUNS = new String[]{ private final static String[] PRONOUNS = new String[]{
"hers", "his", "she", "he" "hers", "his", "she", "he"
}; };
private final static String[] FOOD = new String[]{ private final static String[] FOOD = new String[]{
"veal", "cauliflower", "broccoli", "tomatoes" "veal", "cauliflower", "broccoli", "tomatoes"
}; };
private final static String[] GREEK_LETTERS = new String[]{ private final static String[] GREEK_LETTERS = new String[]{
"Alpha", "Beta", "Gamma" "Alpha", "Beta", "Gamma"
}; };
private final static String[] UNICODE = new String[]{ private final static String[] UNICODE = new String[]{
"turning", "once", "again", "börkü" "turning", "once", "again", "börkü"
}; };
@Test @Test
@ -406,7 +408,7 @@ public class TrieTest {
.onlyWholeWordsWhiteSpaceSeparated() .onlyWholeWordsWhiteSpaceSeparated()
.addKeyword("#sugar-123") .addKeyword("#sugar-123")
.build(); .build();
Collection < Emit > emits = trie.parseText("#sugar-123 #sugar-1234"); // left, middle, right test Collection<Emit> emits = trie.parseText("#sugar-123 #sugar-1234"); // left, middle, right test
assertEquals(1, emits.size()); // Match must not be made assertEquals(1, emits.size()); // Match must not be made
checkEmit(emits.iterator().next(), 0, 9, "#sugar-123"); checkEmit(emits.iterator().next(), 0, 9, "#sugar-123");
} }
@ -415,57 +417,57 @@ public class TrieTest {
public void testLargeString() { public void testLargeString() {
final int interval = 100; final int interval = 100;
final int textSize = 1000000; final int textSize = 1000000;
final String keyword = FOOD[ 1 ]; final String keyword = FOOD[1];
final StringBuilder text = randomNumbers( textSize ); final StringBuilder text = randomNumbers(textSize);
injectKeyword( text, keyword, interval ); injectKeyword(text, keyword, interval);
Trie trie = Trie.builder() Trie trie = Trie.builder()
.onlyWholeWords() .onlyWholeWords()
.addKeyword( keyword ) .addKeyword(keyword)
.build(); .build();
final Collection<Emit> emits = trie.parseText( text ); final Collection<Emit> emits = trie.parseText(text);
assertEquals( textSize / interval, emits.size() ); assertEquals(textSize / interval, emits.size());
} }
/** /**
* Generates a random sequence of ASCII numbers. * Generates a random sequence of ASCII numbers.
* *
* @param count The number of numbers to generate. * @param count The number of numbers to generate.
* @return A character sequence filled with random digits. * @return A character sequence filled with random digits.
*/ */
private StringBuilder randomNumbers( int count ) { private StringBuilder randomNumbers(int count) {
final StringBuilder sb = new StringBuilder( count ); final StringBuilder sb = new StringBuilder(count);
while( --count > 0 ) { while (--count > 0) {
sb.append( randomInt( 0, 10 ) ); sb.append(randomInt(0, 10));
} }
return sb; return sb;
} }
/** /**
* Injects keywords into a string builder. * Injects keywords into a string builder.
* *
* @param source Should contain a bunch of random data that cannot match * @param source Should contain a bunch of random data that cannot match
* any keyword. * any keyword.
* @param keyword A keyword to inject repeatedly in the text. * @param keyword A keyword to inject repeatedly in the text.
* @param interval How often to inject the keyword. * @param interval How often to inject the keyword.
*/ */
private void injectKeyword( private void injectKeyword(
final StringBuilder source, final StringBuilder source,
final String keyword, final String keyword,
final int interval ) { final int interval) {
final int length = source.length(); final int length = source.length();
for( int i = 0; i < length; i += interval ) { for (int i = 0; i < length; i += interval) {
source.replace( i, i + keyword.length(), keyword ); source.replace(i, i + keyword.length(), keyword);
} }
} }
private int randomInt( final int min, final int max ) { private int randomInt(final int min, final int max) {
return ThreadLocalRandom.current().nextInt( min, max ); return ThreadLocalRandom.current().nextInt(min, max);
} }
private void checkEmit(Emit next, int expectedStart, int expectedEnd, String expectedKeyword) { private void checkEmit(Emit next, int expectedStart, int expectedEnd, String expectedKeyword) {