diff --git a/.gitignore b/.gitignore index 94cdb5e..fd4e33f 100644 --- a/.gitignore +++ b/.gitignore @@ -14,4 +14,5 @@ apache-solr/solr-tagging/data apache-solr/solr-tagging/conf/dataimport.properties apache-solr/solr-clustering/data apache-solr/solr-clustering/conf/dataimport.properties +apache-solr/example/solr/data diff --git a/README b/README index 2579ba8..fd9d9d8 100644 --- a/README +++ b/README @@ -29,6 +29,15 @@ started, you will need: cd opennlp-models wget -nd -np -r http://maven.tamingtext.com/opennlp-models/models-1.5/ rm index.html* + + + or using wget (https://eternallybored.org/misc/wget/) and 7-Zip (http://www.7-zip.org/) on windows (both must be added to the path environment variable): + + md opennlp-models + cd opennlp-models + wget -nd -np -r http://maven.tamingtext.com/opennlp-models/models-1.5/ + del index.htm* + 4. Get WordNet 3.0 and place it in the TT_HOME directory. @@ -37,11 +46,22 @@ started, you will need: wget -nd -np -m http://maven.tamingtext.com/wordnet/ rm index.html* - tar -xf Wordnet-3.0.tar.gz + tar -xf WordNet-3.0.tar.gz + + or using wget (https://eternallybored.org/misc/wget/) and 7-Zip (http://www.7-zip.org/) on windows (both must be added to the path environment variable): + + wget -nd -np -r http://maven.tamingtext.com/wordnet/ + del index.html* + 7z x WordNet-3.0.tar.gz + 7z x WordNet-3.0.tar Building the Source ------------------- +Prior to building the source, for those previously unfamiliar with Maven, +it may be wise to read this to avoid future hassles: +http://maven.apache.org/guides/getting-started/maven-in-five-minutes.html + To build the source, in TT_HOME: mvn clean package diff --git a/apache-solr/example/solr/conf/solrconfig.xml b/apache-solr/example/solr/conf/solrconfig.xml index 5f50ab5..543a2e0 100755 --- a/apache-solr/example/solr/conf/solrconfig.xml +++ b/apache-solr/example/solr/conf/solrconfig.xml @@ -76,20 +76,11 @@ files in that directory which completely match the regex (anchored on both ends) will be included. --> - - - - - - - - - - - - - + + + + @@ -612,10 +612,11 @@ + + tamingtext.maven2.repository diff --git a/src/main/java/com/tamingtext/util/SplitInput.java b/src/main/java/com/tamingtext/util/SplitInput.java index 9f1ba0e..f3c316d 100644 --- a/src/main/java/com/tamingtext/util/SplitInput.java +++ b/src/main/java/com/tamingtext/util/SplitInput.java @@ -26,7 +26,8 @@ import java.io.Writer; import java.nio.charset.Charset; import java.util.BitSet; -import java.util.Collections; +import java.util.HashSet; +import java.util.Set; import org.apache.commons.cli2.CommandLine; import org.apache.commons.cli2.Group; @@ -353,16 +354,19 @@ else if (fs.getFileStatus(inputFile).isDir()) { BufferedReader reader = new BufferedReader(new InputStreamReader(fs.open(inputFile), charset)); Writer trainingWriter = new OutputStreamWriter(fs.create(trainingOutputFile), charset); Writer testWriter = new OutputStreamWriter(fs.create(testOutputFile), charset); + Set writers = new HashSet(); + writers.add(trainingWriter); + writers.add(testWriter); int pos = 0; int trainCount = 0; int testCount = 0; String line; + Writer writer; while ((line = reader.readLine()) != null) { pos++; - Writer writer; if (testRandomSelectionPct > 0) { // Randomly choose writer = randomSel.get(pos) ? testWriter : trainingWriter; } else { // Choose based on location @@ -384,9 +388,8 @@ else if (fs.getFileStatus(inputFile).isDir()) { writer.write(line); writer.write('\n'); } - - IOUtils.close(Collections.singleton(trainingWriter)); - IOUtils.close(Collections.singleton(testWriter)); + + IOUtils.close(writers); log.info("file: {}, input: {} train: {}, test: {} starting at {}", new Object[] {inputFile.getName(), lineCount, trainCount, testCount, testSplitStart}); @@ -580,7 +583,12 @@ public static int countLines(FileSystem fs, Path inputFile, Charset charset) thr lineCount++; } } finally { - IOUtils.close(Collections.singleton(countReader)); + try { + countReader.close(); + } + catch (IOException ex) { + log.warn("Could not close line count reader", ex); + } } return lineCount; diff --git a/src/test/java/com/tamingtext/frankenstein/Frankenstein.java b/src/test/java/com/tamingtext/frankenstein/Frankenstein.java index 7979908..f3d3e91 100644 --- a/src/test/java/com/tamingtext/frankenstein/Frankenstein.java +++ b/src/test/java/com/tamingtext/frankenstein/Frankenstein.java @@ -220,13 +220,13 @@ private void addMetadata(Document doc, int lines, int paragraphs, int paragraphL */ private void init() throws IOException { System.out.println("Initializing Frankenstein"); - File models = new File("../../opennlp-models"); - File wordnet = new File("../../WordNet-3.0"); + File models = new File("./opennlp-models"); + File wordnet = new File("./WordNet-3.0"); if (models.exists() == false) { - throw new FileNotFoundException("../../opennlp-models"); + throw new FileNotFoundException("./opennlp-models"); } - System.setProperty("model.dir", "../../opennlp-models"); - System.setProperty("wordnet.dir", "../../WordNet-3.0"); + System.setProperty("model.dir", "./opennlp-models"); + System.setProperty("wordnet.dir", "./WordNet-3.0"); File modelFile = new File(models, "en-sent.bin"); InputStream modelStream = new FileInputStream(modelFile); diff --git a/src/test/java/com/tamingtext/opennlp/NameFinderTest.java b/src/test/java/com/tamingtext/opennlp/NameFinderTest.java index fd1eb51..fef2eb4 100644 --- a/src/test/java/com/tamingtext/opennlp/NameFinderTest.java +++ b/src/test/java/com/tamingtext/opennlp/NameFinderTest.java @@ -77,6 +77,7 @@ private void displayNames(Span[] names, String[] tokens) { // private void removeConflicts(List allAnnotations) { + if (allAnnotations.size() < 2) return; // java.util.Collections.sort(allAnnotations); // List stack = new ArrayList(); // stack.add(allAnnotations.get(0)); @@ -119,6 +120,7 @@ private void removeConflicts(List allAnnotations) { /* + Exit early if there will be no conflicts. Sort the names based on their span's start index ascending then end index decending. Initialize a stack to keep track of previous names. Iterate over each name.