diff --git a/.gitignore b/.gitignore
index 94cdb5e..fd4e33f 100644
--- a/.gitignore
+++ b/.gitignore
@@ -14,4 +14,5 @@ apache-solr/solr-tagging/data
apache-solr/solr-tagging/conf/dataimport.properties
apache-solr/solr-clustering/data
apache-solr/solr-clustering/conf/dataimport.properties
+apache-solr/example/solr/data
diff --git a/README b/README
index 2579ba8..fd9d9d8 100644
--- a/README
+++ b/README
@@ -29,6 +29,15 @@ started, you will need:
cd opennlp-models
wget -nd -np -r http://maven.tamingtext.com/opennlp-models/models-1.5/
rm index.html*
+
+
+ or using wget (https://eternallybored.org/misc/wget/) and 7-Zip (http://www.7-zip.org/) on windows (both must be added to the path environment variable):
+
+ md opennlp-models
+ cd opennlp-models
+ wget -nd -np -r http://maven.tamingtext.com/opennlp-models/models-1.5/
+ del index.htm*
+
4. Get WordNet 3.0 and place it in the TT_HOME directory.
@@ -37,11 +46,22 @@ started, you will need:
wget -nd -np -m http://maven.tamingtext.com/wordnet/
rm index.html*
- tar -xf Wordnet-3.0.tar.gz
+ tar -xf WordNet-3.0.tar.gz
+
+ or using wget (https://eternallybored.org/misc/wget/) and 7-Zip (http://www.7-zip.org/) on windows (both must be added to the path environment variable):
+
+ wget -nd -np -r http://maven.tamingtext.com/wordnet/
+ del index.html*
+ 7z x WordNet-3.0.tar.gz
+ 7z x WordNet-3.0.tar
Building the Source
-------------------
+Prior to building the source, for those previously unfamiliar with Maven,
+it may be wise to read this to avoid future hassles:
+http://maven.apache.org/guides/getting-started/maven-in-five-minutes.html
+
To build the source, in TT_HOME:
mvn clean package
diff --git a/apache-solr/example/solr/conf/solrconfig.xml b/apache-solr/example/solr/conf/solrconfig.xml
index 5f50ab5..543a2e0 100755
--- a/apache-solr/example/solr/conf/solrconfig.xml
+++ b/apache-solr/example/solr/conf/solrconfig.xml
@@ -76,20 +76,11 @@
files in that directory which completely match the regex
(anchored on both ends) will be included.
-->
-
-
-
-
-
-
-
-
-
-
-
-
-
+
+
+
+
@@ -612,10 +612,11 @@
+
+
tamingtext.maven2.repository
diff --git a/src/main/java/com/tamingtext/util/SplitInput.java b/src/main/java/com/tamingtext/util/SplitInput.java
index 9f1ba0e..f3c316d 100644
--- a/src/main/java/com/tamingtext/util/SplitInput.java
+++ b/src/main/java/com/tamingtext/util/SplitInput.java
@@ -26,7 +26,8 @@
import java.io.Writer;
import java.nio.charset.Charset;
import java.util.BitSet;
-import java.util.Collections;
+import java.util.HashSet;
+import java.util.Set;
import org.apache.commons.cli2.CommandLine;
import org.apache.commons.cli2.Group;
@@ -353,16 +354,19 @@ else if (fs.getFileStatus(inputFile).isDir()) {
BufferedReader reader = new BufferedReader(new InputStreamReader(fs.open(inputFile), charset));
Writer trainingWriter = new OutputStreamWriter(fs.create(trainingOutputFile), charset);
Writer testWriter = new OutputStreamWriter(fs.create(testOutputFile), charset);
+ Set writers = new HashSet();
+ writers.add(trainingWriter);
+ writers.add(testWriter);
int pos = 0;
int trainCount = 0;
int testCount = 0;
String line;
+ Writer writer;
while ((line = reader.readLine()) != null) {
pos++;
- Writer writer;
if (testRandomSelectionPct > 0) { // Randomly choose
writer = randomSel.get(pos) ? testWriter : trainingWriter;
} else { // Choose based on location
@@ -384,9 +388,8 @@ else if (fs.getFileStatus(inputFile).isDir()) {
writer.write(line);
writer.write('\n');
}
-
- IOUtils.close(Collections.singleton(trainingWriter));
- IOUtils.close(Collections.singleton(testWriter));
+
+ IOUtils.close(writers);
log.info("file: {}, input: {} train: {}, test: {} starting at {}",
new Object[] {inputFile.getName(), lineCount, trainCount, testCount, testSplitStart});
@@ -580,7 +583,12 @@ public static int countLines(FileSystem fs, Path inputFile, Charset charset) thr
lineCount++;
}
} finally {
- IOUtils.close(Collections.singleton(countReader));
+ try {
+ countReader.close();
+ }
+ catch (IOException ex) {
+ log.warn("Could not close line count reader", ex);
+ }
}
return lineCount;
diff --git a/src/test/java/com/tamingtext/frankenstein/Frankenstein.java b/src/test/java/com/tamingtext/frankenstein/Frankenstein.java
index 7979908..f3d3e91 100644
--- a/src/test/java/com/tamingtext/frankenstein/Frankenstein.java
+++ b/src/test/java/com/tamingtext/frankenstein/Frankenstein.java
@@ -220,13 +220,13 @@ private void addMetadata(Document doc, int lines, int paragraphs, int paragraphL
*/
private void init() throws IOException {
System.out.println("Initializing Frankenstein");
- File models = new File("../../opennlp-models");
- File wordnet = new File("../../WordNet-3.0");
+ File models = new File("./opennlp-models");
+ File wordnet = new File("./WordNet-3.0");
if (models.exists() == false) {
- throw new FileNotFoundException("../../opennlp-models");
+ throw new FileNotFoundException("./opennlp-models");
}
- System.setProperty("model.dir", "../../opennlp-models");
- System.setProperty("wordnet.dir", "../../WordNet-3.0");
+ System.setProperty("model.dir", "./opennlp-models");
+ System.setProperty("wordnet.dir", "./WordNet-3.0");
File modelFile = new File(models, "en-sent.bin");
InputStream modelStream = new FileInputStream(modelFile);
diff --git a/src/test/java/com/tamingtext/opennlp/NameFinderTest.java b/src/test/java/com/tamingtext/opennlp/NameFinderTest.java
index fd1eb51..fef2eb4 100644
--- a/src/test/java/com/tamingtext/opennlp/NameFinderTest.java
+++ b/src/test/java/com/tamingtext/opennlp/NameFinderTest.java
@@ -77,6 +77,7 @@ private void displayNames(Span[] names, String[] tokens) {
//
private void removeConflicts(List allAnnotations) {
+ if (allAnnotations.size() < 2) return; //
java.util.Collections.sort(allAnnotations); //
List stack = new ArrayList(); //
stack.add(allAnnotations.get(0));
@@ -119,6 +120,7 @@ private void removeConflicts(List allAnnotations) {
/*
+ Exit early if there will be no conflicts.
Sort the names based on their span's start index ascending then end index decending.
Initialize a stack to keep track of previous names.
Iterate over each name.