diff --git a/.build/dependencies.props b/.build/dependencies.props
index c5ecdaf4c4..c204fd7c6f 100644
--- a/.build/dependencies.props
+++ b/.build/dependencies.props
@@ -35,8 +35,6 @@
Just make sure they are adjusted to the right version of ICU/Lucene.
[60.1,60.2) -->
[60.1.0-alpha.440,60.1.0-alpha.446)
- 8.7.5
- 1.6.7
[2.2.0-alpha-0053, 3.0.0)
1.0.9
@@ -66,6 +64,7 @@
$(MorfologikFsaPackageVersion)
$(MorfologikFsaPackageVersion)
2.0.3
+ 1.9.5-beta.2
4.6.0
3.14.0
2.7.8
@@ -83,8 +82,4 @@
4.3.1
6.1.0
-
- 1.9.1
- 4.2.0
-
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj b/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj
index cbce7cae1f..83a14c71ce 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj
+++ b/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj
@@ -30,16 +30,9 @@
-
-
-
- net10.0;net8.0
- net8.0
- $(TargetFrameworks);net472
-
+ $(LibraryTargetFrameworks)
Lucene.Net.Analysis.OpenNLP
- $(PackageTags);analysis;natural;language;processing;opennlp
+ $(PackageTags);analysis;natural;language;processing;opennlp;nopennlp
bin\$(Configuration)\$(TargetFramework)\$(AssemblyName).xml
$(NoWarn);1591;1573
Lucene.Net.Analysis.OpenNlp
@@ -53,13 +46,7 @@
-
-
-
-
-
-
-
+
diff --git a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs
index 067f9bfdd6..59d3cf99c9 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs
@@ -3,7 +3,7 @@
using ICU4N.Text;
using Lucene.Net.Analysis.OpenNlp.Tools;
using Lucene.Net.Analysis.Util;
-using opennlp.tools.util;
+using NOpenNLP.Tools.Util;
using System;
using System.Diagnostics;
using System.Text;
@@ -249,7 +249,7 @@ public override void SetText(CharacterIterator newText)
for (int i = 0; i < spans.Length; ++i)
{
// Adjust start positions to match those of the passed-in CharacterIterator
- sentenceStarts[i] = spans[i].getStart() + text.BeginIndex;
+ sentenceStarts[i] = spans[i].Start + text.BeginIndex;
}
}
diff --git a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs
index 037a011c80..e9f051f93f 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs
@@ -3,7 +3,7 @@
using Lucene.Net.Analysis.TokenAttributes;
using Lucene.Net.Analysis.Util;
using Lucene.Net.Util;
-using opennlp.tools.util;
+using NOpenNLP.Tools.Util;
using System;
using System.IO;
@@ -92,9 +92,9 @@ protected override bool IncrementWord()
}
ClearAttributes();
Span term = termSpans[termNum];
- termAtt.CopyBuffer(m_buffer, sentenceStart + term.getStart(), term.length());
- offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.getStart()),
- CorrectOffset(m_offset + sentenceStart + term.getEnd()));
+ termAtt.CopyBuffer(m_buffer, sentenceStart + term.Start, term.Length);
+ offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.Start),
+ CorrectOffset(m_offset + sentenceStart + term.End));
if (termNum == termSpans.Length - 1)
{
flagsAtt.Flags = flagsAtt.Flags | EOS_FLAG_BIT; // mark the last token in the sentence with EOS_FLAG_BIT
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs
index c23ee3cdbb..9e3d6b93a3 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs
@@ -1,7 +1,6 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
-using opennlp.tools.chunker;
-
+using NOpenNLP.Tools.Chunker;
namespace Lucene.Net.Analysis.OpenNlp.Tools
{
@@ -40,9 +39,9 @@ public virtual string[] GetChunks(string[] words, string[] tags, double[] probs)
UninterruptableMonitor.Enter(this);
try
{
- string[] chunks = chunker.chunk(words, tags);
+ string[] chunks = chunker.Chunk(words, tags);
if (probs != null)
- chunker.probs(probs);
+ chunker.Probs(probs);
return chunks;
}
finally
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs
index 223fab5193..ae990a5294 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs
@@ -1,5 +1,6 @@
// Lucene version compatibility level 8.2.0
-using opennlp.tools.lemmatizer;
+
+using NOpenNLP.Tools.Lemmatizer;
using System.Diagnostics;
using System.IO;
@@ -39,7 +40,7 @@ public class NLPLemmatizerOp
public NLPLemmatizerOp(Stream dictionary, LemmatizerModel lemmatizerModel)
{
Debug.Assert(dictionary != null || lemmatizerModel != null, "At least one parameter must be non-null");
- dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(new ikvm.io.InputStreamWrapper(dictionary));
+ dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(dictionary);
lemmatizerME = lemmatizerModel is null ? null : new LemmatizerME(lemmatizerModel);
}
@@ -49,7 +50,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
string[] maxEntLemmas = null;
if (dictionaryLemmatizer != null)
{
- lemmas = dictionaryLemmatizer.lemmatize(words, postags);
+ lemmas = dictionaryLemmatizer.Lemmatize(words, postags);
for (int i = 0; i < lemmas.Length; ++i)
{
if (lemmas[i].Equals("O"))
@@ -58,7 +59,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
{ // fall back to the MaxEnt lemmatizer if it's enabled
if (maxEntLemmas is null)
{
- maxEntLemmas = lemmatizerME.lemmatize(words, postags);
+ maxEntLemmas = lemmatizerME.Lemmatize(words, postags);
}
if ("_".Equals(maxEntLemmas[i]))
{
@@ -78,7 +79,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
}
else
{ // there is only a MaxEnt lemmatizer
- maxEntLemmas = lemmatizerME.lemmatize(words, postags);
+ maxEntLemmas = lemmatizerME.Lemmatize(words, postags);
for (int i = 0; i < maxEntLemmas.Length; ++i)
{
if ("_".Equals(maxEntLemmas[i]))
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs
index f824957cf1..38bed341f2 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
-using opennlp.tools.namefind;
-using opennlp.tools.util;
+using NOpenNLP.Tools.Namefind;
+using NOpenNLP.Tools.Util;
namespace Lucene.Net.Analysis.OpenNlp.Tools
{
@@ -39,7 +39,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools
///
public class NLPNERTaggerOp
{
- private readonly TokenNameFinder nameFinder;
+ private readonly ITokenNameFinder nameFinder;
public NLPNERTaggerOp(TokenNameFinderModel model)
{
@@ -48,7 +48,7 @@ public NLPNERTaggerOp(TokenNameFinderModel model)
public virtual Span[] GetNames(string[] words)
{
- Span[] names = nameFinder.find(words);
+ Span[] names = nameFinder.Find(words);
return names;
}
@@ -57,7 +57,7 @@ public virtual void Reset()
UninterruptableMonitor.Enter(this);
try
{
- nameFinder.clearAdaptiveData();
+ nameFinder.ClearAdaptiveData();
}
finally
{
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs
index cb82a7d644..c7cf038052 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs
@@ -1,6 +1,6 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
-using opennlp.tools.postag;
+using NOpenNLP.Tools.Postag;
namespace Lucene.Net.Analysis.OpenNlp.Tools
{
@@ -27,7 +27,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools
///
public class NLPPOSTaggerOp
{
- private readonly POSTagger tagger = null;
+ private readonly IPOSTagger tagger = null;
public NLPPOSTaggerOp(POSModel model)
{
@@ -39,7 +39,7 @@ public virtual string[] GetPOSTags(string[] words)
UninterruptableMonitor.Enter(this);
try
{
- return tagger.tag(words);
+ return tagger.Tag(words);
}
finally
{
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs
index 49651dc281..1437b8beee 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
-using opennlp.tools.sentdetect;
-using opennlp.tools.util;
+using NOpenNLP.Tools.Sentdetect;
+using NOpenNLP.Tools.Util;
namespace Lucene.Net.Analysis.OpenNlp.Tools
{
@@ -47,7 +47,7 @@ public virtual Span[] SplitSentences(string line)
{
if (sentenceSplitter != null)
{
- return sentenceSplitter.sentPosDetect(line);
+ return sentenceSplitter.SentPosDetect(line);
}
else
{
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs
index ef6842f53c..e2a66676ff 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
-using opennlp.tools.tokenize;
-using opennlp.tools.util;
+using NOpenNLP.Tools.Tokenize;
+using NOpenNLP.Tools.Util;
namespace Lucene.Net.Analysis.OpenNlp.Tools
{
@@ -51,7 +51,7 @@ public virtual Span[] GetTerms(string sentence)
span1[0] = new Span(0, sentence.Length);
return span1;
}
- return tokenizer.tokenizePos(sentence);
+ return tokenizer.TokenizePos(sentence);
}
finally
{
diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs
index 23c17fda8f..5215b0b449 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs
+++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs
@@ -1,11 +1,11 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Analysis.Util;
-using opennlp.tools.chunker;
-using opennlp.tools.lemmatizer;
-using opennlp.tools.namefind;
-using opennlp.tools.postag;
-using opennlp.tools.sentdetect;
-using opennlp.tools.tokenize;
+using NOpenNLP.Tools.Chunker;
+using NOpenNLP.Tools.Lemmatizer;
+using NOpenNLP.Tools.Namefind;
+using NOpenNLP.Tools.Postag;
+using NOpenNLP.Tools.Sentdetect;
+using NOpenNLP.Tools.Tokenize;
using System.Collections.Concurrent;
using System.Diagnostics;
using System.IO;
@@ -63,7 +63,7 @@ public static SentenceModel GetSentenceModel(string modelName, IResourceLoader l
return sentenceModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new SentenceModel(new ikvm.io.InputStreamWrapper(resource));
+ return new SentenceModel(resource);
});
}
@@ -86,7 +86,7 @@ public static TokenizerModel GetTokenizerModel(string modelName, IResourceLoader
return tokenizerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new TokenizerModel(new ikvm.io.InputStreamWrapper(resource));
+ return new TokenizerModel(resource);
});
}
@@ -102,7 +102,7 @@ public static POSModel GetPOSTaggerModel(string modelName, IResourceLoader loade
return posTaggerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new POSModel(new ikvm.io.InputStreamWrapper(resource));
+ return new POSModel(resource);
});
}
@@ -118,7 +118,7 @@ public static ChunkerModel GetChunkerModel(string modelName, IResourceLoader loa
return chunkerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new ChunkerModel(new ikvm.io.InputStreamWrapper(resource));
+ return new ChunkerModel(resource);
});
}
@@ -134,7 +134,7 @@ public static TokenNameFinderModel GetNERTaggerModel(string modelName, IResource
return nerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new TokenNameFinderModel(new ikvm.io.InputStreamWrapper(resource));
+ return new TokenNameFinderModel(resource);
});
}
@@ -178,7 +178,7 @@ public static LemmatizerModel GetLemmatizerModel(string modelName, IResourceLoad
return lemmatizerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
- return new LemmatizerModel(new ikvm.io.InputStreamWrapper(resource));
+ return new LemmatizerModel(resource);
});
}
diff --git a/src/Lucene.Net.Analysis.OpenNLP/overview.md b/src/Lucene.Net.Analysis.OpenNLP/overview.md
index 33e23f4b6f..b65723cc88 100644
--- a/src/Lucene.Net.Analysis.OpenNLP/overview.md
+++ b/src/Lucene.Net.Analysis.OpenNLP/overview.md
@@ -38,22 +38,11 @@ Since the is not store
- copies the value to the
- creates a cloned token at the same position as each tagged token, and copies the value to the , optionally with a customized prefix (so that tags effectively occupy a different namespace from token text).
-Named Entity Recognition is also supported by OpenNLP, but there is no OpenNLPNERFilter included. For an implementation, see the [lucenenet-opennlp-mavenreference-demo](https://github.com/NightOwl888/lucenenet-opennlp-mavenreference-demo).
+Named Entity Recognition is also supported by OpenNLP, but there is no OpenNLPNERFilter included.
-## MavenReference Primer
+## Underlying NLP library
-When a `` is included for this NuGet package in your SDK-style MSBuild project, it will automatically include transitive dependencies to [`opennlp-tools` on maven.org](https://search.maven.org/artifact/org.apache.opennlp/opennlp-tools/1.9.4/bundle). The transitive dependency will automatically include a `` in your MSBuild project.
-
-The `` item group operates similar to a dependency in Maven. All transitive dependencies are collected and resolved, and then the final output is produced. However, unlike `PackageReference`s, `MavenReference`s are collected by the final output project, and reassessed. That is, each dependent Project within your .NET SDK-style solution contributes its `MavenReference`s to project(s) which include it, and each project makes its own dependency graph. Projects do not contribute their final built assemblies up. They only contribute their dependencies. Allowing each project in a complicated solution to make its own local conflict resolution attempt.
-
-> [!NOTE]
-> `` is only supported on SDK-style MSBuild projects.
-
-## MavenReference Example
-
-This means this package can be combined with other related packages on Maven in your project and they can be accessed using the same path as in Java like a namespace in .NET. For example, you can add a `` to your project to include a reference to [`opennlp-uima`](https://search.maven.org/artifact/org.apache.opennlp/opennlp-uima/1.9.1/jar). The UIMA (Unstructured Information Management Architecture) integration module is designed to work with the Apache UIMA framework. UIMA is a framework for building applications that analyze unstructured information, and it's often used for processing natural language text. The opennlp-uima module allows you to integrate OpenNLP functionality into UIMA pipelines, leveraging the capabilities of both frameworks.
-
-Here's a basic outline of how you might extend an existing Lucene.NET analyzer to incorporate OpenNLP-UIMA annotators:
+This module is built on [NOpenNLP](https://github.com/nopennlp/nopennlp), a C# port of Apache OpenNLP 1.9.5. NOpenNLP is referenced as an ordinary NuGet package, so using this module requires nothing beyond a ``:
```xml
@@ -62,62 +51,9 @@ Here's a basic outline of how you might extend an existing Lucene.NET analyzer t
-
-
-
-
-
+
```
-```c#
-using Lucene.Net.Analysis;
-using Lucene.Net.Analysis.Core;
-using Lucene.Net.Analysis.Util;
-using Lucene.Net.Util;
-using org.apache.uima.analysis_engine;
-using System.IO;
-
-public class CustomOpenNLPAnalyzer : OpenNLPTokenizerFactory
-{
- // ... constructor and other methods ...
-
- public override Tokenizer Create(AttributeFactory factory, TextReader reader)
- {
- Tokenizer tokenizer = base.Create(factory, reader);
-
- // Wrap the tokenizer with UIMA annotators
- AnalysisEngineDescription uimaSentenceAnnotator = CreateUIMASentenceAnnotator();
- AnalysisEngineDescription uimaTokenAnnotator = CreateUIMATokenAnnotator();
-
- // Combine OpenNLP-UIMA annotators with the existing tokenizer
- AnalysisEngine tokenizerAndUIMAAnnotators = CreateAggregate(uimaSentenceAnnotator, uimaTokenAnnotator);
-
- return new UIMATokenizer(tokenizer, tokenizerAndUIMAAnnotators);
- }
-
- // ... other methods ...
-
- private AnalysisEngineDescription CreateUIMASentenceAnnotator() {
- // Create and configure UIMA sentence annotator
- // ...
-
- return /* UIMA sentence annotator description */;
- }
-
- private AnalysisEngineDescription CreateUIMATokenAnnotator() {
- // Create and configure UIMA token annotator
- // ...
-
- return /* UIMA token annotator description */;
- }
-}
-```
-
-In the above example, `CustomOpenNLPAnalyzer` extends `OpenNLPTokenizerFactory` (assuming that's the analyzer you're using), and it wraps the OpenNLP tokenizer with UIMA annotators. You'll need to replace the placeholder methods (`CreateUIMASentenceAnnotator` and `CreateUIMATokenAnnotator`) with the actual code to create and configure your UIMA annotators. Please note that configuring NLP can be complex. See the [OpenNLP 1.9.4 Manual](https://opennlp.apache.org/docs/1.9.4/manual/opennlp.html) and [OpenNLP UIMA 1.9.4 API Documentation](https://opennlp.apache.org/docs/1.9.4/apidocs/opennlp-uima/index.html) for details.
-
-> [!NOTE]
-> IKVM (and ``) does not support Java SE higher than version 8. So it will not be possible to add a `` to OpenNLP 2.x until support is added for it in IKVM.
-
-For a more complete example, see the [lucenenet-opennlp-mavenreference-demo](https://github.com/NightOwl888/lucenenet-opennlp-mavenreference-demo).
+NOpenNLP ports the `opennlp-tools` module, keeping the upstream OpenNLP type and member names with .NET casing conventions applied, so the [OpenNLP Manual](https://opennlp.apache.org/docs/1.9.4/manual/opennlp.html) and Java examples generally carry over. Its API reference is at [nopennlp.github.io/nopennlp](https://nopennlp.github.io/nopennlp/).