diff --git a/.build/dependencies.props b/.build/dependencies.props index c5ecdaf4c4..c204fd7c6f 100644 --- a/.build/dependencies.props +++ b/.build/dependencies.props @@ -35,8 +35,6 @@ Just make sure they are adjusted to the right version of ICU/Lucene. [60.1,60.2) --> [60.1.0-alpha.440,60.1.0-alpha.446) - 8.7.5 - 1.6.7 [2.2.0-alpha-0053, 3.0.0) 1.0.9 @@ -66,6 +64,7 @@ $(MorfologikFsaPackageVersion) $(MorfologikFsaPackageVersion) 2.0.3 + 1.9.5-beta.2 4.6.0 3.14.0 2.7.8 @@ -83,8 +82,4 @@ 4.3.1 6.1.0 - - 1.9.1 - 4.2.0 - diff --git a/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj b/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj index cbce7cae1f..83a14c71ce 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj +++ b/src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj @@ -30,16 +30,9 @@ - - - - net10.0;net8.0 - net8.0 - $(TargetFrameworks);net472 - + $(LibraryTargetFrameworks) Lucene.Net.Analysis.OpenNLP - $(PackageTags);analysis;natural;language;processing;opennlp + $(PackageTags);analysis;natural;language;processing;opennlp;nopennlp bin\$(Configuration)\$(TargetFramework)\$(AssemblyName).xml $(NoWarn);1591;1573 Lucene.Net.Analysis.OpenNlp @@ -53,13 +46,7 @@ - - - - - - - + diff --git a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs index 067f9bfdd6..59d3cf99c9 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPSentenceBreakIterator.cs @@ -3,7 +3,7 @@ using ICU4N.Text; using Lucene.Net.Analysis.OpenNlp.Tools; using Lucene.Net.Analysis.Util; -using opennlp.tools.util; +using NOpenNLP.Tools.Util; using System; using System.Diagnostics; using System.Text; @@ -249,7 +249,7 @@ public override void SetText(CharacterIterator newText) for (int i = 0; i < spans.Length; ++i) { // Adjust start positions to match those of the passed-in CharacterIterator - sentenceStarts[i] = spans[i].getStart() + text.BeginIndex; + sentenceStarts[i] = spans[i].Start + text.BeginIndex; } } diff --git a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs index 037a011c80..e9f051f93f 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs @@ -3,7 +3,7 @@ using Lucene.Net.Analysis.TokenAttributes; using Lucene.Net.Analysis.Util; using Lucene.Net.Util; -using opennlp.tools.util; +using NOpenNLP.Tools.Util; using System; using System.IO; @@ -92,9 +92,9 @@ protected override bool IncrementWord() } ClearAttributes(); Span term = termSpans[termNum]; - termAtt.CopyBuffer(m_buffer, sentenceStart + term.getStart(), term.length()); - offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.getStart()), - CorrectOffset(m_offset + sentenceStart + term.getEnd())); + termAtt.CopyBuffer(m_buffer, sentenceStart + term.Start, term.Length); + offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.Start), + CorrectOffset(m_offset + sentenceStart + term.End)); if (termNum == termSpans.Length - 1) { flagsAtt.Flags = flagsAtt.Flags | EOS_FLAG_BIT; // mark the last token in the sentence with EOS_FLAG_BIT diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs index c23ee3cdbb..9e3d6b93a3 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs @@ -1,7 +1,6 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Support.Threading; -using opennlp.tools.chunker; - +using NOpenNLP.Tools.Chunker; namespace Lucene.Net.Analysis.OpenNlp.Tools { @@ -40,9 +39,9 @@ public virtual string[] GetChunks(string[] words, string[] tags, double[] probs) UninterruptableMonitor.Enter(this); try { - string[] chunks = chunker.chunk(words, tags); + string[] chunks = chunker.Chunk(words, tags); if (probs != null) - chunker.probs(probs); + chunker.Probs(probs); return chunks; } finally diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs index 223fab5193..ae990a5294 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs @@ -1,5 +1,6 @@ // Lucene version compatibility level 8.2.0 -using opennlp.tools.lemmatizer; + +using NOpenNLP.Tools.Lemmatizer; using System.Diagnostics; using System.IO; @@ -39,7 +40,7 @@ public class NLPLemmatizerOp public NLPLemmatizerOp(Stream dictionary, LemmatizerModel lemmatizerModel) { Debug.Assert(dictionary != null || lemmatizerModel != null, "At least one parameter must be non-null"); - dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(new ikvm.io.InputStreamWrapper(dictionary)); + dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(dictionary); lemmatizerME = lemmatizerModel is null ? null : new LemmatizerME(lemmatizerModel); } @@ -49,7 +50,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags) string[] maxEntLemmas = null; if (dictionaryLemmatizer != null) { - lemmas = dictionaryLemmatizer.lemmatize(words, postags); + lemmas = dictionaryLemmatizer.Lemmatize(words, postags); for (int i = 0; i < lemmas.Length; ++i) { if (lemmas[i].Equals("O")) @@ -58,7 +59,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags) { // fall back to the MaxEnt lemmatizer if it's enabled if (maxEntLemmas is null) { - maxEntLemmas = lemmatizerME.lemmatize(words, postags); + maxEntLemmas = lemmatizerME.Lemmatize(words, postags); } if ("_".Equals(maxEntLemmas[i])) { @@ -78,7 +79,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags) } else { // there is only a MaxEnt lemmatizer - maxEntLemmas = lemmatizerME.lemmatize(words, postags); + maxEntLemmas = lemmatizerME.Lemmatize(words, postags); for (int i = 0; i < maxEntLemmas.Length; ++i) { if ("_".Equals(maxEntLemmas[i])) diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs index f824957cf1..38bed341f2 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs @@ -1,7 +1,7 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Support.Threading; -using opennlp.tools.namefind; -using opennlp.tools.util; +using NOpenNLP.Tools.Namefind; +using NOpenNLP.Tools.Util; namespace Lucene.Net.Analysis.OpenNlp.Tools { @@ -39,7 +39,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools /// public class NLPNERTaggerOp { - private readonly TokenNameFinder nameFinder; + private readonly ITokenNameFinder nameFinder; public NLPNERTaggerOp(TokenNameFinderModel model) { @@ -48,7 +48,7 @@ public NLPNERTaggerOp(TokenNameFinderModel model) public virtual Span[] GetNames(string[] words) { - Span[] names = nameFinder.find(words); + Span[] names = nameFinder.Find(words); return names; } @@ -57,7 +57,7 @@ public virtual void Reset() UninterruptableMonitor.Enter(this); try { - nameFinder.clearAdaptiveData(); + nameFinder.ClearAdaptiveData(); } finally { diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs index cb82a7d644..c7cf038052 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs @@ -1,6 +1,6 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Support.Threading; -using opennlp.tools.postag; +using NOpenNLP.Tools.Postag; namespace Lucene.Net.Analysis.OpenNlp.Tools { @@ -27,7 +27,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools /// public class NLPPOSTaggerOp { - private readonly POSTagger tagger = null; + private readonly IPOSTagger tagger = null; public NLPPOSTaggerOp(POSModel model) { @@ -39,7 +39,7 @@ public virtual string[] GetPOSTags(string[] words) UninterruptableMonitor.Enter(this); try { - return tagger.tag(words); + return tagger.Tag(words); } finally { diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs index 49651dc281..1437b8beee 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPSentenceDetectorOp.cs @@ -1,7 +1,7 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Support.Threading; -using opennlp.tools.sentdetect; -using opennlp.tools.util; +using NOpenNLP.Tools.Sentdetect; +using NOpenNLP.Tools.Util; namespace Lucene.Net.Analysis.OpenNlp.Tools { @@ -47,7 +47,7 @@ public virtual Span[] SplitSentences(string line) { if (sentenceSplitter != null) { - return sentenceSplitter.sentPosDetect(line); + return sentenceSplitter.SentPosDetect(line); } else { diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs index ef6842f53c..e2a66676ff 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs @@ -1,7 +1,7 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Support.Threading; -using opennlp.tools.tokenize; -using opennlp.tools.util; +using NOpenNLP.Tools.Tokenize; +using NOpenNLP.Tools.Util; namespace Lucene.Net.Analysis.OpenNlp.Tools { @@ -51,7 +51,7 @@ public virtual Span[] GetTerms(string sentence) span1[0] = new Span(0, sentence.Length); return span1; } - return tokenizer.tokenizePos(sentence); + return tokenizer.TokenizePos(sentence); } finally { diff --git a/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs b/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs index 23c17fda8f..5215b0b449 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs +++ b/src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs @@ -1,11 +1,11 @@ // Lucene version compatibility level 8.2.0 using Lucene.Net.Analysis.Util; -using opennlp.tools.chunker; -using opennlp.tools.lemmatizer; -using opennlp.tools.namefind; -using opennlp.tools.postag; -using opennlp.tools.sentdetect; -using opennlp.tools.tokenize; +using NOpenNLP.Tools.Chunker; +using NOpenNLP.Tools.Lemmatizer; +using NOpenNLP.Tools.Namefind; +using NOpenNLP.Tools.Postag; +using NOpenNLP.Tools.Sentdetect; +using NOpenNLP.Tools.Tokenize; using System.Collections.Concurrent; using System.Diagnostics; using System.IO; @@ -63,7 +63,7 @@ public static SentenceModel GetSentenceModel(string modelName, IResourceLoader l return sentenceModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new SentenceModel(new ikvm.io.InputStreamWrapper(resource)); + return new SentenceModel(resource); }); } @@ -86,7 +86,7 @@ public static TokenizerModel GetTokenizerModel(string modelName, IResourceLoader return tokenizerModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new TokenizerModel(new ikvm.io.InputStreamWrapper(resource)); + return new TokenizerModel(resource); }); } @@ -102,7 +102,7 @@ public static POSModel GetPOSTaggerModel(string modelName, IResourceLoader loade return posTaggerModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new POSModel(new ikvm.io.InputStreamWrapper(resource)); + return new POSModel(resource); }); } @@ -118,7 +118,7 @@ public static ChunkerModel GetChunkerModel(string modelName, IResourceLoader loa return chunkerModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new ChunkerModel(new ikvm.io.InputStreamWrapper(resource)); + return new ChunkerModel(resource); }); } @@ -134,7 +134,7 @@ public static TokenNameFinderModel GetNERTaggerModel(string modelName, IResource return nerModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new TokenNameFinderModel(new ikvm.io.InputStreamWrapper(resource)); + return new TokenNameFinderModel(resource); }); } @@ -178,7 +178,7 @@ public static LemmatizerModel GetLemmatizerModel(string modelName, IResourceLoad return lemmatizerModels.GetOrAdd(modelName, (modelName) => { using Stream resource = loader.OpenResource(modelName); - return new LemmatizerModel(new ikvm.io.InputStreamWrapper(resource)); + return new LemmatizerModel(resource); }); } diff --git a/src/Lucene.Net.Analysis.OpenNLP/overview.md b/src/Lucene.Net.Analysis.OpenNLP/overview.md index 33e23f4b6f..b65723cc88 100644 --- a/src/Lucene.Net.Analysis.OpenNLP/overview.md +++ b/src/Lucene.Net.Analysis.OpenNLP/overview.md @@ -38,22 +38,11 @@ Since the is not store - copies the value to the - creates a cloned token at the same position as each tagged token, and copies the value to the , optionally with a customized prefix (so that tags effectively occupy a different namespace from token text). -Named Entity Recognition is also supported by OpenNLP, but there is no OpenNLPNERFilter included. For an implementation, see the [lucenenet-opennlp-mavenreference-demo](https://github.com/NightOwl888/lucenenet-opennlp-mavenreference-demo). +Named Entity Recognition is also supported by OpenNLP, but there is no OpenNLPNERFilter included. -## MavenReference Primer +## Underlying NLP library -When a `` is included for this NuGet package in your SDK-style MSBuild project, it will automatically include transitive dependencies to [`opennlp-tools` on maven.org](https://search.maven.org/artifact/org.apache.opennlp/opennlp-tools/1.9.4/bundle). The transitive dependency will automatically include a `` in your MSBuild project. - -The `` item group operates similar to a dependency in Maven. All transitive dependencies are collected and resolved, and then the final output is produced. However, unlike `PackageReference`s, `MavenReference`s are collected by the final output project, and reassessed. That is, each dependent Project within your .NET SDK-style solution contributes its `MavenReference`s to project(s) which include it, and each project makes its own dependency graph. Projects do not contribute their final built assemblies up. They only contribute their dependencies. Allowing each project in a complicated solution to make its own local conflict resolution attempt. - -> [!NOTE] -> `` is only supported on SDK-style MSBuild projects. - -## MavenReference Example - -This means this package can be combined with other related packages on Maven in your project and they can be accessed using the same path as in Java like a namespace in .NET. For example, you can add a `` to your project to include a reference to [`opennlp-uima`](https://search.maven.org/artifact/org.apache.opennlp/opennlp-uima/1.9.1/jar). The UIMA (Unstructured Information Management Architecture) integration module is designed to work with the Apache UIMA framework. UIMA is a framework for building applications that analyze unstructured information, and it's often used for processing natural language text. The opennlp-uima module allows you to integrate OpenNLP functionality into UIMA pipelines, leveraging the capabilities of both frameworks. - -Here's a basic outline of how you might extend an existing Lucene.NET analyzer to incorporate OpenNLP-UIMA annotators: +This module is built on [NOpenNLP](https://github.com/nopennlp/nopennlp), a C# port of Apache OpenNLP 1.9.5. NOpenNLP is referenced as an ordinary NuGet package, so using this module requires nothing beyond a ``: ```xml @@ -62,62 +51,9 @@ Here's a basic outline of how you might extend an existing Lucene.NET analyzer t - - - - - + ``` -```c# -using Lucene.Net.Analysis; -using Lucene.Net.Analysis.Core; -using Lucene.Net.Analysis.Util; -using Lucene.Net.Util; -using org.apache.uima.analysis_engine; -using System.IO; - -public class CustomOpenNLPAnalyzer : OpenNLPTokenizerFactory -{ - // ... constructor and other methods ... - - public override Tokenizer Create(AttributeFactory factory, TextReader reader) - { - Tokenizer tokenizer = base.Create(factory, reader); - - // Wrap the tokenizer with UIMA annotators - AnalysisEngineDescription uimaSentenceAnnotator = CreateUIMASentenceAnnotator(); - AnalysisEngineDescription uimaTokenAnnotator = CreateUIMATokenAnnotator(); - - // Combine OpenNLP-UIMA annotators with the existing tokenizer - AnalysisEngine tokenizerAndUIMAAnnotators = CreateAggregate(uimaSentenceAnnotator, uimaTokenAnnotator); - - return new UIMATokenizer(tokenizer, tokenizerAndUIMAAnnotators); - } - - // ... other methods ... - - private AnalysisEngineDescription CreateUIMASentenceAnnotator() { - // Create and configure UIMA sentence annotator - // ... - - return /* UIMA sentence annotator description */; - } - - private AnalysisEngineDescription CreateUIMATokenAnnotator() { - // Create and configure UIMA token annotator - // ... - - return /* UIMA token annotator description */; - } -} -``` - -In the above example, `CustomOpenNLPAnalyzer` extends `OpenNLPTokenizerFactory` (assuming that's the analyzer you're using), and it wraps the OpenNLP tokenizer with UIMA annotators. You'll need to replace the placeholder methods (`CreateUIMASentenceAnnotator` and `CreateUIMATokenAnnotator`) with the actual code to create and configure your UIMA annotators. Please note that configuring NLP can be complex. See the [OpenNLP 1.9.4 Manual](https://opennlp.apache.org/docs/1.9.4/manual/opennlp.html) and [OpenNLP UIMA 1.9.4 API Documentation](https://opennlp.apache.org/docs/1.9.4/apidocs/opennlp-uima/index.html) for details. - -> [!NOTE] -> IKVM (and ``) does not support Java SE higher than version 8. So it will not be possible to add a `` to OpenNLP 2.x until support is added for it in IKVM. - -For a more complete example, see the [lucenenet-opennlp-mavenreference-demo](https://github.com/NightOwl888/lucenenet-opennlp-mavenreference-demo). +NOpenNLP ports the `opennlp-tools` module, keeping the upstream OpenNLP type and member names with .NET casing conventions applied, so the [OpenNLP Manual](https://opennlp.apache.org/docs/1.9.4/manual/opennlp.html) and Java examples generally carry over. Its API reference is at [nopennlp.github.io/nopennlp](https://nopennlp.github.io/nopennlp/).