Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 1 addition & 6 deletions .build/dependencies.props
Original file line number Diff line number Diff line change
Expand Up @@ -35,8 +35,6 @@
Just make sure they are adjusted to the right version of ICU/Lucene.
<ICU4NPackageVersion>[60.1,60.2)</ICU4NPackageVersion> -->
<ICU4NPackageVersion>[60.1.0-alpha.440,60.1.0-alpha.446)</ICU4NPackageVersion>
<IKVMPackageVersion>8.7.5</IKVMPackageVersion>
<IKVMMavenSdkPackageVersion>1.6.7</IKVMMavenSdkPackageVersion>
<!-- J2N will break binary compatibility in 3.0.0 to fix the APIs of collection types -->
<J2NPackageVersion>[2.2.0-alpha-0053, 3.0.0)</J2NPackageVersion>
<LiquidTestReportsMarkdownPackageVersion>1.0.9</LiquidTestReportsMarkdownPackageVersion>
Expand Down Expand Up @@ -66,6 +64,7 @@
<MorfologikPolishPackageVersion>$(MorfologikFsaPackageVersion)</MorfologikPolishPackageVersion>
<MorfologikStemmingPackageVersion>$(MorfologikFsaPackageVersion)</MorfologikStemmingPackageVersion>
<NETStandardLibrary20PackageVersion>2.0.3</NETStandardLibrary20PackageVersion>
<NOpenNLPPackageVersion>1.9.5-beta.2</NOpenNLPPackageVersion>
<NUnit3TestAdapterPackageVersion>4.6.0</NUnit3TestAdapterPackageVersion>
<NUnitPackageVersion>3.14.0</NUnitPackageVersion>
<RandomizedTestingGeneratorsPackageVersion>2.7.8</RandomizedTestingGeneratorsPackageVersion>
Expand All @@ -83,8 +82,4 @@
<SystemTextRegularExpressionsPackageVersion>4.3.1</SystemTextRegularExpressionsPackageVersion>
<TimeZoneConverterPackageVersion>6.1.0</TimeZoneConverterPackageVersion>
</PropertyGroup>
<PropertyGroup Label="Maven Package Reference Versions">
<OpenNLPToolsMavenReferenceVersion>1.9.1</OpenNLPToolsMavenReferenceVersion>
<OSGICoreMavenReferenceVersion>4.2.0</OSGICoreMavenReferenceVersion>
</PropertyGroup>
</Project>
19 changes: 3 additions & 16 deletions src/Lucene.Net.Analysis.OpenNLP/Lucene.Net.Analysis.OpenNLP.csproj
Original file line number Diff line number Diff line change
Expand Up @@ -30,16 +30,9 @@
<Import Project="$(SolutionDir).build/nuget.props" />

<PropertyGroup>
<!-- Currently, IKVM doesn't officially support building NetFX on anything but Windows, so we skip it for contributors who may be on various platforms.
We can remove the condition once that has been addressed. See: https://github.com/ikvmnet/ikvm-maven/issues/49 -->
<!-- net10.0 is conditional on VS version (VS2024+ = 18.0) -->
<!-- .NET Framework is conditional on VS version (VS2024+ = 18.0) due to invalid build warnings with C# 14.0 build features -->
<TargetFrameworks>net10.0;net8.0</TargetFrameworks>
<TargetFrameworks Condition=" '$(IsLegacyVisualStudioVersion)' == 'true' ">net8.0</TargetFrameworks>
<TargetFrameworks Condition=" $([MSBuild]::IsOsPlatform('Windows')) And '$(IsLegacyVisualStudioVersion)' != 'true' ">$(TargetFrameworks);net472</TargetFrameworks>

<TargetFrameworks>$(LibraryTargetFrameworks)</TargetFrameworks>
<AssemblyTitle>Lucene.Net.Analysis.OpenNLP</AssemblyTitle>
<PackageTags>$(PackageTags);analysis;natural;language;processing;opennlp</PackageTags>
<PackageTags>$(PackageTags);analysis;natural;language;processing;opennlp;nopennlp</PackageTags>
<DocumentationFile>bin\$(Configuration)\$(TargetFramework)\$(AssemblyName).xml</DocumentationFile>
<NoWarn>$(NoWarn);1591;1573</NoWarn>
<RootNamespace>Lucene.Net.Analysis.OpenNlp</RootNamespace>
Expand All @@ -53,13 +46,7 @@

<ItemGroup>
<PackageReference Include="ICU4N" Version="$(ICU4NPackageVersion)" />
<PackageReference Include="IKVM" Version="$(IKVMPackageVersion)" />
<PackageReference Include="IKVM.Maven.Sdk" Version="$(IKVMMavenSdkPackageVersion)" />
</ItemGroup>

<ItemGroup>
<MavenReference Include="org.apache.opennlp:opennlp-tools" Version="$(OpenNLPToolsMavenReferenceVersion)" />
<MavenReference Include="org.osgi:org.osgi.core" Version="$(OSGICoreMavenReferenceVersion)" />
<PackageReference Include="NOpenNLP.Tools" Version="$(NOpenNLPPackageVersion)" />
</ItemGroup>

</Project>
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
using ICU4N.Text;
using Lucene.Net.Analysis.OpenNlp.Tools;
using Lucene.Net.Analysis.Util;
using opennlp.tools.util;
using NOpenNLP.Tools.Util;
using System;
using System.Diagnostics;
using System.Text;
Expand Down Expand Up @@ -249,7 +249,7 @@ public override void SetText(CharacterIterator newText)
for (int i = 0; i < spans.Length; ++i)
{
// Adjust start positions to match those of the passed-in CharacterIterator
sentenceStarts[i] = spans[i].getStart() + text.BeginIndex;
sentenceStarts[i] = spans[i].Start + text.BeginIndex;
}
}

Expand Down
8 changes: 4 additions & 4 deletions src/Lucene.Net.Analysis.OpenNLP/OpenNLPTokenizer.cs
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
using Lucene.Net.Analysis.TokenAttributes;
using Lucene.Net.Analysis.Util;
using Lucene.Net.Util;
using opennlp.tools.util;
using NOpenNLP.Tools.Util;
using System;
using System.IO;

Expand Down Expand Up @@ -92,9 +92,9 @@ protected override bool IncrementWord()
}
ClearAttributes();
Span term = termSpans[termNum];
termAtt.CopyBuffer(m_buffer, sentenceStart + term.getStart(), term.length());
offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.getStart()),
CorrectOffset(m_offset + sentenceStart + term.getEnd()));
termAtt.CopyBuffer(m_buffer, sentenceStart + term.Start, term.Length);
offsetAtt.SetOffset(CorrectOffset(m_offset + sentenceStart + term.Start),
CorrectOffset(m_offset + sentenceStart + term.End));
if (termNum == termSpans.Length - 1)
{
flagsAtt.Flags = flagsAtt.Flags | EOS_FLAG_BIT; // mark the last token in the sentence with EOS_FLAG_BIT
Expand Down
7 changes: 3 additions & 4 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/NLPChunkerOp.cs
Original file line number Diff line number Diff line change
@@ -1,7 +1,6 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
using opennlp.tools.chunker;

using NOpenNLP.Tools.Chunker;

namespace Lucene.Net.Analysis.OpenNlp.Tools
{
Expand Down Expand Up @@ -40,9 +39,9 @@ public virtual string[] GetChunks(string[] words, string[] tags, double[] probs)
UninterruptableMonitor.Enter(this);
try
{
string[] chunks = chunker.chunk(words, tags);
string[] chunks = chunker.Chunk(words, tags);
if (probs != null)
chunker.probs(probs);
chunker.Probs(probs);
return chunks;
}
finally
Expand Down
11 changes: 6 additions & 5 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/NLPLemmatizerOp.cs
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
// Lucene version compatibility level 8.2.0
using opennlp.tools.lemmatizer;

using NOpenNLP.Tools.Lemmatizer;
using System.Diagnostics;
using System.IO;

Expand Down Expand Up @@ -39,7 +40,7 @@ public class NLPLemmatizerOp
public NLPLemmatizerOp(Stream dictionary, LemmatizerModel lemmatizerModel)
{
Debug.Assert(dictionary != null || lemmatizerModel != null, "At least one parameter must be non-null");
dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(new ikvm.io.InputStreamWrapper(dictionary));
dictionaryLemmatizer = dictionary is null ? null : new DictionaryLemmatizer(dictionary);
lemmatizerME = lemmatizerModel is null ? null : new LemmatizerME(lemmatizerModel);
}

Expand All @@ -49,7 +50,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
string[] maxEntLemmas = null;
if (dictionaryLemmatizer != null)
{
lemmas = dictionaryLemmatizer.lemmatize(words, postags);
lemmas = dictionaryLemmatizer.Lemmatize(words, postags);
for (int i = 0; i < lemmas.Length; ++i)
{
if (lemmas[i].Equals("O"))
Expand All @@ -58,7 +59,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
{ // fall back to the MaxEnt lemmatizer if it's enabled
if (maxEntLemmas is null)
{
maxEntLemmas = lemmatizerME.lemmatize(words, postags);
maxEntLemmas = lemmatizerME.Lemmatize(words, postags);
}
if ("_".Equals(maxEntLemmas[i]))
{
Expand All @@ -78,7 +79,7 @@ public virtual string[] Lemmatize(string[] words, string[] postags)
}
else
{ // there is only a MaxEnt lemmatizer
maxEntLemmas = lemmatizerME.lemmatize(words, postags);
maxEntLemmas = lemmatizerME.Lemmatize(words, postags);
for (int i = 0; i < maxEntLemmas.Length; ++i)
{
if ("_".Equals(maxEntLemmas[i]))
Expand Down
10 changes: 5 additions & 5 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/NLPNERTaggerOp.cs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
using opennlp.tools.namefind;
using opennlp.tools.util;
using NOpenNLP.Tools.Namefind;
using NOpenNLP.Tools.Util;

namespace Lucene.Net.Analysis.OpenNlp.Tools
{
Expand Down Expand Up @@ -39,7 +39,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools
/// </summary>
public class NLPNERTaggerOp
{
private readonly TokenNameFinder nameFinder;
private readonly ITokenNameFinder nameFinder;

public NLPNERTaggerOp(TokenNameFinderModel model)
{
Expand All @@ -48,7 +48,7 @@ public NLPNERTaggerOp(TokenNameFinderModel model)

public virtual Span[] GetNames(string[] words)
{
Span[] names = nameFinder.find(words);
Span[] names = nameFinder.Find(words);
return names;
}

Expand All @@ -57,7 +57,7 @@ public virtual void Reset()
UninterruptableMonitor.Enter(this);
try
{
nameFinder.clearAdaptiveData();
nameFinder.ClearAdaptiveData();
}
finally
{
Expand Down
6 changes: 3 additions & 3 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/NLPPOSTaggerOp.cs
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
using opennlp.tools.postag;
using NOpenNLP.Tools.Postag;

namespace Lucene.Net.Analysis.OpenNlp.Tools
{
Expand All @@ -27,7 +27,7 @@ namespace Lucene.Net.Analysis.OpenNlp.Tools
/// </summary>
public class NLPPOSTaggerOp
{
private readonly POSTagger tagger = null;
private readonly IPOSTagger tagger = null;

public NLPPOSTaggerOp(POSModel model)
{
Expand All @@ -39,7 +39,7 @@ public virtual string[] GetPOSTags(string[] words)
UninterruptableMonitor.Enter(this);
try
{
return tagger.tag(words);
return tagger.Tag(words);
}
finally
{
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
using opennlp.tools.sentdetect;
using opennlp.tools.util;
using NOpenNLP.Tools.Sentdetect;
using NOpenNLP.Tools.Util;

namespace Lucene.Net.Analysis.OpenNlp.Tools
{
Expand Down Expand Up @@ -47,7 +47,7 @@ public virtual Span[] SplitSentences(string line)
{
if (sentenceSplitter != null)
{
return sentenceSplitter.sentPosDetect(line);
return sentenceSplitter.SentPosDetect(line);
}
else
{
Expand Down
6 changes: 3 additions & 3 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/NLPTokenizerOp.cs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Support.Threading;
using opennlp.tools.tokenize;
using opennlp.tools.util;
using NOpenNLP.Tools.Tokenize;
using NOpenNLP.Tools.Util;

namespace Lucene.Net.Analysis.OpenNlp.Tools
{
Expand Down Expand Up @@ -51,7 +51,7 @@ public virtual Span[] GetTerms(string sentence)
span1[0] = new Span(0, sentence.Length);
return span1;
}
return tokenizer.tokenizePos(sentence);
return tokenizer.TokenizePos(sentence);
}
finally
{
Expand Down
24 changes: 12 additions & 12 deletions src/Lucene.Net.Analysis.OpenNLP/Tools/OpenNLPOpsFactory.cs
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
// Lucene version compatibility level 8.2.0
using Lucene.Net.Analysis.Util;
using opennlp.tools.chunker;
using opennlp.tools.lemmatizer;
using opennlp.tools.namefind;
using opennlp.tools.postag;
using opennlp.tools.sentdetect;
using opennlp.tools.tokenize;
using NOpenNLP.Tools.Chunker;
using NOpenNLP.Tools.Lemmatizer;
using NOpenNLP.Tools.Namefind;
using NOpenNLP.Tools.Postag;
using NOpenNLP.Tools.Sentdetect;
using NOpenNLP.Tools.Tokenize;
using System.Collections.Concurrent;
using System.Diagnostics;
using System.IO;
Expand Down Expand Up @@ -63,7 +63,7 @@ public static SentenceModel GetSentenceModel(string modelName, IResourceLoader l
return sentenceModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new SentenceModel(new ikvm.io.InputStreamWrapper(resource));
return new SentenceModel(resource);
});
}

Expand All @@ -86,7 +86,7 @@ public static TokenizerModel GetTokenizerModel(string modelName, IResourceLoader
return tokenizerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new TokenizerModel(new ikvm.io.InputStreamWrapper(resource));
return new TokenizerModel(resource);
});
}

Expand All @@ -102,7 +102,7 @@ public static POSModel GetPOSTaggerModel(string modelName, IResourceLoader loade
return posTaggerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new POSModel(new ikvm.io.InputStreamWrapper(resource));
return new POSModel(resource);
});
}

Expand All @@ -118,7 +118,7 @@ public static ChunkerModel GetChunkerModel(string modelName, IResourceLoader loa
return chunkerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new ChunkerModel(new ikvm.io.InputStreamWrapper(resource));
return new ChunkerModel(resource);
});
}

Expand All @@ -134,7 +134,7 @@ public static TokenNameFinderModel GetNERTaggerModel(string modelName, IResource
return nerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new TokenNameFinderModel(new ikvm.io.InputStreamWrapper(resource));
return new TokenNameFinderModel(resource);
});
}

Expand Down Expand Up @@ -178,7 +178,7 @@ public static LemmatizerModel GetLemmatizerModel(string modelName, IResourceLoad
return lemmatizerModels.GetOrAdd(modelName, (modelName) =>
{
using Stream resource = loader.OpenResource(modelName);
return new LemmatizerModel(new ikvm.io.InputStreamWrapper(resource));
return new LemmatizerModel(resource);
});
}

Expand Down
Loading