create a general ngram tagger.
This commit is contained in:
parent
fbd4c69b56
commit
e97a38da0d
|
|
@ -14,22 +14,15 @@ namespace BotSharp.NLP.UnitTest
|
||||||
[TestMethod]
|
[TestMethod]
|
||||||
public void TagInCoNLL2000()
|
public void TagInCoNLL2000()
|
||||||
{
|
{
|
||||||
var sentences = new CoNLLReader()
|
var tokenizer = new TokenizerFactory<RegexTokenizer>(new TokenizationOptions { }, SupportedLanguage.English);
|
||||||
.Read(new ReaderOptions
|
var tokens = tokenizer.Tokenize("How are you doing?");
|
||||||
{
|
|
||||||
DataDir = AppDomain.CurrentDomain.BaseDirectory,
|
|
||||||
FileName = "conll2000_chunking_train"
|
|
||||||
});
|
|
||||||
|
|
||||||
var tagger = new TaggerFactory<DefaultTagger>();
|
var tagger = new TaggerFactory<DefaultTagger>(new TagOptions
|
||||||
|
{
|
||||||
|
Tag = "NN"
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
// tokenize
|
tagger.Tag(new Sentence { Words = tokens });
|
||||||
|
|
||||||
tagger.Tag(null,
|
|
||||||
new TagOptions
|
|
||||||
{
|
|
||||||
|
|
||||||
});
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
46
BotSharp.NLP.UnitTest/NGramTaggerTest.cs
Normal file
46
BotSharp.NLP.UnitTest/NGramTaggerTest.cs
Normal file
|
|
@ -0,0 +1,46 @@
|
||||||
|
using BotSharp.NLP.Corpus;
|
||||||
|
using BotSharp.NLP.Tag;
|
||||||
|
using BotSharp.NLP.Tokenize;
|
||||||
|
using Microsoft.VisualStudio.TestTools.UnitTesting;
|
||||||
|
using System;
|
||||||
|
using System.Collections.Generic;
|
||||||
|
using System.IO;
|
||||||
|
using System.Text;
|
||||||
|
|
||||||
|
namespace BotSharp.NLP.UnitTest
|
||||||
|
{
|
||||||
|
[TestClass]
|
||||||
|
public class NGramTaggerTest
|
||||||
|
{
|
||||||
|
[TestMethod]
|
||||||
|
public void TagInCoNLL2000()
|
||||||
|
{
|
||||||
|
// tokenization
|
||||||
|
var tokenizer = new TokenizerFactory<RegexTokenizer>(new TokenizationOptions
|
||||||
|
{
|
||||||
|
Pattern = RegexTokenizer.WORD_PUNC
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
|
var tokens = tokenizer.Tokenize("How are you doing?");
|
||||||
|
|
||||||
|
// get training corpus
|
||||||
|
string corpusDir = Environment.GetEnvironmentVariable("BOTSHARP_CORPUS_PATH", EnvironmentVariableTarget.User);
|
||||||
|
var sentences = new CoNLLReader()
|
||||||
|
.Read(new ReaderOptions
|
||||||
|
{
|
||||||
|
DataDir = Path.Combine(corpusDir, "CoNLL"),
|
||||||
|
FileName = "conll2000_chunking_train.txt"
|
||||||
|
});
|
||||||
|
|
||||||
|
// start tag
|
||||||
|
var tagger = new TaggerFactory<NGramTagger>(new TagOptions
|
||||||
|
{
|
||||||
|
NGram = 2,
|
||||||
|
Tag = "NN",
|
||||||
|
Corpus = sentences
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
|
tagger.Tag(new Sentence { Words = tokens });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -12,13 +12,12 @@ namespace BotSharp.NLP.UnitTest
|
||||||
[TestMethod]
|
[TestMethod]
|
||||||
public void StemInDefault()
|
public void StemInDefault()
|
||||||
{
|
{
|
||||||
var stemmer = new StemmerFactory<RegexStemmer>();
|
var stemmer = new StemmerFactory<RegexStemmer>(new StemOptions
|
||||||
|
{
|
||||||
|
Pattern = RegexStemmer.DEFAULT
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
var stem = stemmer.Stem("doing",
|
var stem = stemmer.Stem("doing");
|
||||||
new StemOptions
|
|
||||||
{
|
|
||||||
Pattern = RegexStemmer.DEFAULT
|
|
||||||
});
|
|
||||||
|
|
||||||
Assert.IsTrue(stem == "do");
|
Assert.IsTrue(stem == "do");
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -9,13 +9,12 @@ namespace BotSharp.NLP.UnitTest
|
||||||
[TestMethod]
|
[TestMethod]
|
||||||
public void TokenizeInWhiteSpace()
|
public void TokenizeInWhiteSpace()
|
||||||
{
|
{
|
||||||
var tokenizer = new TokenizerFactory<RegexTokenizer>();
|
var tokenizer = new TokenizerFactory<RegexTokenizer>(new TokenizationOptions
|
||||||
|
{
|
||||||
|
Pattern = RegexTokenizer.WHITE_SPACE
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
var tokens = tokenizer.Tokenize("Chop into pieces, isn't it?",
|
var tokens = tokenizer.Tokenize("Chop into pieces, isn't it?");
|
||||||
new TokenizationOptions
|
|
||||||
{
|
|
||||||
Pattern = RegexTokenizer.WHITE_SPACE
|
|
||||||
});
|
|
||||||
|
|
||||||
Assert.IsTrue(tokens[0].Start == 0);
|
Assert.IsTrue(tokens[0].Start == 0);
|
||||||
Assert.IsTrue(tokens[0].Text == "Chop");
|
Assert.IsTrue(tokens[0].Text == "Chop");
|
||||||
|
|
@ -36,13 +35,12 @@ namespace BotSharp.NLP.UnitTest
|
||||||
[TestMethod]
|
[TestMethod]
|
||||||
public void TokenizeInWordPunctuation()
|
public void TokenizeInWordPunctuation()
|
||||||
{
|
{
|
||||||
var tokenizer = new TokenizerFactory<RegexTokenizer>();
|
var tokenizer = new TokenizerFactory<RegexTokenizer>(new TokenizationOptions
|
||||||
|
{
|
||||||
|
Pattern = RegexTokenizer.WORD_PUNC
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
var tokens = tokenizer.Tokenize("Chop into pieces, isn't it?",
|
var tokens = tokenizer.Tokenize("Chop into pieces, isn't it?");
|
||||||
new TokenizationOptions
|
|
||||||
{
|
|
||||||
Pattern = RegexTokenizer.WORD_PUNC
|
|
||||||
});
|
|
||||||
|
|
||||||
Assert.IsTrue(tokens[0].Start == 0);
|
Assert.IsTrue(tokens[0].Start == 0);
|
||||||
Assert.IsTrue(tokens[0].Text == "Chop");
|
Assert.IsTrue(tokens[0].Text == "Chop");
|
||||||
|
|
@ -75,17 +73,16 @@ namespace BotSharp.NLP.UnitTest
|
||||||
[TestMethod]
|
[TestMethod]
|
||||||
public void TokenizeInBlankLine()
|
public void TokenizeInBlankLine()
|
||||||
{
|
{
|
||||||
var tokenizer = new TokenizerFactory<RegexTokenizer>();
|
var tokenizer = new TokenizerFactory<RegexTokenizer>(new TokenizationOptions
|
||||||
|
{
|
||||||
|
Pattern = RegexTokenizer.BLANK_LINE
|
||||||
|
}, SupportedLanguage.English);
|
||||||
|
|
||||||
var tokens = tokenizer.Tokenize(@"Chop into pieces,
|
var tokens = tokenizer.Tokenize(@"Chop into pieces,
|
||||||
|
|
||||||
isn't
|
isn't
|
||||||
|
|
||||||
it?",
|
it?");
|
||||||
new TokenizationOptions
|
|
||||||
{
|
|
||||||
Pattern = RegexTokenizer.BLANK_LINE
|
|
||||||
});
|
|
||||||
|
|
||||||
Assert.IsTrue(tokens[0].Start == 0);
|
Assert.IsTrue(tokens[0].Start == 0);
|
||||||
Assert.IsTrue(tokens[0].Text == "Chop into pieces,");
|
Assert.IsTrue(tokens[0].Text == "Chop into pieces,");
|
||||||
|
|
|
||||||
|
|
@ -1 +1,2 @@
|
||||||
conll2000_chunking is downloaded from https://www.clips.uantwerpen.be/conll2000/chunking/
|
conll2000_chunking is downloaded from https://www.clips.uantwerpen.be/conll2000/chunking/
|
||||||
|
The train and test data consist of three columns separated by spaces. Each word has been put on a separate line and there is an empty line after each sentence. The first column contains the current word, the second its part-of-speech tag as derived by the Brill tagger and the third its chunk tag as derived from the WSJ corpus.
|
||||||
|
|
@ -1,4 +1,5 @@
|
||||||
using System;
|
using BotSharp.NLP.Tokenize;
|
||||||
|
using System;
|
||||||
using System.Collections.Generic;
|
using System.Collections.Generic;
|
||||||
using System.IO;
|
using System.IO;
|
||||||
using System.Text;
|
using System.Text;
|
||||||
|
|
@ -17,9 +18,36 @@ namespace BotSharp.NLP.Corpus
|
||||||
{
|
{
|
||||||
public List<Sentence> Read(ReaderOptions options)
|
public List<Sentence> Read(ReaderOptions options)
|
||||||
{
|
{
|
||||||
string corpus = File.ReadAllText(Path.Combine(options.DataDir, "conll2000_chunking_train.txt"));
|
var sentences = new List<Sentence>();
|
||||||
|
using(StreamReader reader = new StreamReader(Path.Combine(options.DataDir, options.FileName)))
|
||||||
|
{
|
||||||
|
string line = reader.ReadLine();
|
||||||
|
var sentence = new Sentence { Words = new List<Token> { } };
|
||||||
|
|
||||||
return null;
|
while (!reader.EndOfStream)
|
||||||
|
{
|
||||||
|
if (String.IsNullOrEmpty(line))
|
||||||
|
{
|
||||||
|
sentences.Add(sentence);
|
||||||
|
sentence = new Sentence { Words = new List<Token> { } };
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
var columns = line.Split(' ');
|
||||||
|
|
||||||
|
sentence.Words.Add(new Token
|
||||||
|
{
|
||||||
|
Text = columns[0],
|
||||||
|
Pos = columns[1]
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
line = reader.ReadLine();
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
return sentences;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -12,11 +12,6 @@ namespace BotSharp.NLP.Stem
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public interface IStemmer
|
public interface IStemmer
|
||||||
{
|
{
|
||||||
/// <summary>
|
|
||||||
/// Language
|
|
||||||
/// </summary>
|
|
||||||
SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Strip affixes from the token and return the stem.
|
/// Strip affixes from the token and return the stem.
|
||||||
/// </summary>
|
/// </summary>
|
||||||
|
|
|
||||||
|
|
@ -15,8 +15,6 @@ namespace BotSharp.NLP.Stem
|
||||||
{
|
{
|
||||||
public const string DEFAULT = "ing$|s$|e$|able$";
|
public const string DEFAULT = "ing$|s$|e$|able$";
|
||||||
|
|
||||||
public SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
private Regex _regex;
|
private Regex _regex;
|
||||||
|
|
||||||
public string Stem(string word, StemOptions options)
|
public string Stem(string word, StemOptions options)
|
||||||
|
|
|
||||||
|
|
@ -14,16 +14,22 @@ namespace BotSharp.NLP.Stem
|
||||||
/// <typeparam name="IStem"></typeparam>
|
/// <typeparam name="IStem"></typeparam>
|
||||||
public class StemmerFactory<IStem> where IStem : IStemmer, new()
|
public class StemmerFactory<IStem> where IStem : IStemmer, new()
|
||||||
{
|
{
|
||||||
|
private SupportedLanguage _lang { get; set; }
|
||||||
|
|
||||||
private IStem _stemmer;
|
private IStem _stemmer;
|
||||||
|
|
||||||
public StemmerFactory()
|
private StemOptions _options;
|
||||||
|
|
||||||
|
public StemmerFactory(StemOptions options, SupportedLanguage lang)
|
||||||
{
|
{
|
||||||
|
_lang = lang;
|
||||||
|
_options = options;
|
||||||
_stemmer = new IStem();
|
_stemmer = new IStem();
|
||||||
}
|
}
|
||||||
|
|
||||||
public string Stem(string word, StemOptions options)
|
public string Stem(string word)
|
||||||
{
|
{
|
||||||
return _stemmer.Stem(word, options);
|
return _stemmer.Stem(word, _options);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -12,16 +12,14 @@ namespace BotSharp.NLP.Tag
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public class DefaultTagger : ITagger
|
public class DefaultTagger : ITagger
|
||||||
{
|
{
|
||||||
public SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
public void Tag(Sentence sentence, TagOptions options)
|
public void Tag(Sentence sentence, TagOptions options)
|
||||||
{
|
{
|
||||||
throw new NotImplementedException();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
public void Train(List<Sentence> sentences, TagOptions options)
|
public void Train(List<Sentence> sentences, TagOptions options)
|
||||||
{
|
{
|
||||||
throw new NotImplementedException();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -11,11 +11,6 @@ namespace BotSharp.NLP.Tag
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public interface ITagger
|
public interface ITagger
|
||||||
{
|
{
|
||||||
/// <summary>
|
|
||||||
/// Language
|
|
||||||
/// </summary>
|
|
||||||
SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
///
|
///
|
||||||
/// </summary>
|
/// </summary>
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,6 @@
|
||||||
using System;
|
using System;
|
||||||
using System.Collections.Generic;
|
using System.Collections.Generic;
|
||||||
|
using System.Linq;
|
||||||
using System.Text;
|
using System.Text;
|
||||||
using BotSharp.NLP.Tokenize;
|
using BotSharp.NLP.Tokenize;
|
||||||
|
|
||||||
|
|
@ -11,16 +12,55 @@ namespace BotSharp.NLP.Tag
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public class NGramTagger : ITagger
|
public class NGramTagger : ITagger
|
||||||
{
|
{
|
||||||
public SupportedLanguage Lang { get; set; }
|
public Dictionary<string, string> ContextMapping { get; set; }
|
||||||
|
|
||||||
public void Tag(Sentence sentence, TagOptions options)
|
public void Tag(Sentence sentence, TagOptions options)
|
||||||
{
|
{
|
||||||
throw new NotImplementedException();
|
// need training to generate model
|
||||||
|
if(ContextMapping == null)
|
||||||
|
{
|
||||||
|
var cache = new List<Tuple<String, String>>();
|
||||||
|
var contextTag = new List<Tuple<String, String, int>>();
|
||||||
|
|
||||||
|
ContextMapping = new Dictionary<string, string>();
|
||||||
|
|
||||||
|
options.Corpus.ForEach(sent =>
|
||||||
|
{
|
||||||
|
// Supplementary place
|
||||||
|
for (int ngram = 1; ngram < options.NGram; ngram++)
|
||||||
|
{
|
||||||
|
sent.Words.Insert(0, new Token { Text = "NIL", Pos = options.Tag, Start = (ngram - 1) * 3 });
|
||||||
|
}
|
||||||
|
|
||||||
|
int pos = options.NGram - 1;
|
||||||
|
for(pos = 1; pos < sent.Words.Count; pos++)
|
||||||
|
{
|
||||||
|
Token pre = sent.Words[pos - 1];
|
||||||
|
Token cur = sent.Words[pos];
|
||||||
|
|
||||||
|
cache.Add(new Tuple<string, string>($"{pre.Pos} {cur.Text}", cur.Pos));// Dictionary.Add($"{pre.Pos} {cur.Text}", cur.Pos);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
var results = (from c in cache
|
||||||
|
group c by c.Item1 into g
|
||||||
|
select new { g.Key, Count = g.Count() }).ToList();
|
||||||
|
|
||||||
|
results.ForEach(x =>
|
||||||
|
{
|
||||||
|
int count = cache.Count(c => c.Item1 == x.Key);
|
||||||
|
});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
public void Train(List<Sentence> sentences, TagOptions options)
|
public void Train(List<Sentence> sentences, TagOptions options)
|
||||||
{
|
{
|
||||||
throw new NotImplementedException();
|
throw new NotImplementedException();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private class NGramFreq
|
||||||
|
{
|
||||||
|
public string Key { get; set; }
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -16,5 +16,15 @@ namespace BotSharp.NLP.Tag
|
||||||
/// Used in DefaultTagger
|
/// Used in DefaultTagger
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public string Tag { get; set; }
|
public string Tag { get; set; }
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// N-Gram number
|
||||||
|
/// </summary>
|
||||||
|
public int NGram { get; set; }
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Tagged corpus used for training a model
|
||||||
|
/// </summary>
|
||||||
|
public List<Sentence> Corpus { get; set; }
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -6,16 +6,22 @@ namespace BotSharp.NLP.Tag
|
||||||
{
|
{
|
||||||
public class TaggerFactory<ITag> where ITag : ITagger, new()
|
public class TaggerFactory<ITag> where ITag : ITagger, new()
|
||||||
{
|
{
|
||||||
|
private SupportedLanguage _lang;
|
||||||
|
|
||||||
private ITag _tagger;
|
private ITag _tagger;
|
||||||
|
|
||||||
public TaggerFactory()
|
private TagOptions _options;
|
||||||
|
|
||||||
|
public TaggerFactory(TagOptions options, SupportedLanguage lang)
|
||||||
{
|
{
|
||||||
|
_lang = lang;
|
||||||
|
_options = options;
|
||||||
_tagger = new ITag();
|
_tagger = new ITag();
|
||||||
}
|
}
|
||||||
|
|
||||||
public void Tag(Sentence sentence, TagOptions options)
|
public void Tag(Sentence sentence)
|
||||||
{
|
{
|
||||||
_tagger.Tag(sentence, options);
|
_tagger.Tag(sentence, _options);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -10,17 +10,12 @@ namespace BotSharp.NLP.Tokenize
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public interface ITokenizer
|
public interface ITokenizer
|
||||||
{
|
{
|
||||||
/// <summary>
|
|
||||||
/// Language
|
|
||||||
/// </summary>
|
|
||||||
SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Tokenize
|
/// Tokenize
|
||||||
/// </summary>
|
/// </summary>
|
||||||
/// <param name="sentence">input sentence</param>
|
/// <param name="sentence">input sentence</param>
|
||||||
/// <param name="options">Options such as: regex expression</param>
|
/// <param name="options">Options such as: regex expression</param>
|
||||||
/// <returns></returns>
|
/// <returns></returns>
|
||||||
Token[] Tokenize(string sentence, TokenizationOptions options);
|
List<Token> Tokenize(string sentence, TokenizationOptions options);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -11,8 +11,6 @@ namespace BotSharp.NLP.Tokenize
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public class RegexTokenizer : ITokenizer
|
public class RegexTokenizer : ITokenizer
|
||||||
{
|
{
|
||||||
public SupportedLanguage Lang { get; set; }
|
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Tokenize a text into a sequence of alphabetic and non-alphabetic characters
|
/// Tokenize a text into a sequence of alphabetic and non-alphabetic characters
|
||||||
/// </summary>
|
/// </summary>
|
||||||
|
|
@ -34,7 +32,7 @@ namespace BotSharp.NLP.Tokenize
|
||||||
|
|
||||||
private Regex _regex;
|
private Regex _regex;
|
||||||
|
|
||||||
public Token[] Tokenize(string sentence, TokenizationOptions options)
|
public List<Token> Tokenize(string sentence, TokenizationOptions options)
|
||||||
{
|
{
|
||||||
_regex = new Regex(options.Pattern);
|
_regex = new Regex(options.Pattern);
|
||||||
|
|
||||||
|
|
@ -65,7 +63,7 @@ namespace BotSharp.NLP.Tokenize
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return tokens.ToArray();
|
return tokens.ToList();
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
|
@ -73,7 +71,7 @@ namespace BotSharp.NLP.Tokenize
|
||||||
{
|
{
|
||||||
Text = x.Value,
|
Text = x.Value,
|
||||||
Start = x.Index
|
Start = x.Index
|
||||||
}).ToArray();
|
}).ToList();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -55,5 +55,10 @@ namespace BotSharp.NLP.Tokenize
|
||||||
return Start + Text.Length - 1;
|
return Start + Text.Length - 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public override string ToString()
|
||||||
|
{
|
||||||
|
return $"{Text} {Start} {Pos}";
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -12,16 +12,22 @@ namespace BotSharp.NLP.Tokenize
|
||||||
/// </summary>
|
/// </summary>
|
||||||
public class TokenizerFactory<ITokenize> where ITokenize : ITokenizer, new()
|
public class TokenizerFactory<ITokenize> where ITokenize : ITokenizer, new()
|
||||||
{
|
{
|
||||||
|
private SupportedLanguage _lang;
|
||||||
|
|
||||||
private ITokenize _tokenizer;
|
private ITokenize _tokenizer;
|
||||||
|
|
||||||
public TokenizerFactory()
|
private TokenizationOptions _options;
|
||||||
|
|
||||||
|
public TokenizerFactory(TokenizationOptions options, SupportedLanguage lang)
|
||||||
{
|
{
|
||||||
|
_lang = lang;
|
||||||
|
_options = options;
|
||||||
_tokenizer = new ITokenize();
|
_tokenizer = new ITokenize();
|
||||||
}
|
}
|
||||||
|
|
||||||
public Token[] Tokenize(string sentence, TokenizationOptions options)
|
public List<Token> Tokenize(string sentence)
|
||||||
{
|
{
|
||||||
return _tokenizer.Tokenize(sentence, options);
|
return _tokenizer.Tokenize(sentence, _options);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue