using System; using System.Collections.Generic; using System.Text; namespace BotSharp.NLP.Tokenize { /// /// A tokenizer is a component used for dividing text intotokens. /// A tokenizer is language specific and takes into account the peculiarities of the language, e.g. don’t in English is tokenized as two tokens. /// public interface ITokenizer { /// /// Tokenize /// /// input sentence /// Options such as: regex expression /// List Tokenize(string sentence, TokenizationOptions options); } }