diff --git a/BotSharp.NLP.UnitTest/SVMClassifierTest.cs b/BotSharp.NLP.UnitTest/SVMClassifierTest.cs index f9b1d48a..732af1f5 100644 --- a/BotSharp.NLP.UnitTest/SVMClassifierTest.cs +++ b/BotSharp.NLP.UnitTest/SVMClassifierTest.cs @@ -1,6 +1,5 @@ using BotSharp.NLP.Classify; using BotSharp.NLP.Corpus; -using BotSharp.NLP.Models.TF_IDF; using BotSharp.NLP.Tokenize; using Microsoft.Extensions.Configuration; using Microsoft.VisualStudio.TestTools.UnitTesting; @@ -29,8 +28,8 @@ namespace BotSharp.NLP.UnitTest "see you Bolo", "byebye Haiping" }; - TFIDFGenerator tfidfGenerator = new TFIDFGenerator(); - List> weights = tfidfGenerator.TFIDFWeightVectorsForSentences(documents); + /*TFIDFGenerator tfidfGenerator = new TFIDFGenerator(); + List> weights = tfidfGenerator.TFIDFWeightVectorsForSentences(documents);*/ } [TestMethod] diff --git a/BotSharp.NLP/Classify/NaiveBayesClassifier.cs b/BotSharp.NLP/Classify/NaiveBayesClassifier.cs index e29a881d..3baaef1b 100644 --- a/BotSharp.NLP/Classify/NaiveBayesClassifier.cs +++ b/BotSharp.NLP/Classify/NaiveBayesClassifier.cs @@ -53,6 +53,10 @@ namespace BotSharp.NLP.Classify public void Train(List sentences, ClassifyOptions options) { + var tfidf = new TFIDF(); + tfidf.Sentences = sentences; + words = tfidf.EncodeAll(); + var encoder = new OneHotEncoder(); encoder.Sentences = sentences; words = encoder.EncodeAll(); diff --git a/BotSharp.NLP/Models/TF-IDF/TFIDF.cs b/BotSharp.NLP/Models/TF-IDF/TFIDF.cs deleted file mode 100644 index 1fcf67f6..00000000 --- a/BotSharp.NLP/Models/TF-IDF/TFIDF.cs +++ /dev/null @@ -1,232 +0,0 @@ -using BotSharp.NLP.Tokenize; -using System; -using System.Collections.Generic; -using System.IO; -using System.Linq; -using System.Runtime.Serialization.Formatters.Binary; -using System.Text; -using System.Text.RegularExpressions; - -namespace BotSharp.NLP.Models.TF_IDF -{ - /// - /// Copyright (c) 2018 Bo Peng - /// - /// Permission is hereby granted, free of charge, to any person obtaining - /// a copy of this software and associated documentation files (the - /// "Software"), to deal in the Software without restriction, including - /// without limitation the rights to use, copy, modify, merge, publish, - /// distribute, sublicense, and/or sell copies of the Software, and to - /// permit persons to whom the Software is furnished to do so, subject to - /// the following conditions: - /// - /// The above copyright notice and this permission notice shall be - /// included in all copies or substantial portions of the Software. - /// - public class TFIDF - { - List vocabulary { get; set; } - - public TFIDF() - { - } - - /// - /// Document vocabulary, containing each word's IDF value. - /// - private static Dictionary _vocabularyIDF = new Dictionary(); - - public static List> GetTFIDFWeightsVectors(string[] documents, int vocabularyThreshold = 1) - { - List> stemmedDocs; - List vocabulary; - // Get the vocabulary and stem the documents at the same time. - vocabulary = GetVocabulary(documents, out stemmedDocs, vocabularyThreshold); - if (_vocabularyIDF.Count == 0) - { - // Calculate the IDF for each vocabulary term. - foreach (var term in vocabulary) - { - double numberOfDocsContainingTerm = stemmedDocs.Where(d => d.Contains(term)).Count(); - _vocabularyIDF[term] = Math.Log((double)stemmedDocs.Count / ((double)1 + numberOfDocsContainingTerm)); - } - } - // Transform each document into a vector of tfidf values. - List> vectors = new List>(); - foreach (var doc in stemmedDocs) - { - List vector = new List(); - foreach (string word in doc) - { - double tf = doc.Where(d => d == word).Count(); - double tfidf = tf * _vocabularyIDF[word]; - vector.Add(tfidf); - } - vectors.Add(vector); - } - return vectors; - } - - - /// - /// Normalizes a TF*IDF array of vectors using L2-Norm. - /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - /// - /// List> - /// List> - public static List> Normalize(List> vectors) - { - // Normalize the vectors using L2-Norm. - List> normalizedVectors = new List>(); - foreach (var vector in vectors) - { - var normalized = Normalize(vector); - normalizedVectors.Add(normalized); - } - - return normalizedVectors; - } - - /// - /// Normalizes a TF*IDF vector using L2-Norm. - /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - /// - /// List - /// List - public static List Normalize(List vector) - { - List result = new List(); - - double sumSquared = 0; - foreach (var value in vector) - { - sumSquared += value * value; - } - - double SqrtSumSquared = Math.Sqrt(sumSquared); - - foreach (var value in vector) - { - // L2-norm: Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - result.Add(value / SqrtSumSquared); - } - return result; - } - - /// - /// Saves the TFIDF vocabulary to disk. - /// - /// File path - public static void Save(string filePath = "vocabulary.dat") - { - // Save result to disk. - using (FileStream fs = new FileStream(filePath, FileMode.Create)) - { - BinaryFormatter formatter = new BinaryFormatter(); - formatter.Serialize(fs, _vocabularyIDF); - } - } - - /// - /// Loads the TFIDF vocabulary from disk. - /// - /// File path - public static void Load(string filePath = "vocabulary.dat") - { - // Load from disk. - using (FileStream fs = new FileStream(filePath, FileMode.Open)) - { - BinaryFormatter formatter = new BinaryFormatter(); - _vocabularyIDF = (Dictionary)formatter.Deserialize(fs); - } - } - - /// - /// Parses and tokenizes a list of documents, returning a vocabulary of words. - /// - /// string[] - /// List of List of string - /// Vocabulary (list of strings) - private static List GetVocabulary(string[] docs, out List> stemmedDocs, int vocabularyThreshold) - { - List vocabulary = new List(); - Dictionary wordCountList = new Dictionary(); - stemmedDocs = new List>(); - int docIndex = 0; - var tokenizer = new TokenizerFactory(new TokenizationOptions - { - Pattern = RegexTokenizer.WHITE_SPACE - }, SupportedLanguage.English); - - foreach (var doc in docs) - { - List stemmedDoc = new List(); - docIndex++; - if (docIndex % 100 == 0) - { - Console.WriteLine("Processing " + docIndex + "/" + docs.Length); - } - - List tokens = tokenizer.Tokenize(doc); - List list = new List(); - tokenizer.Tokenize(doc).ForEach( token => { - list.Add(token.Text.ToLower()); - }); - string[] parts2 = list.ToArray(); - //string[] parts2 = Tokenize(doc); - List words = new List(); - foreach (string part in parts2) - { - // Strip non-alphanumeric characters. - string stripped = Regex.Replace(part, "[^a-zA-Z0-9]", ""); - try - { - var english = new EnglishWord(stripped); - string stem = english.Stem; - words.Add(stem); - - if (stem.Length > 0) - { - // Build the word count list. - if (wordCountList.ContainsKey(stem)) - { - wordCountList[stem]++; - } - else - { - wordCountList.Add(stem, 0); - } - stemmedDoc.Add(stem); - } - } - catch - { - } - } - stemmedDocs.Add(stemmedDoc); - } - // Get the top words. - var vocabList = wordCountList.Where(w => w.Value >= vocabularyThreshold); - foreach (var item in vocabList) - { - vocabulary.Add(item.Key); - } - return vocabulary; - } - - - } - public class EnglishWord - { - public EnglishWord(string input) - { - this.Original = input; - this.Stem = input; - this.Length = input.Length; - } - - public string Stem { get; set; } - public string Original { get; } - public int Length { get; } - } -} diff --git a/BotSharp.NLP/Models/TF-IDF/TFIDFGenerator.cs b/BotSharp.NLP/Models/TF-IDF/TFIDFGenerator.cs deleted file mode 100644 index 670f84b3..00000000 --- a/BotSharp.NLP/Models/TF-IDF/TFIDFGenerator.cs +++ /dev/null @@ -1,30 +0,0 @@ -using System; -using System.Collections.Generic; -using System.Text; - -namespace BotSharp.NLP.Models.TF_IDF -{ - /// - /// Copyright (c) 2018 Bo Peng - /// - /// Permission is hereby granted, free of charge, to any person obtaining - /// a copy of this software and associated documentation files (the - /// "Software"), to deal in the Software without restriction, including - /// without limitation the rights to use, copy, modify, merge, publish, - /// distribute, sublicense, and/or sell copies of the Software, and to - /// permit persons to whom the Software is furnished to do so, subject to - /// the following conditions: - /// - /// The above copyright notice and this permission notice shall be - /// included in all copies or substantial portions of the Software. - /// - public class TFIDFGenerator - { - public List> TFIDFWeightVectorsForSentences(string[]documents) - { - List> res = TFIDF.GetTFIDFWeightsVectors(documents, 0); - res = TFIDF.Normalize(res); - return res; - } - } -} diff --git a/BotSharp.NLP/Txt2Vec/TFIDF.cs b/BotSharp.NLP/Txt2Vec/TFIDF.cs new file mode 100644 index 00000000..b7e64a94 --- /dev/null +++ b/BotSharp.NLP/Txt2Vec/TFIDF.cs @@ -0,0 +1,148 @@ +/// +/// Copyright (c) 2018 Bo Peng +/// +/// Permission is hereby granted, free of charge, to any person obtaining +/// a copy of this software and associated documentation files (the +/// "Software"), to deal in the Software without restriction, including +/// without limitation the rights to use, copy, modify, merge, publish, +/// distribute, sublicense, and/or sell copies of the Software, and to +/// permit persons to whom the Software is furnished to do so, subject to +/// the following conditions: +/// +/// The above copyright notice and this permission notice shall be +/// included in all copies or substantial portions of the Software. +/// +/// +using BotSharp.NLP.Tokenize; +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Runtime.Serialization.Formatters.Binary; +using System.Text; +using System.Text.RegularExpressions; + +namespace BotSharp.NLP.Txt2Vec +{ + public class TFIDF + { + public List Sentences { get; set; } + + public List Words { get; set; } + + public void Encode(Sentence sentence) + { + InitDictionary(); + + // var featureSets = Sentences.Select(x => new Tuple(x.Label, x.Vector)).ToList(); + + var labelDist = Sentences.Select(x => x.Label).Distinct().ToList(); + + labelDist.ForEach(label => + { + // https://zhuanlan.zhihu.com/p/31197209 + // calculate TF + // all words in the article + List words = new List(); + Sentences.Where(x => x.Label == label).ToList().ForEach(sent => + { + words.AddRange(sent.Words.Select(w => w.Text)); + }); + + List> tfs = new List>(); + words.Distinct().ToList().ForEach(w => + { + // TF + int c1 = words.Count(x => x == w); + double tf = (c1 + 1.0) / words.Count(); + + // IDF + var sents = Sentences.Where(s => s.Words.Select(x => x.Text).Contains(w)).ToList(); + double idf = Math.Log(Sentences.Count / (sents.Count() + 1.0)); + + tfs.Add(new Tuple(w, tf * idf)); + }); + + tfs = tfs.OrderByDescending(x => x.Item2).Take(words.Count / 10).ToList(); + }); + + + + sentence.Words.ForEach(w => + { + int index = Words.IndexOf(w.Text.ToLower()); + }); + } + + public List EncodeAll() + { + InitDictionary(); + + Sentences.ForEach(sent => Encode(sent)); + //Parallel.ForEach(Sentences, sent => Encode(sent)); + + return Words; + } + + private List InitDictionary() + { + if (Words == null) + { + Words = new List(); + Sentences.ForEach(x => + { + Words.AddRange(x.Words.Where(w => w.IsAlpha).Select(w => w.Text.ToLower())); + }); + Words = Words.Distinct().OrderBy(x => x).ToList(); + } + + return Words; + } + + + /// + /// Normalizes a TF*IDF array of vectors using L2-Norm. + /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + /// + /// List> + /// List> + public static List> Normalize(List> vectors) + { + // Normalize the vectors using L2-Norm. + List> normalizedVectors = new List>(); + foreach (var vector in vectors) + { + var normalized = Normalize(vector); + normalizedVectors.Add(normalized); + } + + return normalizedVectors; + } + + /// + /// Normalizes a TF*IDF vector using L2-Norm. + /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + /// + /// List + /// List + public static List Normalize(List vector) + { + List result = new List(); + + double sumSquared = 0; + foreach (var value in vector) + { + sumSquared += value * value; + } + + double SqrtSumSquared = Math.Sqrt(sumSquared); + + foreach (var value in vector) + { + // L2-norm: Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + result.Add(value / SqrtSumSquared); + } + return result; + } + } +} diff --git a/BotSharp.NLP/Txt2Vec/VectorGenerator.cs b/BotSharp.NLP/Txt2Vec/VectorGenerator.cs index dba24342..7c848031 100644 --- a/BotSharp.NLP/Txt2Vec/VectorGenerator.cs +++ b/BotSharp.NLP/Txt2Vec/VectorGenerator.cs @@ -5,7 +5,6 @@ using System.Text; using System.Threading.Tasks; using System.IO; using System.Threading; -using BotSharp.NLP.Models.TF_IDF; //using AdvUtils; namespace Txt2Vec @@ -37,8 +36,8 @@ namespace Txt2Vec public List Sentence2Vec(List sentences, WeightingScheme weightingScheme = WeightingScheme.AVG) { // Inplementing TF-IDF - TFIDFGenerator tfidfGenerator = new TFIDFGenerator(); - List> weights = tfidfGenerator.TFIDFWeightVectorsForSentences(sentences.ToArray()); + // TFIDFGenerator tfidfGenerator = new TFIDFGenerator(); + List> weights = null;// tfidfGenerator.TFIDFWeightVectorsForSentences(sentences.ToArray()); List> matixList = new List>();