From a8c5242c7b0d7f8266f71c7907c66db2b9a0fddf Mon Sep 17 00:00:00 2001 From: Esther2013 Date: Wed, 12 Sep 2018 23:33:34 -0500 Subject: [PATCH] extractor key words from document. --- BotSharp.NLP/Classify/NaiveBayesClassifier.cs | 8 +- BotSharp.NLP/Featuring/IFeatureExtractor.cs | 10 ++ .../Featuring/TfIdfFeatureExtractor.cs | 155 ++++++++++++++++++ BotSharp.NLP/Tokenize/Token.cs | 2 + BotSharp.NLP/Tokenize/TokenizerFactory.cs | 5 +- BotSharp.NLP/Txt2Vec/OneHotEncoder.cs | 9 +- BotSharp.NLP/Txt2Vec/TFIDF.cs | 148 ----------------- 7 files changed, 178 insertions(+), 159 deletions(-) create mode 100644 BotSharp.NLP/Featuring/IFeatureExtractor.cs create mode 100644 BotSharp.NLP/Featuring/TfIdfFeatureExtractor.cs delete mode 100644 BotSharp.NLP/Txt2Vec/TFIDF.cs diff --git a/BotSharp.NLP/Classify/NaiveBayesClassifier.cs b/BotSharp.NLP/Classify/NaiveBayesClassifier.cs index 3baaef1b..38f28d7b 100644 --- a/BotSharp.NLP/Classify/NaiveBayesClassifier.cs +++ b/BotSharp.NLP/Classify/NaiveBayesClassifier.cs @@ -22,6 +22,7 @@ using BotSharp.Algorithm.Estimators; using BotSharp.Algorithm.Extensions; using BotSharp.Algorithm.Features; using BotSharp.Algorithm.Statistics; +using BotSharp.NLP.Featuring; using BotSharp.NLP.Txt2Vec; using Newtonsoft.Json; using System; @@ -53,10 +54,11 @@ namespace BotSharp.NLP.Classify public void Train(List sentences, ClassifyOptions options) { - var tfidf = new TFIDF(); + var tfidf = new TfIdfFeatureExtractor(); tfidf.Sentences = sentences; - words = tfidf.EncodeAll(); - + tfidf.CalBasedOnCategory(); + var keyWords = tfidf.Features(); + string keywords2 = String.Join(",", keyWords.ToArray()); var encoder = new OneHotEncoder(); encoder.Sentences = sentences; words = encoder.EncodeAll(); diff --git a/BotSharp.NLP/Featuring/IFeatureExtractor.cs b/BotSharp.NLP/Featuring/IFeatureExtractor.cs new file mode 100644 index 00000000..785f1d30 --- /dev/null +++ b/BotSharp.NLP/Featuring/IFeatureExtractor.cs @@ -0,0 +1,10 @@ +using System; +using System.Collections.Generic; +using System.Text; + +namespace BotSharp.NLP.Featuring +{ + public interface IFeatureExtractor + { + } +} diff --git a/BotSharp.NLP/Featuring/TfIdfFeatureExtractor.cs b/BotSharp.NLP/Featuring/TfIdfFeatureExtractor.cs new file mode 100644 index 00000000..8f3ccf77 --- /dev/null +++ b/BotSharp.NLP/Featuring/TfIdfFeatureExtractor.cs @@ -0,0 +1,155 @@ +/* + * BotSharp.NLP Library + * Copyright (C) 2018 Haiping Chen + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, either version 3 of the License, or + * (at your option) any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + */ + +using BotSharp.NLP.Tokenize; +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Runtime.Serialization.Formatters.Binary; +using System.Text; +using System.Text.RegularExpressions; + +namespace BotSharp.NLP.Featuring +{ + public class TfIdfFeatureExtractor : IFeatureExtractor + { + public List Sentences { get; set; } + + private List> tfs; + + private List Categories { get; set; } + + public void Extract(Sentence sentence) + { + + } + + public List Features() + { + var tfs2 = tfs.OrderByDescending(x => x.Item2) + .Select(x => x.Item1) + .Distinct() + .Take(Sentences.Count / Categories.Count) + .ToList(); + + return tfs2; + } + + public void CalBasedOnSentence() + { + Categories = Sentences.Select(x => x.Label).Distinct().ToList(); + + tfs = new List>(); + + Sentences.ForEach(sent => + { + sent.Words.ForEach(word => + { + // TF + int c1 = sent.Words.Count(x => x.Lemma == word.Lemma); + double tf = (c1 + 1.0) / sent.Words.Count(); + + // IDF + var c2 = Sentences.Count(s => s.Words.Select(x => x.Lemma).Contains(word.Lemma)); + double idf = Math.Log(Sentences.Count / (c2 + 1.0)); + + word.Vector = tf * idf; + + tfs.Add(new Tuple(word.Lemma, word.Vector)); + }); + }); + } + + public void CalBasedOnCategory() + { + tfs = new List>(); + + Categories = Sentences.Select(x => x.Label).Distinct().ToList(); + + Categories.ForEach(label => + { + var allTokens = new List(); + Sentences.Where(x => x.Label == label) + .ToList() + .ForEach(s => allTokens.AddRange(s.Words)); + + allTokens.Select(x => x.Lemma).Distinct() + .ToList() + .ForEach(word => + { + // TF + int c1 = allTokens.Count(x => x.Lemma == word); + double tf = (c1 + 1.0) / allTokens.Count(); + + // IDF + var c2 = Sentences.Where(s => s.Words.Select(x => x.Lemma).Contains(word)) + .GroupBy(x => x.Label).Count(); + double idf = Math.Log(Categories.Count / (c2 + 1.0)); + + tfs.Add(new Tuple(word, tf * idf)); + }); + }); + } + + /// + /// Normalizes a TF*IDF array of vectors using L2-Norm. + /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + /// + /// List> + /// List> + public static List> Normalize(List> vectors) + { + // Normalize the vectors using L2-Norm. + List> normalizedVectors = new List>(); + foreach (var vector in vectors) + { + var normalized = Normalize(vector); + normalizedVectors.Add(normalized); + } + + return normalizedVectors; + } + + /// + /// Normalizes a TF*IDF vector using L2-Norm. + /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + /// + /// List + /// List + public static List Normalize(List vector) + { + List result = new List(); + + double sumSquared = 0; + foreach (var value in vector) + { + sumSquared += value * value; + } + + double SqrtSumSquared = Math.Sqrt(sumSquared); + + foreach (var value in vector) + { + // L2-norm: Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) + result.Add(value / SqrtSumSquared); + } + return result; + } + } +} diff --git a/BotSharp.NLP/Tokenize/Token.cs b/BotSharp.NLP/Tokenize/Token.cs index e6637642..4013fe52 100644 --- a/BotSharp.NLP/Tokenize/Token.cs +++ b/BotSharp.NLP/Tokenize/Token.cs @@ -67,5 +67,7 @@ namespace BotSharp.NLP.Tokenize { return $"{Text} {Start} {Pos}"; } + + public double Vector { get; set; } } } diff --git a/BotSharp.NLP/Tokenize/TokenizerFactory.cs b/BotSharp.NLP/Tokenize/TokenizerFactory.cs index 5518326e..fd8b22ec 100644 --- a/BotSharp.NLP/Tokenize/TokenizerFactory.cs +++ b/BotSharp.NLP/Tokenize/TokenizerFactory.cs @@ -29,7 +29,9 @@ namespace BotSharp.NLP.Tokenize public List Tokenize(string sentence) { - return _tokenizer.Tokenize(sentence, _options); + var tokens = _tokenizer.Tokenize(sentence, _options); + tokens.ForEach(x => x.Lemma = x.Text.ToLower()); + return tokens; } public List Tokenize(List sentences) @@ -39,6 +41,7 @@ namespace BotSharp.NLP.Tokenize Parallel.ForEach(sents, (sentence) => { sentence.Words = Tokenize(sentence.Text); + sentence.Words.ForEach(x => x.Lemma = x.Text.ToLower()); }); return sents; diff --git a/BotSharp.NLP/Txt2Vec/OneHotEncoder.cs b/BotSharp.NLP/Txt2Vec/OneHotEncoder.cs index 6c3152a0..510a37d9 100644 --- a/BotSharp.NLP/Txt2Vec/OneHotEncoder.cs +++ b/BotSharp.NLP/Txt2Vec/OneHotEncoder.cs @@ -25,7 +25,7 @@ namespace BotSharp.NLP.Txt2Vec sentence.Words.ForEach(w => { - int index = Words.IndexOf(w.Text.ToLower()); + int index = Words.IndexOf(w.Lemma.ToLower()); if(index > 0) { vector[index] = 1; @@ -49,12 +49,7 @@ namespace BotSharp.NLP.Txt2Vec { if (Words == null) { - Words = new List(); - Sentences.ForEach(x => - { - Words.AddRange(x.Words.Where(w => w.IsAlpha).Select(w => w.Text.ToLower())); - }); - Words = Words.Distinct().OrderBy(x => x).ToList(); + // Words = "shuffle,pause,resume,next,stop,previous,continue,mode,repeat,back,music,play,enough,off,them,playlist,skip,restart,favourites,on,add,go,again,turn,save,my,station,favourite,start,by,playing,please,now,running,move".Split(',').ToList(); } return Words; diff --git a/BotSharp.NLP/Txt2Vec/TFIDF.cs b/BotSharp.NLP/Txt2Vec/TFIDF.cs deleted file mode 100644 index b7e64a94..00000000 --- a/BotSharp.NLP/Txt2Vec/TFIDF.cs +++ /dev/null @@ -1,148 +0,0 @@ -/// -/// Copyright (c) 2018 Bo Peng -/// -/// Permission is hereby granted, free of charge, to any person obtaining -/// a copy of this software and associated documentation files (the -/// "Software"), to deal in the Software without restriction, including -/// without limitation the rights to use, copy, modify, merge, publish, -/// distribute, sublicense, and/or sell copies of the Software, and to -/// permit persons to whom the Software is furnished to do so, subject to -/// the following conditions: -/// -/// The above copyright notice and this permission notice shall be -/// included in all copies or substantial portions of the Software. -/// -/// -using BotSharp.NLP.Tokenize; -using System; -using System.Collections.Generic; -using System.IO; -using System.Linq; -using System.Runtime.Serialization.Formatters.Binary; -using System.Text; -using System.Text.RegularExpressions; - -namespace BotSharp.NLP.Txt2Vec -{ - public class TFIDF - { - public List Sentences { get; set; } - - public List Words { get; set; } - - public void Encode(Sentence sentence) - { - InitDictionary(); - - // var featureSets = Sentences.Select(x => new Tuple(x.Label, x.Vector)).ToList(); - - var labelDist = Sentences.Select(x => x.Label).Distinct().ToList(); - - labelDist.ForEach(label => - { - // https://zhuanlan.zhihu.com/p/31197209 - // calculate TF - // all words in the article - List words = new List(); - Sentences.Where(x => x.Label == label).ToList().ForEach(sent => - { - words.AddRange(sent.Words.Select(w => w.Text)); - }); - - List> tfs = new List>(); - words.Distinct().ToList().ForEach(w => - { - // TF - int c1 = words.Count(x => x == w); - double tf = (c1 + 1.0) / words.Count(); - - // IDF - var sents = Sentences.Where(s => s.Words.Select(x => x.Text).Contains(w)).ToList(); - double idf = Math.Log(Sentences.Count / (sents.Count() + 1.0)); - - tfs.Add(new Tuple(w, tf * idf)); - }); - - tfs = tfs.OrderByDescending(x => x.Item2).Take(words.Count / 10).ToList(); - }); - - - - sentence.Words.ForEach(w => - { - int index = Words.IndexOf(w.Text.ToLower()); - }); - } - - public List EncodeAll() - { - InitDictionary(); - - Sentences.ForEach(sent => Encode(sent)); - //Parallel.ForEach(Sentences, sent => Encode(sent)); - - return Words; - } - - private List InitDictionary() - { - if (Words == null) - { - Words = new List(); - Sentences.ForEach(x => - { - Words.AddRange(x.Words.Where(w => w.IsAlpha).Select(w => w.Text.ToLower())); - }); - Words = Words.Distinct().OrderBy(x => x).ToList(); - } - - return Words; - } - - - /// - /// Normalizes a TF*IDF array of vectors using L2-Norm. - /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - /// - /// List> - /// List> - public static List> Normalize(List> vectors) - { - // Normalize the vectors using L2-Norm. - List> normalizedVectors = new List>(); - foreach (var vector in vectors) - { - var normalized = Normalize(vector); - normalizedVectors.Add(normalized); - } - - return normalizedVectors; - } - - /// - /// Normalizes a TF*IDF vector using L2-Norm. - /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - /// - /// List - /// List - public static List Normalize(List vector) - { - List result = new List(); - - double sumSquared = 0; - foreach (var value in vector) - { - sumSquared += value * value; - } - - double SqrtSumSquared = Math.Sqrt(sumSquared); - - foreach (var value in vector) - { - // L2-norm: Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) - result.Add(value / SqrtSumSquared); - } - return result; - } - } -}