/* * BotSharp.NLP Library * Copyright (C) 2018 Haiping Chen * * This program is free software: you can redistribute it and/or modify * it under the terms of the GNU General Public License as published by * the Free Software Foundation, either version 3 of the License, or * (at your option) any later version. * * This program is distributed in the hope that it will be useful, * but WITHOUT ANY WARRANTY; without even the implied warranty of * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * GNU General Public License for more details. * * You should have received a copy of the GNU General Public License * along with this program. If not, see . */ using BotSharp.NLP.Tokenize; using System; using System.Collections.Generic; using System.IO; using System.Linq; using System.Runtime.Serialization.Formatters.Binary; using System.Text; using System.Text.RegularExpressions; namespace BotSharp.NLP.Featuring { public class TfIdfFeatureExtractor : IFeatureExtractor { public List Sentences { get; set; } private List> tfs; private List Categories { get; set; } public void Extract(Sentence sentence) { } public List Features() { var tfs2 = tfs.OrderByDescending(x => x.Item2) .Select(x => x.Item1) .Distinct() .Take(Sentences.Count / Categories.Count) .ToList(); return tfs2; } public void CalBasedOnSentence() { Categories = Sentences.Select(x => x.Label).Distinct().ToList(); tfs = new List>(); Sentences.ForEach(sent => { sent.Words.ForEach(word => { // TF int c1 = sent.Words.Count(x => x.Lemma == word.Lemma); double tf = (c1 + 1.0) / sent.Words.Count(); // IDF var c2 = Sentences.Count(s => s.Words.Select(x => x.Lemma).Contains(word.Lemma)); double idf = Math.Log(Sentences.Count / (c2 + 1.0)); word.Vector = tf * idf; tfs.Add(new Tuple(word.Lemma, word.Vector)); }); }); } public void CalBasedOnCategory() { tfs = new List>(); Categories = Sentences.Select(x => x.Label).Distinct().ToList(); Categories.ForEach(label => { var allTokens = new List(); Sentences.Where(x => x.Label == label) .ToList() .ForEach(s => allTokens.AddRange(s.Words)); allTokens.Select(x => x.Lemma).Distinct() .ToList() .ForEach(word => { // TF int c1 = allTokens.Count(x => x.Lemma == word); double tf = (c1 + 1.0) / allTokens.Count(); // IDF var c2 = Sentences.Where(s => s.Words.Select(x => x.Lemma).Contains(word)) .GroupBy(x => x.Label).Count(); double idf = Math.Log(Categories.Count / (c2 + 1.0)); tfs.Add(new Tuple(word, tf * idf)); }); }); } /// /// Normalizes a TF*IDF array of vectors using L2-Norm. /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) /// /// List> /// List> public static List> Normalize(List> vectors) { // Normalize the vectors using L2-Norm. List> normalizedVectors = new List>(); foreach (var vector in vectors) { var normalized = Normalize(vector); normalizedVectors.Add(normalized); } return normalizedVectors; } /// /// Normalizes a TF*IDF vector using L2-Norm. /// Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) /// /// List /// List public static List Normalize(List vector) { List result = new List(); double sumSquared = 0; foreach (var value in vector) { sumSquared += value * value; } double SqrtSumSquared = Math.Sqrt(sumSquared); foreach (var value in vector) { // L2-norm: Xi = Xi / Sqrt(X0^2 + X1^2 + .. + Xn^2) result.Add(value / SqrtSumSquared); } return result; } } }