.NET下文本相似度算法余弦定理和SimHash浅析及应用实例分析

2019-05-23 06:12:16刘景俊

        }
 
        private void GenerateTermFrequency()
        {
            for(int i=0; i < _numDocs  ; i++)
            {                               
                string curDoc=_docs[i];
                IDictionary freq=GetWordFrequency(curDoc);
                IDictionaryEnumerator enums=freq.GetEnumerator() ;
                _maxTermFreq[i]=int.MinValue ;
                while (enums.MoveNext())
                {
                    string word=(string)enums.Key;
                    int wordFreq=(int)enums.Value ;
                    int termIndex=GetTermIndex(word);
 
                    _termFreq [termIndex][i]=wordFreq;
                    _docFreq[termIndex] ++;
 
                    if (wordFreq > _maxTermFreq[i]) _maxTermFreq[i]=wordFreq;                   
                }
            }
        }

        private void GenerateTermWeight()
        {           
            for(int i=0; i < _numTerms   ; i++)
            {
                for(int j=0; j < _numDocs ; j++)