#-*-coding:utf-8-*- import MySQLdb import MySQLdb as mdb import os,sys,string import jieba import codecs reload(sys) sys.setdefaultencoding('utf-8') #連接數據庫 try: conn=mdb.connect(host='127.0.0.1',user='root',passwd='kongjunli',db='test1',charset='utf8') except Exception,e: print e sys.exit() #獲取cursor對象操作數據庫 cursor=conn.cursor(mdb.cursors.DictCursor) #cursor游標 #獲取內容 sql='SELECT link,content FROM test1.spider;' cursor.execute(sql) #execute()方法,將字符串當命令執行 data=cursor.fetchall()#fetchall()接收全部返回結果行 f=codecs.open('C:UserskkDesktophello-result1.txt','w','utf-8') for row in data: #row接收結果行的每行數據 seg='/'.join(list(jieba.cut(row['content'],cut_all='False'))) f.write(row['link']+' '+seg+' ') f.close() cursor.close() #提交事務,在插入數據時必須
jiansuo.py
#-*-coding:utf-8-*- import sys import string import MySQLdb import MySQLdb as mdb import gensim from gensim import corpora,models,similarities from gensim.similarities import MatrixSimilarity import logging import codecs reload(sys) sys.setdefaultencoding('utf-8') con=mdb.connect(host='127.0.0.1',user='root',passwd='kongjunli',db='test1',charset='utf8') with con: cur=con.cursor() cur.execute('SELECT * FROM cutresult_copy') rows=cur.fetchall() class MyCorpus(object): def __iter__(self): for row in rows: yield str(row[1]).split('/') #開啟日志 logging.basicConfig(format='%(asctime)s:%(levelname)s:%(message)s',level=logging.INFO) Corp=MyCorpus() #將網頁文檔轉化為tf-idf dictionary=corpora.Dictionary(Corp) corpus=[dictionary.doc2bow(text) for text in Corp] #將文檔轉化為詞袋模型 #print corpus tfidf=models.TfidfModel(corpus)#使用tf-idf模型得出文檔的tf-idf模型 corpus_tfidf=tfidf[corpus]#計算得出tf-idf值 #for doc in corpus_tfidf: #print doc ### ''' q_file=open('C:UserskkDesktopq.txt','r') query=q_file.readline() q_file.close() vec_bow=dictionary.doc2bow(query.split(' '))#將請求轉化為詞帶模型 vec_tfidf=tfidf[vec_bow]#計算出請求的tf-idf值 #for t in vec_tfidf: # print t ''' ### query=raw_input('Enter your query:') vec_bow=dictionary.doc2bow(query.split()) vec_tfidf=tfidf[vec_bow] index=similarities.MatrixSimilarity(corpus_tfidf) sims=index[vec_tfidf] similarity=list(sims) print sorted(similarity,reverse=True)
encodings.xml
<?xml version="1.0" encoding="UTF-8"?>
misc.xml
<?xml version="1.0" encoding="UTF-8"?>
modules.xml
<?xml version="1.0" encoding="UTF-8"?>
聲明:本網頁內容旨在傳播知識,若有侵權等問題請及時與本網聯系,我們將在第一時間刪除處理。TEL:177 7030 7066 E-MAIL:11247931@qq.com