基于Gensim构建Doc2Vec模型并查找同商品高相似高有用性评论
解决Doc2Vec相似评论查找及TaggedReviewDocument属性缺失问题
一、修复TaggedReviewDocument类的属性为空问题
你定义的TaggedReviewDocument需要在生成TaggedDocument的同时,主动保留product(asin)、helpfulness、reviewerID这些属性,否则实例化后自然为空。可以通过继承gensim的TaggedDocument来扩展字段:
from gensim.models.doc2vec import TaggedDocument class TaggedReviewDocument(TaggedDocument): def __init__(self, words, tags, product, reviewer_id, helpfulness): super().__init__(words, tags) self.product = product # 存储商品asin self.reviewer_id = reviewer_id self.helpfulness = helpfulness # 存储有用性得分,比如(2,3)元组或计算后的数值
生成实例时要传入对应参数,同时给每条评论设置唯一标签(方便后续模型调用):
# 假设从txt文件读取的单条评论格式为:reviewText\treviewerID\thelpfulness tokenized_text = review_text.split() # 这里用训练模型时的分词逻辑,比如自定义分词器 # 标签用"asin_reviewerID"的组合,保证唯一 doc = TaggedReviewDocument( words=tokenized_text, tags=(f"{asin}_{reviewer_id}",), product=asin, reviewer_id=reviewer_id, helpfulness=eval(helpful_str) # 将字符串格式的有用性转成元组 )
二、实现find_similar_reviews函数
核心步骤
- 读取目标asin对应的评论文件,解析所有评论的文本、评论者ID、有用性得分
- 定位目标评论,用Doc2Vec模型推断其向量
- 遍历同商品评论,计算相似度并筛选符合条件的结果
完整代码实现
import gensim from collections import namedtuple # 定义结构存储评论信息,方便后续处理 Review = namedtuple('Review', ['review_text', 'reviewer_id', 'helpfulness', 'tokenized_text']) def find_similar_reviews(asin, target_reviewer_id, model, product_dir='product'): # 1. 读取目标商品的所有评论 review_file = f"{product_dir}/{asin}.txt" all_reviews = [] target_tokenized = None with open(review_file, 'r', encoding='utf-8') as f: for line in f: # 根据你的存储格式调整分隔符,这里假设是制表符分隔 review_text, reviewer_id, helpful_str = line.strip().split('\t') # 分词逻辑必须和训练模型时完全一致 tokenized = review_text.split() # 计算有用性得分:用有用投票数/总投票数,避免除零 helpful_tuple = eval(helpful_str) helpful_score = helpful_tuple[0] / helpful_tuple[1] if helpful_tuple[1] != 0 else 0 all_reviews.append(Review(review_text, reviewer_id, helpful_score, tokenized)) # 记录目标评论的分词结果 if reviewer_id == target_reviewer_id: target_tokenized = tokenized if not target_tokenized: return [] # 未找到目标评论 # 2. 推断目标评论的向量,epochs和训练模型时保持一致 target_vec = model.infer_vector(target_tokenized, epochs=model.epochs) # 3. 计算相似度并筛选符合条件的评论 similar_candidates = [] for review in all_reviews: if review.reviewer_id == target_reviewer_id: continue # 跳过自身评论 # 推断当前评论的向量 curr_vec = model.infer_vector(review.tokenized_text, epochs=model.epochs) # 计算余弦相似度 similarity = gensim.matutils.cosine_similarity(target_vec, curr_vec)[0][0] if similarity >= 0.8: similar_candidates.append((similarity, review.helpfulness, review)) # 4. 按有用性得分降序、相似度降序排序,取前5条 similar_candidates.sort(key=lambda x: (-x[1], -x[0])) top5_results = [(rev.review_text, rev.reviewer_id, rev.helpfulness, sim) for sim, _, rev in similar_candidates[:5]] return top5_results
优化技巧:利用模型的docvecs提升效率
如果训练时给每条评论设置了唯一标签(如asin_reviewerID),可以直接调用model.docvecs.most_similar获取相似向量,减少重复推断的开销:
def find_similar_reviews(asin, target_reviewer_id, model, product_dir='product'): target_tag = f"{asin}_{target_reviewer_id}" if target_tag not in model.docvecs: return [] # 直接从模型获取相似文档 similar_docs = model.docvecs.most_similar(target_tag, topn=100) # 读取目标商品的所有评论 review_file = f"{product_dir}/{asin}.txt" all_reviews = {} with open(review_file, 'r', encoding='utf-8') as f: for line in f: review_text, reviewer_id, helpful_str = line.strip().split('\t') helpful_tuple = eval(helpful_str) helpful_score = helpful_tuple[0] / helpful_tuple[1] if helpful_tuple[1] != 0 else 0 all_reviews[reviewer_id] = (review_text, helpful_score) # 筛选同商品、相似度≥0.8的评论 similar_candidates = [] for tag, similarity in similar_docs: doc_asin, doc_reviewer_id = tag.split('_') if doc_asin == asin and similarity >= 0.8 and doc_reviewer_id != target_reviewer_id: review_text, helpful_score = all_reviews[doc_reviewer_id] similar_candidates.append((similarity, helpful_score, review_text, doc_reviewer_id)) # 排序取前5 similar_candidates.sort(key=lambda x: (-x[1], -x[0])) return [(text, rid, score, sim) for sim, score, text, rid in similar_candidates[:5]]
内容的提问来源于stack exchange,提问作者Alex
相关产品推荐
相关产品推荐

