You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Gensim构建Doc2Vec模型并查找同商品高相似高有用性评论

解决Doc2Vec相似评论查找及TaggedReviewDocument属性缺失问题

一、修复TaggedReviewDocument类的属性为空问题

你定义的TaggedReviewDocument需要在生成TaggedDocument的同时,主动保留product(asin)、helpfulness、reviewerID这些属性,否则实例化后自然为空。可以通过继承gensim的TaggedDocument来扩展字段:

from gensim.models.doc2vec import TaggedDocument

class TaggedReviewDocument(TaggedDocument):
    def __init__(self, words, tags, product, reviewer_id, helpfulness):
        super().__init__(words, tags)
        self.product = product  # 存储商品asin
        self.reviewer_id = reviewer_id
        self.helpfulness = helpfulness  # 存储有用性得分,比如(2,3)元组或计算后的数值

生成实例时要传入对应参数,同时给每条评论设置唯一标签(方便后续模型调用):

# 假设从txt文件读取的单条评论格式为:reviewText\treviewerID\thelpfulness
tokenized_text = review_text.split()  # 这里用训练模型时的分词逻辑,比如自定义分词器
# 标签用"asin_reviewerID"的组合,保证唯一
doc = TaggedReviewDocument(
    words=tokenized_text,
    tags=(f"{asin}_{reviewer_id}",),
    product=asin,
    reviewer_id=reviewer_id,
    helpfulness=eval(helpful_str)  # 将字符串格式的有用性转成元组
)

二、实现find_similar_reviews函数

核心步骤

  1. 读取目标asin对应的评论文件,解析所有评论的文本、评论者ID、有用性得分
  2. 定位目标评论,用Doc2Vec模型推断其向量
  3. 遍历同商品评论,计算相似度并筛选符合条件的结果

完整代码实现

import gensim
from collections import namedtuple

# 定义结构存储评论信息,方便后续处理
Review = namedtuple('Review', ['review_text', 'reviewer_id', 'helpfulness', 'tokenized_text'])

def find_similar_reviews(asin, target_reviewer_id, model, product_dir='product'):
    # 1. 读取目标商品的所有评论
    review_file = f"{product_dir}/{asin}.txt"
    all_reviews = []
    target_tokenized = None
    
    with open(review_file, 'r', encoding='utf-8') as f:
        for line in f:
            # 根据你的存储格式调整分隔符,这里假设是制表符分隔
            review_text, reviewer_id, helpful_str = line.strip().split('\t')
            # 分词逻辑必须和训练模型时完全一致
            tokenized = review_text.split()
            # 计算有用性得分:用有用投票数/总投票数,避免除零
            helpful_tuple = eval(helpful_str)
            helpful_score = helpful_tuple[0] / helpful_tuple[1] if helpful_tuple[1] != 0 else 0
            
            all_reviews.append(Review(review_text, reviewer_id, helpful_score, tokenized))
            # 记录目标评论的分词结果
            if reviewer_id == target_reviewer_id:
                target_tokenized = tokenized
    
    if not target_tokenized:
        return []  # 未找到目标评论
    
    # 2. 推断目标评论的向量,epochs和训练模型时保持一致
    target_vec = model.infer_vector(target_tokenized, epochs=model.epochs)
    
    # 3. 计算相似度并筛选符合条件的评论
    similar_candidates = []
    for review in all_reviews:
        if review.reviewer_id == target_reviewer_id:
            continue  # 跳过自身评论
        # 推断当前评论的向量
        curr_vec = model.infer_vector(review.tokenized_text, epochs=model.epochs)
        # 计算余弦相似度
        similarity = gensim.matutils.cosine_similarity(target_vec, curr_vec)[0][0]
        if similarity >= 0.8:
            similar_candidates.append((similarity, review.helpfulness, review))
    
    # 4. 按有用性得分降序、相似度降序排序,取前5条
    similar_candidates.sort(key=lambda x: (-x[1], -x[0]))
    top5_results = [(rev.review_text, rev.reviewer_id, rev.helpfulness, sim) for sim, _, rev in similar_candidates[:5]]
    
    return top5_results

优化技巧:利用模型的docvecs提升效率

如果训练时给每条评论设置了唯一标签(如asin_reviewerID),可以直接调用model.docvecs.most_similar获取相似向量,减少重复推断的开销:

def find_similar_reviews(asin, target_reviewer_id, model, product_dir='product'):
    target_tag = f"{asin}_{target_reviewer_id}"
    if target_tag not in model.docvecs:
        return []
    
    # 直接从模型获取相似文档
    similar_docs = model.docvecs.most_similar(target_tag, topn=100)
    # 读取目标商品的所有评论
    review_file = f"{product_dir}/{asin}.txt"
    all_reviews = {}
    with open(review_file, 'r', encoding='utf-8') as f:
        for line in f:
            review_text, reviewer_id, helpful_str = line.strip().split('\t')
            helpful_tuple = eval(helpful_str)
            helpful_score = helpful_tuple[0] / helpful_tuple[1] if helpful_tuple[1] != 0 else 0
            all_reviews[reviewer_id] = (review_text, helpful_score)
    
    # 筛选同商品、相似度≥0.8的评论
    similar_candidates = []
    for tag, similarity in similar_docs:
        doc_asin, doc_reviewer_id = tag.split('_')
        if doc_asin == asin and similarity >= 0.8 and doc_reviewer_id != target_reviewer_id:
            review_text, helpful_score = all_reviews[doc_reviewer_id]
            similar_candidates.append((similarity, helpful_score, review_text, doc_reviewer_id))
    
    # 排序取前5
    similar_candidates.sort(key=lambda x: (-x[1], -x[0]))
    return [(text, rid, score, sim) for sim, score, text, rid in similar_candidates[:5]]

内容的提问来源于stack exchange,提问作者Alex

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.30 04:43:23