利用gensim 直接生成文档向量
def gen_d2v_corpus(self, lines): with open("./data/ques2_result.txt", "wb") as fw: for line in lines: fw.write(" ".join(jieba.lcut(line)) + "\n") sents = doc2vec.TaggedLineDocument("./data/ques2_result.txt") model = doc2vec.Doc2Vec(sents, size = 50, window = 5, alpha = 0.015) model.train(sents) corpus = model.docvecs np.save("./output/d2v.corpus.npy", corpus) return np.asarray(corpus)
时间: 2024-10-27 06:14:59