清洗文本高频词、情感分析、情感分类、主题建模挖掘主题

import pandas as pd

import re

import nltk

from nltk import FreqDist

from nltk.sentiment.vader import SentimentIntensityAnalyzer

from nltk.tokenize import word_tokenize

import spacy

from spacy.lang.en.stop_words import STOP_WORDS

from gensim.corpora import Dictionary

from gensim.models import LdaModel

下载NLTK的停用词、情感分析和词性标注所需的资源

nltk.download('stopwords')

nltk.download('punkt')

nltk.download('vader_lexicon')

加载SpaCy的英文NLP模型

nlp = spacy.load("en_core_web_sm")

读取Excel文件

df = pd.read_excel('nltk分词处理结果第二次.xlsx')

定义文本清洗函数

def clean_text(text):

去除HTML标签

cleaned_text = re.sub(r'<.*?>', '', text)

去除多余空格和换行符

cleaned_text = re.sub(r'\s+', ' ', cleaned_text)

转换为小写

cleaned_text = cleaned_text.lower()

return cleaned_text

清洗文本数据

df'cleaned_content' = df'content'.apply(clean_text)

词频分析

words = \[\]

for text in df'cleaned_content':

words += word_tokenize(text)

freq_dist = FreqDist(words)

print("词频分析结果:", freq_dist.most_common(10))

情感分析

sia = SentimentIntensityAnalyzer()

df'sentiment_score' = df'cleaned_content'.apply(lambda x: sia.polarity_scores(x)'compound')

print("情感分析结果:", df'sentiment_score')

定义阈值

positive_threshold = 0.5

negative_threshold = -0.5

根据情感分数进行分类

def classify_sentiment(score):

if score > positive_threshold:

return '积极'

elif score < negative_threshold:

return '消极'

else:

return '中性'

应用分类函数,创建新的列 'sentiment_category'

df'sentiment_category' = df'sentiment_score'.apply(classify_sentiment)

输出带有情感分类的数据

print(df\['cleaned_content', 'sentiment_score', 'sentiment_category'])

主题建模

tokens = \[token.text.lower() for token in nlp(text) if token.is_alpha and token.text.lower() not in STOP_WORDS for text in df'cleaned_content']

dictionary = Dictionary(tokens)

corpus = dictionary.doc2bow(text) for text in tokens

lda_model = LdaModel(corpus, num_topics=5, id2word=dictionary, passes=15)

topics = lda_model.print_topics(num_words=5)

print("主题建模结果:")

for topic in topics:

print(topic)

相关推荐
u1301304 小时前
AI 日报(2026年10月4日)
人工智能
飞塔老梅子4 小时前
17. Unsloth下载、安装及加载本地大模型 ❀ 老梅子学AI
人工智能·glm·lm studio·5.3·unsolth
打工仔折腾 AI4 小时前
把AI Agent托管在家用电脑:UU远程终端与端口映射实测记录
人工智能·后端·python·langchain·ai agent 实战
零基础1234 小时前
Agent 的 Memory 怎么做科研:以中医诊断场景为例
人工智能·经验分享·python·语言模型
归秋1424 小时前
人声节奏对齐软件推荐:从专业DAW到AI人声制作工具怎么选
人工智能
2601_962966645 小时前
数学与应用数学专业想进管理咨询,2027届秋招需要补哪些商业知识和技能?
人工智能
ss2736 小时前
AI全栈实战 | 3.2-01 Python 基础:四大数据容器怎么选,推导式为什么是 Pythonic 的灵魂
开发语言·人工智能·python
Sweet锦6 小时前
不调 Python,不装向量库:我用纯 Java 写了一套以图搜图引擎
java·人工智能·开源·图搜索
fpcc6 小时前
AI和大模型—JEV模型
人工智能
weixin_446260856 小时前
面向演进式企业AI智能体技能的持续流程级评估
人工智能