import pandas as pd

stopwords_file = 'stopwords.txt'           # 停用词文件
comments_file = 'ratings100.csv'           # 评论语料文件
filename = 'negative_dict.txt'             # 负面情感词典文件

#读取语料词典并进行分词的函数
def fenci_a(stopwords_file, comments_file):
    # 1、读取停用词表
    with open(stopwords_file, 'r', encoding='utf-8') as f:
        stop_words = {line.strip() for line in f}

    # 2、读取语料文件为一个长字符串
    f = open(comments_file, "r", encoding="utf-8")
    comments = f.read()

    # 3、使用jieba进行中文分词,去除停用词,计算词频,结果保存为一个字典类型
    words = {}
    import jieba
    wordslist = list(jieba.cut(comments, cut_all=True))  # 分词结果转换为列表
    for w in wordslist:
        if w not in stop_words:  # 不属于停用词
            if w not in words.keys():
                words[w] = 1  # 第一次添加到字典里,词频为1
            else:
                v = int(words.get(w))  # 读出原有的词频
                words[w] = v + 1  # 词频+1

    print("分词后词语数量:", len(words.keys()))
    print(words)
    return words  # 可选:返回词频字典,便于后续使用


#读取情感词典并初步转化为集合的函数
def duqu_cidian(filename):
    with open(filename, 'r', encoding='utf-8') as f:
        cidian_jihe = set(line.strip().split("\t1")[0] for line in f)
    return cidian_jihe

# 调用示例(如果你需要使用这个函数):
# negwords = duqu_cidian('negative_dict.txt')
# print(negwords)



yuliaoku =  fenci_a(stopwords_file, comments_file)
cidian = duqu_cidian(filename)


#在分词好的语料库中通过读取的情感词典集合筛选出情感词并统计词频
def shaixuan_qingganci(yuliaoku, cidian):

    wordsinter = set(yuliaoku.keys()).intersection(cidian)
    qingganci_cipin = {word: yuliaoku[word] for word in wordsinter}
    print("\n语料中的负面情感词数量:", len(qingganci_cipin))
    print("负面情感词及词频:", qingganci_cipin)
    return qingganci_cipin

pos_freq = shaixuan_qingganci(yuliaoku, cidian)
neg_freq = shaixuan_qingganci(yuliaoku, cidian)



df_pos = pd.DataFrame(list(pos_freq.items()), columns=['词语', '词频'])
df_pos['情感极性'] = '正面'

df_neg = pd.DataFrame(list(neg_freq.items()), columns=['词语', '词频'])
df_neg['情感极性'] = '负面'

# 合并成一个 DataFrame
df_result = pd.concat([df_pos, df_neg], ignore_index=True)
print(df_result)

import matplotlib.pyplot as plt

# 设置中文字体,避免图表中文显示为方框
plt.rcParams['font.family'] = ['SimHei']  # 常见中文字体
plt.rcParams['axes.unicode_minus'] = False  # 正常显示负号

# 从 df_result 中分离正面和负面数据
df_pos = df_result[df_result['情感极性'] == '正面'].head(20)  # 只显示词频最高的前20个,避免图表太拥挤
df_neg = df_result[df_result['情感极性'] == '负面'].head(20)

# 如果数据为空,给出提示
if df_pos.empty and df_neg.empty:
    print("没有找到正面或负面情感词,无法绘图。")
elif df_pos.empty:
    print("没有找到正面情感词。")
elif df_neg.empty:
    print("没有找到负面情感词。")
else:
    # === 绘制正面情感词柱形图 ===
    plt.figure(figsize=(10, 6))
    plt.bar(df_pos['词语'], df_pos['词频'], color='skyblue', edgecolor='navy', alpha=0.8)
    plt.title('语料库中正面情感词词频统计', fontsize=16, fontweight='bold')
    plt.xlabel('正面情感词', fontsize=12)
    plt.ylabel('词频', fontsize=12)
    plt.xticks(rotation=45, fontsize=10)  # 旋转x轴标签,避免重叠
    plt.grid(axis='y', linestyle='--', alpha=0.7)
    plt.tight_layout()  # 自动调整布局
    plt.show()

    # === 绘制负面情感词柱形图 ===
    plt.figure(figsize=(10, 6))
    plt.bar(df_neg['词语'], df_neg['词频'], color='lightcoral', edgecolor='darkred', alpha=0.8)
    plt.title('语料库中负面情感词词频统计', fontsize=16, fontweight='bold')
    plt.xlabel('负面情感词', fontsize=12)
    plt.ylabel('词频', fontsize=12)
    plt.xticks(rotation=45, fontsize=10)
    plt.grid(axis='y', linestyle='--', alpha=0.7)
    plt.tight_layout()
    plt.show()





import re

def load_sentiment_dict(filepath):
    data = []
    with open(filepath, 'r', encoding='utf-8') as f:
        for line_num, line in enumerate(f, start=1):
            line = line.strip()
            if not line:
                continue

            # 尝试从行末尾提取一个数字(整数或小数)
            # 匹配模式:可选负号 + 数字 + 可选小数点和小数
            match = re.search(r'(-?\d+\.?\d*)$', line)
            if not match:
                print(f"跳过第 {line_num} 行(未找到数值): {line}")
                continue

            polarity_str = match.group(1)
            try:
                polarity = float(polarity_str)
            except ValueError:
                print(f"跳过第 {line_num} 行(数值转换失败): {line}")
                continue

            # 数值前面的部分作为词语(去除末尾空格)
            word_part = line[:match.start()].strip()
            # 如果整行就是数字,或前面为空,跳过
            if not word_part:
                print(f"跳过第 {line_num} 行(无词语部分): {line}")
                continue

            data.append([word_part, polarity])

    return pd.DataFrame(data, columns=['词语名称', '极值大小'])

# 读取正向词典
pos_df = load_sentiment_dict('positive_dict.txt')
print(" 正向情感词典加载完成,共 {} 条".format(len(pos_df)))
print(pos_df)

# 读取负向词典
neg_df = load_sentiment_dict('negative_dict.txt')
print(" 负向情感词典加载完成,共 {} 条".format(len(neg_df)))
print(neg_df)



# 初始化得分
positive_score = 0.0
negative_score = 0.0

# 可选:记录每个匹配词的加权值(便于后续分析)
matched_positive = []
matched_negative = []

# 遍历 df_result 的每一行
for _, row in df_result.iterrows():
    word = row['词语']
    freq = row['词频']

    # 在 pos_df 中查找该词语
    pos_match = pos_df[pos_df['词语名称'] == word]
    if not pos_match.empty:
        polarity = pos_match.iloc[0]['极值大小']
        weighted_value = freq * polarity
        positive_score += weighted_value
        # 打印匹配信息
        print(f" 正面词: {word}  词频×极值大小 = {freq} × {polarity} = {weighted_value:.4f}")
        matched_positive.append((word, freq, polarity, weighted_value))

    # 在 neg_df 中查找该词语
    neg_match = neg_df[neg_df['词语名称'] == word]
    if not neg_match.empty:
        polarity = neg_match.iloc[0]['极值大小']
        weighted_value = freq * polarity
        negative_score += weighted_value
        # 打印匹配信息
        print(f" 负面词: {word}  词频×极值大小 = {freq} × {polarity} = {weighted_value:.4f}")
        matched_negative.append((word, freq, polarity, weighted_value))

# 最终输出总分

print(f"正面情感词极值总和: {positive_score:.4f}")
print(f"负面情感词极值总和: {negative_score:.4f}")



'''
你希望实现一个 **情感加权词频计算**,具体需求如下:

---

### 目标

1. 遍历 `df_result` 的每一行(每条词语记录);
2. 如果该行的 `'词语'` 在 `pos_df['词语名称']` 中出现,则:
   - 将其 `词频` × 对应的 `极值大小`(正向情感强度)
   - 累加到 **正向总得分**
3. 如果该行的 `'词语'` 在 `neg_df['词语名称']` 中出现,则:
   - 将其 `词频` × 对应的 `极值大小`(负向情感强度)
   - 累加到 **负向总得分**
4. 最终返回:
   - 正向加权总和(`positive_score`)
   - 负向加权总和(`negative_score`)

---

### 前提假设

- `df_result` 是一个 DataFrame,包含列:
  - `'词语'`:词语
  - `'词频'`:该词语出现的频率(数值)
- `pos_df` 和 `neg_df` 是你已加载的情感词典,包含:
  - `'词语名称'`
  - `'极值大小'`

---

### 完整代码实现

```python
# 初始化得分
positive_score = 0.0
negative_score = 0.0

# 将 pos_df 和 neg_df 转为字典,加快查找速度
pos_dict = dict(zip(pos_df['词语名称'], pos_df['极值大小']))
neg_dict = dict(zip(neg_df['词语名称'], neg_df['极值大小']))

# 遍历 df_result 每一行
for _, row in df_result.iterrows():
    word = row['词语']
    freq = row['词频']

    # 检查是否在正向词典中
    if word in pos_dict:
        positive_score += freq * pos_dict[word]

    # 检查是否在负向词典中
    if word in neg_dict:
        negative_score += freq * neg_dict[word]

# 输出结果
print(f"正向情感加权总得分: {positive_score:.4f}")
print(f"负向情感加权总得分: {negative_score:.4f}")
```

---

### 示例说明

假设:

**df_result:**
| 词语     | 词频 |
|----------|------|
| 很快     | 3    |
| 服务到位 | 2    |
| 差       | 1    |
| 难吃     | 4    |

**pos_df:**
| 词语名称 | 极值大小 |
|----------|----------|
| 很快     | 1.75     |
| 服务到位 | 1.0      |

**neg_df:**
| 词语名称 | 极值大小 |
|----------|----------|
| 差       | 1.5      |
| 难吃     | 2.0      |

**计算:**

- 正向:`3×1.75 + 2×1.0 = 5.25 + 2.0 = 7.25`
- 负向:`1×1.5 + 4×2.0 = 1.5 + 8.0 = 9.5`

输出:
```
正向情感加权总得分: 7.2500
负向情感加权总得分: 9.5000
```

---

### 优化建议(可选)

如果你的 `df_result` 很大,可以用 `merge` 向量化加速:

```python
# 正向匹配
pos_merge = df_result[df_result['词语'].isin(pos_dict.keys())]
pos_merge = pos_merge.copy()
pos_merge['加权值'] = pos_merge['词语'].map(pos_dict) * pos_merge['词频']
positive_score = pos_merge['加权值'].sum()

# 负向匹配
neg_merge = df_result[df_result['词语'].isin(neg_dict.keys())]
neg_merge = neg_merge.copy()
neg_merge['加权值'] = neg_merge['词语'].map(neg_dict) * neg_merge['词频']
negative_score = neg_merge['加权值'].sum()
```

但 **循环方式更清晰易懂**,适合中小数据量。

---

### 总结

你现在已经可以:

-  高效匹配情感词
-  计算 **词频 × 情感强度** 的加权和
-  分别得到正向、负向情感总强度

这个结果可用于:
- 情感倾向分析
- 情感评分模型
- 用户评论打分等场景
'''

import pandas as pd
#1.读语料文件
yuliaos = pd.read_csv('ratings100.csv')
print(yuliaos.shape, yuliaos.columns)  # 打印数据规模

#2.对语料文件按行进行中文分词
#2.1 读取停用词表
import jieba
with open('stopwords.txt', 'r', encoding='utf-8') as file:
    stopwords = file.read().splitlines()
stopwords = set(stopwords)  # 转换为集合

#2.2 按行进行中文分词
# 原始错误代码:yuliaos['segwords'] = []
# 修复后:
yuliaos['segwords'] = [[] for _ in range(len(yuliaos))]  # 每行对应一个空列表

comments = yuliaos['comment']
yuliao_list = []
for txt in comments:
    segwords = []
    try:
        seqlist = jieba.cut(txt, cut_all=False)
        for seg in seqlist:
            if seg not in stopwords:
                segwords.append(seg)
    except Exception as e:
        print(f"处理评论出错: {txt}, 错误: {e}")
        continue
    yuliao_list.append([txt, segwords])

yuliao_df = pd.DataFrame(yuliao_list, columns=['comment', 'segwords'])
print(yuliao_df.head(10))

# 读取情感极值词典
qinggan_jizhi_df = pd.read_csv(filepath_or_buffer="Sentiment/strength.txt", encoding="GBK", sep='\t', names=['word', 'strength'])
print(qinggan_jizhi_df.head(5))
qinggan_jizhi_dict = dict(zip(qinggan_jizhi_df['word'], qinggan_jizhi_df['strength']))

# 4.计算每一行的情感得分,记录到yuliao_df的score字段
yuliao_df['score'] = 0  # 初始化为0
i = 0

def jisuan_jizhi(a):  # 1个用法
    ss = 0
    for w in a:
        try:
            s = qinggan_jizhi_dict.get(w)
            if s is not None:
                ss += s  # 整行的极值为所有情感词极值之和
        except:
            continue
    return ss  # 返回该行的得分

for segwords in yuliao_df['segwords']:
    linescore = jisuan_jizhi(segwords)
    yuliao_df.iloc[i, 2] = linescore  # 写入 score 列(第2列)
    i += 1

print(yuliao_df.head(20))
print("全文情感得分为:", yuliao_df)

'''
comment 的含义
中文意思:评论
数据内容:原始的用户文本评论。
示例值:
"这个手机很好用"
"太差了,不推荐"
"一般般,没什么特别的"


segwords 的含义
中文意思:分词结果(“seg” 是 “segmentation” 的缩写,“words” 是词语)
数据内容:将 comment 中的一条评论通过中文分词工具(如 jieba)切分成的一个个词语组成的列表(list)。
示例:
comment	        segwords(分词后)
"这个手机很好用"	['这个', '手机', '很', '好用']
"太差了,不推荐"	['太', '差了', '不', '推荐']
"一般般,没什么特别的"	['一般般', '没', '什么', '特别']
'''

# 下面是修改变量之前的代码
#1.读语料文件
ratings = pd.read_csv('ratings100.csv')
print(ratings.shape, ratings.columns)  # 打印数据规模

#2.对语料文件按行进行中文分词
#2.1 读取停用词表
import jieba
with open('stopwords.txt', 'r', encoding='utf-8') as file:
    stopwords = file.read().splitlines()
stopwords = set(stopwords)  # 转换为集合

#2.2 按行进行中文分词
# 原始错误代码:ratings['segwords'] = []
# 修复后:
ratings['segwords'] = [[] for _ in range(len(ratings))]  # 每行对应一个空列表

comments = ratings['comment']
ratinglist = []
for txt in comments:
    segwords = []
    try:
        seqlist = jieba.cut(txt, cut_all=False)
        for seg in seqlist:
            if seg not in stopwords:
                segwords.append(seg)
    except Exception as e:
        print(f"处理评论出错: {txt}, 错误: {e}")
        continue
    ratinglist.append([txt, segwords])

ratingdf = pd.DataFrame(ratinglist, columns=['comment', 'segwords'])
print(ratingdf.head(10))

# 读取情感极值词典
strengthdf = pd.read_csv(filepath_or_buffer="Sentiment/strength.txt", encoding="GBK", sep='\t', names=['word', 'strength'])
print(strengthdf.head(5))
strengthdict = dict(zip(strengthdf['word'], strengthdf['strength']))

# 4.计算每一行的情感得分,记录到ratingdf的score字段
ratingdf['score'] = 0  # 初始化为0
i = 0

def calscore(swords):  # 1个用法
    ss = 0
    for w in swords:
        try:
            s = strengthdict.get(w)
            if s is not None:
                ss += s  # 整行的极值为所有情感词极值之和
        except:
            continue
    return ss  # 返回该行的得分

for segwords in ratingdf['segwords']:
    linescore = calscore(segwords)
    ratingdf.iloc[i, 2] = linescore  # 写入 score 列(第2列)
    i += 1

print(ratingdf.head(20))
print("全文情感得分为:", ratingdf)

'''
改动说明(仅名称替换):
原名称 -
新名称
类型
ratings -
yuliaos
DataFrame(语料数据)
ratinglist -
yuliao_list
列表(临时存储评论与分词)
ratingdf -
yuliao_df
DataFrame(处理后的语料)
strengthdf -
qinggan_jizhi_df
DataFrame(情感极值词典)
strengthdict -
qinggan_jizhi_dict
字典(词-强度映射)
calscore(swords)-
jisuan_jizhi(a)
函数(计算情感强度)

'''

更多推荐