#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
公众情绪分析模块 - 人物画像技能
分析公众讨论的情感倾向和主要话题

创建时间：2026-04-03
开发者：瞰宇 (Kàn Yǔ)
"""

import json
from typing import Dict, List, Tuple, Optional
from dataclasses import dataclass
from collections import Counter, defaultdict
import re


@dataclass
class EmotionDistribution:
    """情感分布"""
    positive: float  # 正面比例
    neutral: float  # 中立比例
    negative: float  # 负面比例
    strong_emotion_ratio: float  # 强烈情绪比例


@dataclass
class TopicAnalysis:
    """话题分析"""
    topic: str
    mention_frequency: float  # 提及频率
    sentiment: str  # 情感倾向（正面/负面/中立）
    intensity: str  # 情感强度（弱/中/强）


@dataclass
class SupporterProfile:
    """支持者画像"""
    group_name: str
    demographics: Dict[str, str]  # 人口特征
    ideology: List[str]  # 意识形态特征
    emotion_characteristics: List[str]  # 情感特征


class PublicOpinionAnalyzer:
    """
    公众情绪分析器
    分析公众讨论的情感倾向和主要话题
    """
    
    # 正负面情感词库（示例）
    POSITIVE_WORDS = [
        "支持", "赞同", "优秀", "伟大", "喜欢", "爱", "好", "棒",
        "support", "great", "love", "good", "excellent", "awesome"
    ]
    
    NEGATIVE_WORDS = [
        "反对", "批评", "糟糕", "讨厌", "恨", "坏", "差", "愚蠢",
        "oppose", "criticize", "bad", "terrible", "hate", "stupid"
    ]
    
    STRONG_EMOTION_WORDS = [
        "强烈", "非常", "极其", "绝对", "坚决", "一定", "务必",
        "strongly", "very", "extremely", "absolutely", "definitely"
    ]
    
    def __init__(self):
        """初始化公众情绪分析器"""
        self.social_posts = []
        self.comments = []
    
    def load_social_data(self, posts: List[Dict], comments: List[Dict] = None):
        """
        加载社交媒体数据
        
        Args:
            posts: 帖子列表
            comments: 评论列表（可选）
        """
        self.social_posts = posts
        self.comments = comments or []
    
    def analyze_sentiment(self, text: str) -> Tuple[str, float, bool]:
        """
        分析文本情感
        
        Args:
            text: 文本内容
            
        Returns:
            (情感倾向, 强度分数, 是否强烈情绪)
        """
        text_lower = text.lower()
        
        # 计算正负面词频
        positive_count = sum(1 for word in self.POSITIVE_WORDS if word in text_lower)
        negative_count = sum(1 for word in self.NEGATIVE_WORDS if word in text_lower)
        
        # 计算强烈情绪词频
        strong_count = sum(1 for word in self.STRONG_EMOTION_WORDS if word in text_lower)
        is_strong = strong_count > 0
        
        # 计算得分
        total = positive_count + negative_count
        if total == 0:
            return "neutral", 0.0, is_strong
        
        score = (positive_count - negative_count) / total
        
        # 归一化强度到[0, 1]
        intensity = min(1.0, total / 10.0)
        
        # 判断情感倾向
        if score > 0.2:
            sentiment = "positive"
        elif score < -0.2:
            sentiment = "negative"
        else:
            sentiment = "neutral"
        
        return sentiment, intensity, is_strong
    
    def analyze_overall_sentiment_distribution(self) -> EmotionDistribution:
        """
        分析整体情感分布
        
        Returns:
            EmotionDistribution对象
        """
        all_texts = self.social_posts + self.comments
        
        if not all_texts:
            return EmotionDistribution(0.0, 0.0, 0.0, 0.0)
        
        sentiment_counts = Counter()
        strong_emotion_count = 0
        
        for item in all_texts:
            text = item.get("content", "")
            sentiment, intensity, is_strong = self.analyze_sentiment(text)
            
            sentiment_counts[sentiment] += 1
            if is_strong:
                strong_emotion_count += 1
        
        total = sum(sentiment_counts.values())
        if total == 0:
            return EmotionDistribution(0.0, 0.0, 0.0, 0.0)
        
        return EmotionDistribution(
            positive=sentiment_counts["positive"] / total,
            neutral=sentiment_counts["neutral"] / total,
            negative=sentiment_counts["negative"] / total,
            strong_emotion_ratio=strong_emotion_count / total
        )
    
    def extract_topics(self, min_mentions: int = 5) -> List[str]:
        """
        提取主要话题
        
        Args:
            min_mentions: 最小提及次数
            
        Returns:
            话题列表
        """
        all_texts = [p.get("content", "") for p in self.social_posts]
        combined_text = " ".join(all_texts)
        
        # 简单的关键词提取（实际应该使用NLP方法）
        words = re.findall(r'\b[\w\u4e00-\u9fff]{2,}\b', combined_text)
        
        word_counts = Counter(words)
        
        # 过滤停用词
        stop_words = {
            "的", "了", "在", "是", "和", "有", "说", "这", "那",
            "the", "and", "of", "to", "in", "is", "it", "that"
        }
        for stop_word in stop_words:
            word_counts.pop(stop_word, None)
        
        # 返回高频词作为话题
        return [word for word, count in word_counts.most_common(20) if count >= min_mentions]
    
    def analyze_topic_sentiment(self, topic: str) -> TopicAnalysis:
        """
        分析特定话题的情感
        
        Args:
            topic: 话题关键词
            
        Returns:
            TopicAnalysis对象
        """
        topic_posts = [
            p for p in self.social_posts
            if topic.lower() in p.get("content", "").lower()
        ]
        
        if not topic_posts:
            return TopicAnalysis(topic, 0.0, "neutral", "弱")
        
        # 计算提及频率
        mention_frequency = len(topic_posts) / len(self.social_posts) if self.social_posts else 0.0
        
        # 分析该话题的整体情感
        sentiments = []
        intensities = []
        for post in topic_posts:
            sentiment, intensity, _ = self.analyze_sentiment(post.get("content", ""))
            sentiments.append(sentiment)
            intensities.append(intensity)
        
        # 统计情感倾向
        sentiment_counts = Counter(sentiments)
        dominant_sentiment = sentiment_counts.most_common(1)[0][0] if sentiment_counts else "neutral"
        
        # 计算平均强度
        avg_intensity = sum(intensities) / len(intensities) if intensities else 0.0
        intensity_level = "强" if avg_intensity > 0.6 else ("中" if avg_intensity > 0.3 else "弱")
        
        return TopicAnalysis(
            topic=topic,
            mention_frequency=mention_frequency,
            sentiment=dominant_sentiment,
            intensity=intensity_level
        )
    
    def estimate_support_rate(self) -> Dict[str, float]:
        """
        估算公众支持率
        
        Returns:
            支持率字典
        """
        sentiment_dist = self.analyze_overall_sentiment_distribution()
        
        # 基于情感分布估算支持率（简化版）
        # 实际应该结合社交媒体粉丝数、互动量等数据
        support_rate = sentiment_dist.positive
        opposition_rate = sentiment_dist.negative
        neutral_rate = sentiment_dist.neutral
        
        return {
            "整体支持率": support_rate,
            "反对比例": opposition_rate,
            "中立比例": neutral_rate,
            "极化指数": abs(support_rate - opposition_rate)
        }
    
    def generate_public_opinion_report(self) -> Dict:
        """
        生成公众情绪分析报告
        
        Returns:
            分析报告字典
        """
        # 分析情感分布
        sentiment_dist = self.analyze_overall_sentiment_distribution()
        
        # 提取主要话题
        topics = self.extract_topics()
        
        # 分析每个话题的情感
        topic_analyses = [self.analyze_topic_sentiment(topic) for topic in topics[:10]]
        
        # 估算支持率
        support_rates = self.estimate_support_rate()
        
        report = {
            "情感分布": {
                "正面": f"{sentiment_dist.positive:.1%}",
                "中立": f"{sentiment_dist.neutral:.1%}",
                "负面": f"{sentiment_dist.negative:.1%}",
                "强烈情绪占比": f"{sentiment_dist.strong_emotion_ratio:.1%}"
            },
            "主要话题": [
                {
                    "话题": ta.topic,
                    "提及频率": f"{ta.mention_frequency:.1%}",
                    "情感倾向": ta.sentiment,
                    "情感强度": ta.intensity
                }
                for ta in topic_analyses
            ],
            "支持率估算": {
                "整体支持率": f"{support_rates['整体支持率']:.1%}",
                "反对比例": f"{support_rates['反对比例']:.1%}",
                "中立比例": f"{support_rates['中立比例']:.1%}",
                "极化指数": f"{support_rates['极化指数']:.2f}"
            }
        }
        
        return report


def main():
    """测试公众情绪分析器"""
    analyzer = PublicOpinionAnalyzer()
    
    # 示例社交媒体数据
    posts = [
        {
            "content": "我强烈支持他的政策，他会让国家变得更伟大！",
            "date": "2024-01-01"
        },
        {
            "content": "他的政策简直是一场灾难，完全错误。",
            "date": "2024-01-02"
        },
        {
            "content": "我不确定他的政策效果如何，需要再观察。",
            "date": "2024-01-03"
        },
        {
            "content": "非常好！支持！",
            "date": "2024-01-04"
        },
        {
            "content": "太糟糕了，反对到底！",
            "date": "2024-01-05"
        }
    ]
    
    analyzer.load_social_data(posts)
    
    # 生成报告
    print("=== 公众情绪分析报告 ===")
    report = analyzer.generate_public_opinion_report()
    print(json.dumps(report, ensure_ascii=False, indent=2))


if __name__ == "__main__":
    main()
