#!/usr/bin/env python3
"""
事件分析模块

功能：
1. 事件影响力评估
2. 事件主题分类
3. 事件情感分析
4. 事件传播模式识别
"""

import json
from typing import List, Dict, Any
from collections import defaultdict
import re


def analyze_event_influence(event: Dict[str, Any], interaction_data: Dict[str, Any] = None) -> Dict[str, Any]:
    """
    分析事件影响力

    Args:
        event: 事件数据
        interaction_data: 交互数据（点赞、粉丝列表等）

    Returns:
        Dict: 影响力分析结果
    """
    tweets = event.get("tweets", [])

    # 1. 基础互动量
    total_engagement = event.get("total_engagement", {"total": 0})
    base_score = total_engagement.get("total", 0)

    # 2. 推文数量权重
    tweet_count = event.get("tweet_count", 0)
    tweet_weight = min(tweet_count / 20.0, 1.0) * 100  # 最多20条推文得满分

    # 3. 时间跨度权重（短时间内大量推文 = 热点）
    duration = event.get("duration_hours", 0)
    if duration > 0:
        density = tweet_count / duration  # 推文密度
        density_weight = min(density * 2.0, 1.0) * 50  # 密度越高得分越高
    else:
        density_weight = 0

    # 4. 原创推文比例（原创越多 = 主动性越高）
    original_count = event.get("original_count", 0)
    if tweet_count > 0:
        original_ratio = original_count / tweet_count
        original_weight = original_ratio * 50
    else:
        original_weight = 0

    # 5. 综合影响力评分
    influence_score = (
        base_score * 0.5 +  # 互动量占50%
        tweet_weight * 0.2 +  # 推文数量占20%
        density_weight * 0.2 +  # 推文密度占20%
        original_weight * 0.1  # 原创比例占10%
    ) / 100.0  # 归一化

    # 6. 影响力等级
    if influence_score >= 1000:
        level = "极高"
    elif influence_score >= 500:
        level = "高"
    elif influence_score >= 100:
        level = "中"
    else:
        level = "低"

    return {
        "score": round(influence_score, 2),
        "level": level,
        "factors": {
            "engagement_total": total_engagement.get("total", 0),
            "tweet_count": tweet_count,
            "original_count": original_count,
            "duration_hours": round(duration, 2),
            "density": round(density, 2) if duration > 0 else 0,
        },
        "breakdown": {
            "engagement_contribution": base_score * 0.5,
            "tweet_count_contribution": tweet_weight * 0.2,
            "density_contribution": density_weight * 0.2,
            "original_ratio_contribution": original_weight * 0.1,
        },
    }


def classify_event_topic(event: Dict[str, Any]) -> Dict[str, Any]:
    """
    事件主题分类

    Args:
        event: 事件数据

    Returns:
        Dict: 主题分类结果
    """
    tweets = event.get("tweets", [])
    all_text = " ".join([t.get("text", "") for t in tweets]).lower()

    # 关键词库（简化版，实际应用中应使用NLP模型）
    topic_keywords = {
        "产品发布": ["launch", "release", "product", "announce", "unveil", "debut"],
        "技术更新": ["update", "upgrade", "feature", "improve", "enhance", "fix"],
        "商业动态": ["deal", "acquisition", "merge", "investment", "stock", "revenue"],
        "政治/政策": ["policy", "regulation", "government", "law", "congress", "senate"],
        "社会议题": ["social", "community", "humanitarian", "charity", "support"],
        "个人观点": ["think", "believe", "opinion", "perspective", "view"],
        "争议/冲突": ["dispute", "conflict", "argument", "criticize", "controversy"],
        "娱乐/文化": ["entertainment", "culture", "art", "music", "movie", "game"],
    }

    # 统计关键词匹配
    topic_scores = defaultdict(int)
    for topic, keywords in topic_keywords.items():
        for keyword in keywords:
            # 统计关键词出现次数
            count = all_text.count(keyword)
            topic_scores[topic] += count

    # 识别主题标签
    hashtags = [h["tag"] for h in event.get("hashtags", [])]

    # 话题标签启发式
    hashtag_topics = {
        "product": "产品发布",
        "launch": "产品发布",
        "update": "技术更新",
        "tech": "技术更新",
        "deal": "商业动态",
        "business": "商业动态",
        "politics": "政治/政策",
        "policy": "政治/政策",
        "social": "社会议题",
        "opinion": "个人观点",
    }

    for hashtag in hashtags:
        if hashtag.lower() in hashtag_topics:
            topic_scores[hashtag_topics[hashtag.lower()]] += 3  # 话题标签权重更高

    # 排序主题
    sorted_topics = sorted(topic_scores.items(), key=lambda x: x[1], reverse=True)

    if sorted_topics and sorted_topics[0][1] > 0:
        primary_topic = sorted_topics[0][0]
        primary_confidence = min(sorted_topics[0][1] / 5.0, 1.0)  # 归一化

        secondary_topics = [t[0] for t in sorted_topics[1:4] if t[1] > 0]
    else:
        primary_topic = "未分类"
        primary_confidence = 0.0
        secondary_topics = []

    return {
        "primary_topic": primary_topic,
        "primary_confidence": round(primary_confidence, 2),
        "secondary_topics": secondary_topics,
        "all_topic_scores": dict(sorted_topics),
        "detected_hashtags": hashtags,
    }


def analyze_event_sentiment(event: Dict[str, Any]) -> Dict[str, Any]:
    """
    事件情感分析（简化版）

    实际应用中应使用BERT等NLP模型
    """
    tweets = event.get("tweets", [])
    all_text = " ".join([t.get("text", "") for t in tweets]).lower()

    # 简化情感词典
    positive_words = [
        "good", "great", "awesome", "excellent", "amazing", "love", "excited",
        "happy", "wonderful", "fantastic", "best", "success", "celebrate"
    ]
    negative_words = [
        "bad", "terrible", "awful", "hate", "angry", "frustrated", "disappointed",
        "sad", "worst", "fail", "concern", "worry", "problem", "issue", "crisis"
    ]

    positive_count = sum(all_text.count(word) for word in positive_words)
    negative_count = sum(all_text.count(word) for word in negative_words)

    total = positive_count + negative_count

    if total == 0:
        sentiment = "中性"
        confidence = 0.0
        score = 0.0
    else:
        score = (positive_count - negative_count) / total

        if score > 0.3:
            sentiment = "正面"
        elif score < -0.3:
            sentiment = "负面"
        else:
            sentiment = "中性"

        confidence = min(total / 10.0, 1.0)  # 关键词越多，置信度越高

    return {
        "sentiment": sentiment,
        "confidence": round(confidence, 2),
        "score": round(score, 2),
        "positive_word_count": positive_count,
        "negative_word_count": negative_count,
    }


def identify_propagation_pattern(event: Dict[str, Any]) -> Dict[str, Any]:
    """
    识别传播模式
    """
    tweets = event.get("tweets", [])
    tweet_count = event.get("tweet_count", 0)
    original_count = event.get("original_count", 0)
    retweet_count = event.get("retweet_count", 0)

    # 分析原创/转发比例
    if tweet_count > 0:
        original_ratio = original_count / tweet_count
    else:
        original_ratio = 0

    # 判定传播模式
    if original_ratio >= 0.8:
        pattern = "主动发声"
        description = "以原创内容为主，主动发起话题和讨论"
    elif original_ratio >= 0.5:
        pattern = "混合传播"
        description = "原创与转发并重，既发声也参与讨论"
    elif original_ratio >= 0.2:
        pattern = "参与讨论"
        description = "以转发和引用为主，参与现有话题讨论"
    else:
        pattern = "被动传播"
        description = "几乎全是转发，主要跟随他人话题"

    return {
        "pattern": pattern,
        "description": description,
        "original_ratio": round(original_ratio, 2),
        "original_count": original_count,
        "retweet_count": retweet_count,
    }


def analyze_event_comprehensive(event: Dict[str, Any], interaction_data: Dict[str, Any] = None) -> Dict[str, Any]:
    """
    事件综合分析（主入口）

    Args:
        event: 事件数据
        interaction_data: 交互数据

    Returns:
        Dict: 综合分析结果
    """
    # 1. 影响力分析
    influence = analyze_event_influence(event, interaction_data)

    # 2. 主题分类
    topic = classify_event_topic(event)

    # 3. 情感分析
    sentiment = analyze_event_sentiment(event)

    # 4. 传播模式
    propagation = identify_propagation_pattern(event)

    # 5. 综合评估
    analysis = {
        "event_id": event.get("id"),
        "influence": influence,
        "topic": topic,
        "sentiment": sentiment,
        "propagation_pattern": propagation,
        "summary": f"{topic['primary_topic']}事件，影响力{influence['level']}，情感倾向{sentiment['sentiment']}",
    }

    return analysis


def main():
    """命令行入口，用于测试"""
    import sys

    # 示例事件数据
    example_event = {
        "id": "event_0",
        "tweets": [
            {
                "id": "1",
                "text": "Exciting product launch coming soon! #product #launch",
                "like_count": 1000,
                "retweet_count": 500,
                "reply_count": 100,
            },
            {
                "id": "2",
                "text": "The launch is happening tomorrow #launch",
                "like_count": 2000,
                "retweet_count": 800,
                "reply_count": 200,
            },
        ],
        "total_engagement": {
            "likes": 3000,
            "retweets": 1300,
            "replies": 300,
            "total": 4600,
        },
        "tweet_count": 2,
        "original_count": 2,
        "retweet_count": 0,
        "duration_hours": 24.0,
        "hashtags": [
            {"tag": "product", "count": 1},
            {"tag": "launch", "count": 2},
        ],
    }

    if len(sys.argv) > 1:
        # 从文件读取
        with open(sys.argv[1], 'r', encoding='utf-8') as f:
            example_event = json.load(f)

    result = analyze_event_comprehensive(example_event)
    print(json.dumps(result, ensure_ascii=False, indent=2))


if __name__ == "__main__":
    main()
