#!/usr/bin/env python3
"""
主题与叙事分析模块

功能：
1. 话题与立场分析（（关键词提取、主题聚类、情感分析）
2. 核心叙事识别（反复强调的核心主张）
"""

import json
import re
from typing import List, Dict, Any
from collections import defaultdict


def extract_keywords(text: str, top_n: int = 20) -> List[Dict[str, Any]]:
    """
    提取关键词（简化版词频统计）

    实际应用中应使用TF-IDF或Rake
    """
    # 移除特殊字符，分词
    words = re.findall(r'\b\w{3,}\b', text.lower())

    # 停用词（简化版）
    stop_words = {
        "the", "and", "for", "are", "but", "not", "you", "all", "can", "had",
        "her", "was", "one", "our", "out", "with", "have", "been", "his",
        "she", "they", "their", "this", "that", "from", "will", "would",
        "there", "about", "which", "when", "what", "where", "who", "whom",
        "just", "like", "get", "got", "going", "being", "been", "doing"
    }

    # 统计词频
    word_counts = defaultdict(int)
    for word in words:
        if word not in stop_words:
            word_counts[word] += 1

    # 返回Top N
    top_keywords = sorted(word_counts.items(), key=lambda x: x[1], reverse=True)[:top_n]

    total = sum(count for _, count in top_keywords)

    return [
        {"word": word, "count": count, "ratio": round(count / total, 4)}
        for word, count in top_keywords
    ]


def classify_topic(text: str) -> Dict[str, Any]:
    """
    主题分类

    Returns:
        Dict: 主题分类结果
    """
    text_lower = text.lower()

    # 主题关键词
    topic_keywords = {
        "政治": ["politic", "government", "policy", "law", "regulation", "congress", "senate", "election", "vote", "campaign"],
        "科技": ["tech", "ai", "technology", "software", "hardware", "internet", "digital", "cyber", "data", "algorithm"],
        "经济": ["economic", "economy", "market", "stock", "finance", "business", "trade", "invest", "money", "bank"],
        "社会": ["social", "society", "community", "culture", "people", "public", "citizen", "rights", "justice"],
        "环境": ["environment", "climate", "energy", "sustain", "green", "carbon", "pollution", "global warming"],
        "军事": ["military", "defense", "army", "navy", "air force", "weapon", "war", "conflict", "security"],
        "娱乐": ["entertain", "movie", "music", "sport", "game", "celebrity", "show", "film", "actor"],
        "健康": ["health", "medical", "virus", "disease", "vaccine", "treatment", "doctor", "hospital"],
    }

    topic_scores = defaultdict(int)
    for topic, keywords in topic_keywords.items():
        for keyword in keywords:
            topic_scores[topic] += text_lower.count(keyword)

    # 排序
    sorted_topics = sorted(topic_scores.items(), key=lambda x: x[1], reverse=True)

    if sorted_topics and sorted_topics[0][1] > 0:
        primary_topic = sorted_topics[0][0]
        primary_confidence = min(sorted_topics[0][1] / 5.0, 1.0)
        secondary_topics = [t[0] for t in sorted_topics[1:4] if t[1] > 0]
    else:
        primary_topic = "未分类"
        primary_confidence = 0.0
        secondary_topics = []

    return {
        "primary_topic": primary_topic,
        "primary_confidence": round(primary_confidence, 2),
        "secondary_topics": secondary_topics,
        "all_topic_scores": dict(sorted_topics)
    }


def analyze_sentiment(text: str) -> Dict[str, Any]:
    """
    情感分析（简化版）

    实际应用中应使用BERT等NLP模型
    """
    text_lower = text.lower()

    # 简化情感词典
    positive_words = [
        "good", "great", "awesome", "excellent", "amazing", "love", "excited",
        "happy", "wonderful", "fantastic", "best", "success", "celebrate",
        "support", "agree", "positive", "hope", "trust"
    ]
    negative_words = [
        "bad", "terrible", "awful", "hate", "angry", "frustrated", "disappointed",
        "sad", "worst", "fail", "concern", "worry", "problem", "issue", "crisis",
        "attack", "oppose", "negative", "fear", "distrust", "dangerous"
    ]

    positive_count = sum(text_lower.count(word) for word in positive_words)
    negative_count = sum(text_lower.count(word) for word in negative_words)

    total = positive_count + negative_count

    if total == 0:
        sentiment = "中性"
        confidence = 0.0
        score = 0.0
    else:
        score = (positive_count - negative_count) / total

        if score > 0.3:
            sentiment = "正面"
        elif score < -0.3:
            sentiment = "负面"
        else:
            sentiment = "中性"

        confidence = min(total / 10.0, 1.0)

    return {
        "sentiment": sentiment,
        "confidence": round(confidence, 2),
        "score": round(score, 2),
        "positive_word_count": positive_count,
        "negative_word_count": negative_count
    }


def identify_core_narratives(tweets: List[Dict[str, Any]], min_occurrences: int = 5) -> List[Dict[str, Any]]:
    """
    识别核心叙事

    Args:
        tweets: 推文列表
        min_occurrences: 最少出现次数

    Returns:
        List[Dict]: 核心叙事列表
    """
    if not tweets:
        return []

    # 提取所有文本
    all_text = " ".join([t.get("text", "") for t in tweets])

    # 提取关键词
    keywords = extract_keywords(all_text, top_n=50)

    # 提取话题标签
    all_hashtags = []
    for tweet in tweets:
        hashtags = re.findall(r'#(\w+)', tweet.get("text", ""))
        all_hashtags.extend([h.lower() for h in hashtags])

    hashtag_counts = defaultdict(int)
    for h in all_hashtags:
        hashtag_counts[h] += 1

    # 过滤低频标签
    frequent_hashtags = [
        {"tag": tag, "count": count}
        for tag, count in hashtag_counts.items()
        if count >= min_occurrences
    ]

    frequent_hashtags.sort(key=lambda x: x["count"], reverse=True)

    # 基于高频关键词和标签生成叙事假设
    narratives = []

    # 基于话题标签生成叙事
    for i, ht in enumerate(frequent_hashtags[:5]):
        # 找到与该标签相关的推文
        related_tweets = [
            t for t in tweets
            if ht["tag"] in t.get("text", "").lower()
        ]

        # 分析这些推文的情感
        related_text = " ".join([t.get("text", "") for t in related_tweets])
        sentiment = analyze_sentiment(related_text)

        narratives.append({
            "narrative_id": f"narrative_{i}",
            "type": "hashtag_based",
            "key_element": ht["tag"],
            "occurrence_count": ht["count"],
            "sentiment": sentiment["sentiment"],
            "confidence": round(sentiment["confidence"], 2),
            "sample_tweets": [
                {"id": t.get("id"), "text": t.get("text", "")[:100]}
                for t in related_tweets[:2]
            ]
        })

    return narratives


def analyze_topics_and_sentiment(tweets: List[Dict[str, Any]]) -> Dict[str, Any]:
    """
    综合主题与情感分析

    Args:
        tweets: 推文列表

    Returns:
        Dict: 分析结果
    """
    if not tweets:
        return {
            "top_keywords": [],
            "topic_classification": {},
            "overall_sentiment": {},
            "core_narratives": []
        }

    # 1. 提取关键词
    all_text = " ".join([t.get("text", "") for t in tweets])
    top_keywords = extract_keywords(all_text, top_n=30)

    # 2. 主题分类
    topic_classification = classify_topic(all_text)

    # 3. 整体情感分析
    overall_sentiment = analyze_sentiment(all_text)

    # 4. 识别核心叙事
    core_narratives = identify_core_narratives(tweets, min_occurrences=3)

    return {
        "top_keywords": top_keywords,
        "topic_classification": topic_classification,
        "overall_sentiment": overall_sentiment,
        "core_narratives": core_narratives
    }


def main():
    """命令行入口，用于测试"""
    import sys

    # 示例数据
    example_tweets = [
        {
            "id": "1",
            "text": "We need urgent government action on climate policy #climate #policy",
        },
        {
            "id": "2",
            "text": "The new climate policy is terrible for our economy #climate",
        },
        {
            "id": "3",
            "text": "Support for climate action is growing across society #climate #support",
        },
        {
            "id": "4",
            "text": "Climate change is a serious threat to our future #climate",
        },
        {
            "id": "5",
            "text": "We must oppose to current climate policy #policy",
        },
    ]

    if len(sys.argv) > 1:
        # 从文件读取
        with open(sys.argv[1], 'r', encoding='utf-8') as f:
            example_tweets = json.load(f)

    result = analyze_topics_and_sentiment(example_tweets)
    print(json.dumps(result, ensure_ascii=False, indent=2))


if __name__ == "__main__":
    main()
