#!/usr/bin/env python3
"""
粉丝与风险评估模块

功能：
1. 粉丝画像评估（高质量用户与机器人/虚假账号比例）
2. 核心社群识别（高频互动账号）
3. 风险信号检测（造谣、煽动、网络暴力、操控舆论）
"""

import json
import re
from typing import List, Dict, Any
from collections import defaultdict


def assess_follower_quality(followers_sample: List[Dict[str, Any]]) -> Dict[str, Any]:
    """
    评估粉丝质量

    Args:
        followers_sample: 粉丝样本列表
            {
                "id": "account_id",
                "username": "username",
                "followers_count": 1000,
                "following_count": 1000,
                "created_at": "2020-01-01",
                "verified": false,
                "default_profile_image": true,
                "description": "bio text"
            }

    Returns:
        Dict: 粉丝质量评估结果
    """
    if not followers_sample:
        return {
            "total": 0,
            "high_quality_count": 0,
            "high_quality_ratio": 0,
            "bot_suspect_count": 0,
            "bot_suspect_ratio": 0,
            "fake_suspect_count": 0,
            "fake_suspect_ratio": 0
        }

    total = len(followers_sample)
    high_quality = 0
    bot_suspects = 0
    fake_suspects = 0

    for follower in followers_sample:
        follower_count = follower.get("followers_count", 0)
        following_count = follower.get("following_count", 0)
        verified = follower.get("verified", False)
        has_description = bool(follower.get("description", "").strip())
        has_default_image = follower.get("default_profile_image", False)

        # 机器人嫌疑指标
        bot_signals = 0
        if follower_count < 100:
            bot_signals += 1
        if following_count > 1000 and follower_count < 500:
            bot_signals += 1
        if not has_description:
            bot_signals += 1
        if has_default_image:
            bot_signals += 1

        # 虚假账号嫌疑指标
        fake_signals = 0
        if follower_count < 50:
            fake_signals += 1
        if not has_description:
            fake_signals += 1
        if has_default_image:
            fake_signals += 1

        # 认证且有一定粉丝 = 高质量
        if verified and follower_count >= 100:
            high_quality += 1
        elif follower_count >= 1000 and has_description and not has_default_image:
            high_quality += 1

        # 嫌疑阈值
        if bot_signals >= 3:
            bot_suspects += 1
        if fake_signals >= 3:
            fake_suspects += 1

    return {
        "total": total,
        "high_quality_count": high_quality,
        "high_quality_ratio": round(high_quality / total * 100, 2) if total > 0 else 0,
        "bot_suspect_count": bot_suspects,
        "bot_suspect_ratio": round(bot_suspects / total * 100, 2) if total > 0 else 0,
        "fake_suspect_count": fake_suspects,
        "fake_suspect_ratio": round(fake_suspects / total * 100, 2) if total > 0 else 0
    }


def identify_core_community(interactions: List[Dict[str, Any]], top_n: int = 10) -> Dict[str, Any]:
    """
    识别核心社群

    Args:
        interactions: 交互记录列表
            {
                "user_id": "account_id",
                "username": "username",
                "interaction_type": "retweet|reply|quote|mention",
                "count": 10
            }

    Returns:
        Dict: 核心社群分析结果
    """
    if not interactions:
        return {
            "total_interactions": 0,
            "unique_users": 0,
            "top_interactors": []
        }

    # 按用户统计互动
    user_stats = defaultdict(lambda: {
        "username": "",
        "retweet_count": 0,
        "reply_count": 0,
        "quote_count": 0,
        "mention_count": 0,
        "total_interactions": 0
    })

    for interaction in interactions:
        user_id = interaction.get("user_id", "")
        username = interaction.get("username", "")
        interaction_type = interaction.get("interaction_type", "")
        count = interaction.get("count", 1)

        user_stats[user_id]["username"] = username
        user_stats[user_id][f"{interaction_type}_count"] += count
        user_stats[user_id]["total_interactions"] += count

    # 转换为列表并排序
    sorted_users = sorted(
        [
            {
                "user_id": uid,
                **stats
            }
            for uid, stats in user_stats.items()
        ],
        key=lambda x: x["total_interactions"],
        reverse=True
    )

    # 分析互动模式
    for user in sorted_users:
        total = user["total_interactions"]
        if total > 0:
            user["retweet_ratio"] = round(user["retweet_count"] / total * 100, 2)
            user["reply_ratio"] = round(user["reply_count"] / total * 100, 2)
        else:
            user["retweet_ratio"] = 0
            user["reply_ratio"] = 0

    # 判定互动类型
    for user in sorted_users:
        if user["retweet_ratio"] >= 80:
            user["interaction_pattern"] = "主要转推者"
        elif user["reply_ratio"] >= 60:
            user["interaction_pattern"] = "主要对话者"
        else:
            user["interaction_pattern"] = "混合互动"

    return {
        "total_interactions": sum(i.get("count", 1) for i in interactions),
        "unique_users": len(user_stats),
        "top_interactors": sorted_users[:top_n]
    }


def detect_risk_signals(tweets: List[Dict[str, Any]]) -> Dict[str, Any]:
    """
    检测风险信号

    Args:
        tweets: 推文列表

    Returns:
        Dict: 风险信号检测结果
    """
    if not tweets:
        return {
            "overall_risk_level": "unknown",
            "risk_signals": [],
            "risk_score": 0
        }

    all_text = " ".join([t.get("text", "") for t in tweets]).lower()

    # 风险信号关键词
    risk_keywords = {
        "谣言/不实信息": ["fake", "false", "hoax", "rumor", "unverified", "misleading", "fabricated"],
        "煽动仇恨": ["hate", "attack", "destroy", "enemy", "threaten", "kill", "violent"],
        "网络暴力": ["harass", "bully", "abuse", "shame", "dox", "threat"],
        "操控舆论": ["manipulat", "bot", "army", "campaign", "coordinat", "troll"],
        "极端言论": ["extrem", "radical", "fundament", "white suprem", "terror"],
        "政治操纵": ["interfere", "rigg", "election fraud", "steal vote", "coup"],
    }

    risk_signals = []
    total_risk_score = 0

    for risk_type, keywords in risk_keywords.items():
        count = sum(all_text.count(keyword) for keyword in keywords)

        if count > 0:
            risk_level = "高" if count >= 5 else "中" if count >= 3 else "低"
            risk_signals.append({
                "type": risk_type,
                "occurrence_count": count,
                "risk_level": risk_level,
                "score": count * 2  # 每个出现2分
            })
            total_risk_score += count * 2

    # 检测大量重复内容（可能是脚本化发布）
    text_set = set(t.get("text", "") for t in tweets)
    if len(text_set) < len(tweets) * 0.5:  # 一半以上内容重复
        duplicate_ratio = (len(tweets) - len(text_set)) / len(tweets)
        risk_signals.append({
            "type": "重复内容嫌疑",
            "occurrence_count": len(tweets) - len(text_set),
            "risk_level": "中",
            "score": int(duplicate_ratio * 50)
        })
        total_risk_score += int(duplicate_ratio * 50)

    # 判定总体风险等级
    if total_risk_score >= 50:
        overall_risk_level = "极高"
    elif total_risk_score >= 30:
        overall_risk_level = "高"
    elif total_risk_score >= 15:
        overall_risk_level = "中"
    elif total_risk_score >= 5:
        overall_risk_level = "低"
    else:
        overall_risk_level = "未发现明显风险"

    return {
        "overall_risk_level": overall_risk_level,
        "risk_signals": risk_signals,
        "risk_score": total_risk_score
    }


def analyze_follower_and_risk(
    tweets: List[Dict[str, Any]],
    followers_sample: List[Dict[str, Any]],
    interactions: List[Dict[str, Any]]
) -> Dict[str, Any]:
    """
    综合粉丝和风险分析

    Args:
        tweets: 推文列表
        followers_sample: 粉丝样本
        interactions: 交互记录

    Returns:
        Dict: 综合分析结果
    """
    # 1. 粉丝质量评估
    follower_quality = assess_follower_quality(followers_sample)

    # 2. 核心社群识别
    core_community = identify_core_community(interactions, top_n=10)

    # 3. 风险信号检测
    risk_signals = detect_risk_signals(tweets)

    # 4. 综合评估
    overall_assessment = {
        "follower_quality": follower_quality,
        "core_community": core_community,
        "risk_signals": risk_signals
    }

    return overall_assessment


def main():
    """命令行入口，用于测试"""
    import sys

    # 示例数据
    example_tweets = [
        {"id": "1", "text": "This is fake news #hoax"},
        {"id": "2", "text": "We need to attack them #threaten"},
        {"id": "3", "text": "Let's harass them #bully"},
    ]

    example_followers = [
        {
            "id": "1",
            "username": "real_user",
            "followers_count": 10000,
            "following_count": 500,
            "verified": True,
            "default_profile_image": False,
            "description": "Real user"
        },
        {
            "id": "2",
            "username": "suspect_bot",
            "followers_count": 10,
            "following_count": 5000,
            "verified": False,
            "default_profile_image": True,
            "description": ""
        },
    ]

    example_interactions = [
        {
            "user_id": "1",
            "username": "frequent_interactor",
            "interaction_type": "retweet",
            "count": 50
        },
        {
            "user_id": "2",
            "username": "dialogue_partner",
            "interaction_type": "reply",
            "count": 30
        },
    ]

    if len(sys.argv) > 1:
        with open(sys.argv[1], 'r', encoding='utf-8') as f:
            data = json.load(f)
            example_tweets = data.get("tweets", example_tweets)
            example_followers = data.get("followers_sample", example_followers)
            example_interactions = data.get("interactions", example_interactions)

    result = analyze_follower_and_risk(example_tweets, example_followers, example_interactions)
    print(json.dumps(result, ensure_ascii=False, indent=2))


if __name__ == "__main__":
    main()
