#!/usr/bin/env python3
"""
社交媒体事件挖掘与分析主脚本

整合：事件聚类 + 事件分析 + 报告生成
"""

import json
import sys
from pathlib import Path
from datetime import datetime, timedelta
from typing import List, Dict, Any

# 添加脚本目录到路径
sys.path.insert(0, str(Path(__file__).parent))

from event_clusterer import cluster_tweets_into_events, parse_tweet_time
from event_analyzer import analyze_event_comprehensive
from generate_report import generate_full_report


def filter_tweets_by_date_range(
    tweets: List[Dict[str, Any]],
    days_back: int = 30
) -> List[Dict[str, Any]]:
    """
    按日期范围过滤推文

    Args:
        tweets: 推文列表
        days_back: 向前推算的天数

    Returns:
        List[Dict]: 过滤后的推文列表
    """
    cutoff_date = datetime.now() - timedelta(days=days_back)

    filtered = []
    for tweet in tweets:
        tweet_time = parse_tweet_time(tweet.get("created_at", ""))
        if tweet_time and tweet_time >= cutoff_date:
            filtered.append(tweet)

    return filtered


def mine_social_media_events(
    account_info: Dict[str, Any],
    tweets: List[Dict[str, Any]],
    interaction_data: Dict[str, Any] = None,
    time_window_hours: float = 48.0,
    min_tweets_per_event: int = 3,
    days_back: int = 30
) -> Dict[str, Any]:
    """
    社交媒体事件挖掘主入口

    Args:
        account_info: 账号信息
        tweets: 推文列表
        interaction_data: 交互数据
        time_window_hours: 时间窗口（小时）
        min_tweets_per_event: 每个事件最少推文数
        days_back: 分析时间范围（天）

    Returns:
        Dict: 挖掘结果
            - events: 事件列表
            - report: Markdown报告
            - summary: 汇总信息
    """
    # Step 1: 过滤时间范围
    filtered_tweets = filter_tweets_by_date_range(tweets, days_back)

    # Step 2: 事件聚类
    events = cluster_tweets_into_events(
        filtered_tweets,
        time_window_hours=time_window_hours,
        min_tweets_per_event=min_tweets_per_event
    )

    # Step 3: 事件分析
    analyzed_events = []
    for event in events:
        analysis = analyze_event_comprehensive(event, interaction_data)
        event["analysis"] = analysis
        analyzed_events.append(event)

    # Step 4: 生成报告
    time_range_desc = f"近{days_back}天"
    report = generate_full_report(analyzed_events, account_info, time_range_desc)

    # Step 5: 生成汇总信息
    total_tweets = len(filtered_tweets)
    total_events = len(analyzed_events)
    total_engagement = sum(e.get("total_engagement", {}).get("total", 0) for e in analyzed_events)

    summary = {
        "time_range_days": days_back,
        "total_tweets_analyzed": total_tweets,
        "original_tweets_count": len(tweets),
        "events_identified": total_events,
        "total_engagement": total_engagement,
        "avg_engagement_per_event": round(total_engagement / total_events, 2) if total_events > 0 else 0,
        "generated_at": datetime.now().strftime("%Y-%m-%dT%H:%M:%SZ"),
    }

    # Step 6: 整合结果
    result = {
        "account_info": account_info,
        "events": analyzed_events,
        "report": report,
        "summary": summary,
        "parameters": {
            "time_window_hours": time_window_hours,
            "min_tweets_per_event": min_tweets_per_event,
            "days_back": days_back,
        },
    }

    return result


def main():
    """命令行入口"""
    if len(sys.argv) > 1:
        # 从文件读取输入数据
        input_file = sys.argv[1]
        with open(input_file, 'r', encoding='utf-8') as f:
            input_data = json.load(f)
    else:
        # 使用示例数据
        input_data = {
            "account_info": {
                "id": "elonmusk",
                "username": "Elon Musk",
                "bio": "Mars, cars, chips & rockets",
                "location": "Austin, Texas",
                "created_at": "2009-06-02",
                "followers_count": 150000000,
                "following_count": 100,
                "verified": True,
            },
            "tweets": [
                {
                    "id": "1",
                    "text": "Exciting product launch coming soon! #product #launch",
                    "created_at": "2024-03-15T10:00:00Z",
                    "like_count": 10000,
                    "retweet_count": 5000,
                    "reply_count": 1000,
                },
                {
                    "id": "2",
                    "text": "The launch is happening tomorrow #launch",
                    "created_at": "2024-03-15T11:00:00Z",
                    "like_count": 20000,
                    "retweet_count": 8000,
                    "reply_count": 2000,
                },
                {
                    "id": "3",
                    "text": "Here are the details of the launch #product #launch",
                    "created_at": "2024-03-15T12:00:00Z",
                    "like_count": 15000,
                    "retweet_count": 6000,
                    "reply_count": 1500,
                },
                {
                    "id": "4",
                    "text": "Big announcement about technology update #tech #update",
                    "created_at": "2024-03-20T10:00:00Z",
                    "like_count": 8000,
                    "retweet_count": 3000,
                    "reply_count": 500,
                },
                {
                    "id": "5",
                    "text": "Update is now available #update",
                    "created_at": "2024-03-20T11:00:00Z",
                    "like_count": 5000,
                    "retweet_count": 2000,
                    "reply_count": 300,
                },
            ],
            "interaction_data": {
                "likes": [],
                "followers_sample": [],
                "key_tweet_interactions": {},
            },
        }

    # 执行挖掘
    result = mine_social_media_events(
        account_info=input_data.get("account_info", {}),
        tweets=input_data.get("tweets", []),
        interaction_data=input_data.get("interaction_data"),
        time_window_hours=input_data.get("time_window_hours", 48.0),
        min_tweets_per_event=input_data.get("min_tweets_per_event", 3),
        days_back=input_data.get("days_back", 30),
    )

    # 输出报告
    print(result["report"])

    # 如果需要完整JSON结果，取消注释以下内容
    # print("\n" + "="*80 + "\n")
    # print("完整的JSON结果：")
    # print(json.dumps(result, ensure.get_ascii=False, indent=2, default=str))


if __name__ == "__main__":
    main()
