#!/usr/bin/env python3
"""
活跃度分析模块

功能：
1. 计算日均发帖数
2. 识别发帖高峰时段
3. 计算活跃天数占比
"""

import json
from datetime import datetime, timedelta
from typing import List, Dict, Any
from collections import defaultdict


def parse_tweet_time(time_str: str) -> datetime:
    """解析推文时间戳"""
    formats = [
        "%Y-%m-%dT%H:%M:%SZ",
        "%Y-%m-%dT%H:%M:%S.%fZ",
        "%Y-%m-%d %H:%M:%S",
    ]

    for fmt in formats:
        try:
            return datetime.strptime(time_str, fmt)
        except (ValueError, TypeError):
            continue

    return None


def calculate_activity_metrics(tweets: List[Dict[str, Any]], days: int = 90) -> Dict[str, Any]:
    """
    计算活跃度指标

    Args:
        tweets: 推文列表
        days: 分析天数（默认90天）

    Returns:
        Dict: 活跃度指标
    """
    if not tweets:
        return {
            "total_tweets": 0,
            "daily_average": 0,
            "peak_hour": None,
            "peak_hour_count": 0,
            "active_days": 0,
            "active_days_ratio": 0,
            "hourly_distribution": [],
        }

    # 解析时间戳
    parsed_tweets = []
    for tweet in tweets:
        tweet_time = parse_tweet_time(tweet.get("created_at", ""))
        if tweet_time:
            parsed_tweets.append({
                "tweet": tweet,
                "time": tweet_time,
                "hour": tweet_time.hour,
                "date": tweet_time.date()
            })

    if not parsed_tweets:
        return {
            "total_tweets": 0,
            "daily_average": 0,
            "peak_hour": None,
            "peak_hour_count": 0,
            "active_days": 0,
            "active_days_ratio": 0,
            "hourly_distribution": [],
        }

    # 1. 计算日均发帖数
    total_tweets = len(parsed_tweets)
    daily_average = total_tweets / days

    # 2. 发帖高峰时段（按小时统计）
    hourly_counts = defaultdict(int)
    for pt in parsed_tweets:
        hourly_counts[pt["hour"]] += 1

    peak_hour = max(hourly_counts.items(), key=lambda x: x[1])[0]
    peak_hour_count = hourly_counts[peak_hour]

    # 生成小时分布
    hourly_distribution = []
    for hour in range(24):
        hourly_distribution.append({
            "hour": hour,
            "count": hourly_counts[hour],
            "ratio": hourly_counts[hour] / total_tweets if total_tweets > 0 else 0
        })

    # 3. 活跃天数占比
    active_dates = set(pt["date"] for pt in parsed_tweets)
    active_days = len(active_dates)
    active_days_ratio = active_days / days

    # 4. 按天统计发帖量
    daily_counts = defaultdict(int)
    for pt in parsed_tweets:
        daily_counts[pt["date"]] += 1

    daily_stats = sorted(
        [{"date": str(d), "count": c} for d, c in daily_counts.items()],
        key=lambda x: x["date"],
        reverse=True
    )[:30]  # 最近30天

    return {
        "total_tweets": total_tweets,
        "daily_average": round(daily_average, 2),
        "peak_hour": peak_hour,
        "peak_hour_str": f"{peak_hour:02d}:00-{(peak_hour+1)%24:02d}:00",
        "peak_hour_count": peak_hour_count,
        "active_days": active_days,
        "active_days_ratio": round(active_days_ratio, 4),
        "hourly_distribution": hourly_distribution,
        "daily_stats": daily_stats,
    }


def main():
    """命令行入口，用于测试"""
    import sys

    # 示例数据
    example_tweets = [
        {"id": "1", "created_at": "2026-04-01T10:00:00Z"},
        {"id": "2", "created_at": "2026-04-01T11:00:00Z"},
        {"id": "3", "created_at": "2026-04-01T12:00:00Z"},
        {"id": "4", "created_at": "2026-04-02T10:00:00Z"},
        {"id": "5", "created_at": "2026-04-02T11:00:00Z"},
    ]

    if len(sys.argv) > 1:
        with open(sys.argv[1], 'r', encoding='utf-8') as f:
            example_tweets = json.load(f)

    result = calculate_activity_metrics(example_tweets, days=90)
    print(json.dumps(result, ensure_ascii=False, indent=2, default=str))


if __name__ == "__main__":
    main()
