#!/usr/bin/env python3
"""
首发媒体识别脚本

功能：
1. 根据发布时间对新闻进行排序
2. 识别首发媒体（最早发布的媒体）
3. 分析传播路径和关键节点
"""

import json
from datetime import datetime
from typing import List, Dict, Any, Optional
from collections import defaultdict


def parse_timestamp(timestamp_str: str) -> Optional[datetime]:
    """
    解析时间戳字符串

    支持多种格式：
    - ISO 8601: 2024-03-15T10:30:00Z
    - 中文格式: 2024年3月15日 10:30
    - Unix时间戳: 1710504600
    """
    formats = [
        "%Y-%m-%dT%H:%M:%SZ",
        "%Y-%m-%d %H:%M:%S",
        "%Y-%m-%d %H:%M",
        "%Y年%m月%d日 %H:%M",
        "%Y年%m月%d日",
        "%Y-%m-%d",
    ]

    # 尝试解析为Unix时间戳
    try:
        return datetime.fromtimestamp(int(timestamp_str))
    except (ValueError, TypeError):
        pass

    # 尝试各种格式
    for fmt in formats:
        try:
            return datetime.strptime(timestamp_str, fmt)
        except (ValueError, TypeError):
            continue

    return None


def identify_first_publisher(related_news: List[Dict[str, Any]]) -> Dict[str, Any]:
    """
    识别首发媒体

    Args:
        related_news: 关联新闻列表，每个元素包含：
            - media_name: 媒体名称
            - media_url: 媒体URL（可选）
            - article_url: 文章URL
            - title: 文章标题
            - publish_time: 发布时间
            - content: 文章内容摘要（可选）

    Returns:
        Dict: 首发媒体分析结果
            - first_publisher: 首发媒体信息
            - sorted_news: 按时间排序的新闻列表
            - propagation_timeline: 传播时间线
            - analysis_notes: 分析备注
    """
    if not related_news:
        return {
            "error": "输入新闻列表为空",
            "first_publisher": None
        }

    # 1. 解析并标准化时间
    processed_news = []
    for idx, news in enumerate(related_news):
        publish_time = news.get("publish_time", "")
        parsed_time = parse_timestamp(publish_time)

        processed_news.append({
            "index": idx,
            "media_name": news.get("media_name", "未知媒体"),
            "media_url": news.get("media_url", ""),
            "article_url": news.get("article_url", ""),
            "title": news.get("title", ""),
            "content": news.get("content", ""),
            "publish_time_original": publish_time,
            "publish_time_parsed": parsed_time,
            "timestamp": parsed_time.timestamp() if parsed_time else float('inf'),
        })

    # 2. 按时间排序（有效时间在前，无效时间在后）
    valid_news = [n for n in processed_news if n["publish_time_parsed"] is not None]
    invalid_news = [n for n in processed_news if n["publish_time_parsed"] is None]

    sorted_valid = sorted(valid_news, key=lambda x: x["timestamp"])
    sorted_news = sorted_valid + invalid_news

    # 3. 识别首发媒体
    if sorted_valid:
        first_publisher = sorted_valid[0]
    else:
        # 所有新闻时间都无效，按原顺序第一个
        first_publisher = processed_news[0]

    # 4. 构建传播时间线
    timeline = []
    for i, news in enumerate(sorted_valid[:10]):  # 取前10个有效新闻
        if i == 0:
            delay = 0
        else:
            delay = news["timestamp"] - sorted_valid[i-1]["timestamp"]

        timeline.append({
            "rank": i + 1,
            "media_name": news["media_name"],
            "publish_time": news["publish_time_original"],
            "delay_from_previous_minutes": round(delay / 60, 2) if delay > 0 else 0,
            "article_url": news["article_url"],
        })

    # 5. 生成分析备注
    notes = []
    if len(invalid_news) > 0:
        notes.append(f"注意：{len(invalid_news)}条新闻的时间戳无法解析，已移至末尾")

    if len(valid_news) > 1:
        first_two_delay = sorted_valid[1]["timestamp"] - sorted_valid[0]["timestamp"]
        if first_two_delay < 300:  # 5分钟内
            notes.append(f"前两名媒体发布时间间隔极短（{first_two_delay/60:.1f}分钟），可能为同步发布或数据源同步问题")

    # 统计媒体出现频率
    media_counts = defaultdict(int)
    for news in sorted_valid:
        media_counts[news["media_name"]] += 1

    if len([m for m, c in media_counts.items() if c > 1]) > 0:
        notes.append("检测到同一媒体发布多条相关新闻，可能为持续跟踪报道")

    return {
        "first_publisher": {
            "media_name": first_publisher["media_name"],
            "media_url": first_publisher["media_url"],
            "article_url": first_publisher["article_url"],
            "title": first_publisher["title"],
            "publish_time": first_publisher["publish_time_original"],
            "confidence": "high" if len(valid_news) > 0 else "low",
        },
        "sorted_news": sorted_news,
        "propagation_timeline": timeline,
        "analysis_notes": notes,
        "total_news_count": len(related_news),
        "valid_time_count": len(valid_news),
    }


def main():
    """命令行入口，用于测试"""
    import sys

    # 示例数据
    example_data = [
        {
            "media_name": "路透社",
            "media_url": "https://www.reuters.com",
            "article_url": "https://www.reuters.com/world/example-article",
            "title": "Example News Title",
            "publish_time": "2024-03-15T10:30:00Z",
            "content": "News content..."
        },
        {
            "media_name": "BBC",
            "media_url": "https://www.bbc.com",
            "article_url": "https://www.bbc.com/news/example",
            "title": "Example Title",
            "publish_time": "2024-03-15T11:00:00Z",
        },
        {
            "media_name": "CNN",
            "article_url": "https://www.cnn.com/world/example",
            "title": "Another Title",
            "publish_time": "2024-03-15T10:45:00Z",
        },
    ]

    if len(sys.argv) > 1:
        # 从文件读取
        with open(sys.argv[1], 'r', encoding='utf-8') as f:
            example_data = json.load(f)

    result = identify_first_publisher(example_data)
    print(json.dumps(result, ensure_ascii=False, indent=2))


if __name__ == "__main__":
    main()
