#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
正义使命舆情采集脚本
每小时自动采集美以伊冲突相关舆情信息
"""

import json
import os
from datetime import datetime
from pathlib import Path

# 输出目录
OUTPUT_DIR = Path("/home/admin/.openclaw/workspace/reports/justice-mission/hourly")
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)

# 采集源列表
SOURCES = [
    {"name": "Reuters", "url": "https://www.reuters.com/world/middle-east/"},
    {"name": "AP News", "url": "https://apnews.com/hub/israel-palestinian-conflict"},
    {"name": "SCMP", "url": "https://www.scmp.com/news/china/military"},
    {"name": "Global Times", "url": "https://www.globaltimes.cn/world/"},
    {"name": "China Daily", "url": "https://www.chinadaily.com.cn/world"},
    {"name": "NYTimes CN", "url": "https://cn.nytimes.com/"},
    {"name": "NHK", "url": "https://www3.nhk.or.jp/news/html/topic/china.html"},
    {"name": "CCTV", "url": "https://news.cctv.com/world/"},
]

# 搜索关键词
KEYWORDS = [
    "US Israel Iran war",
    "美以伊冲突",
    "伊朗 最高领袖",
    "霍尔木兹海峡",
    "中东 石油",
    "台湾 军售",
    "Iran missile attack",
    "Hormuz Strait",
]

def collect_hourly_data():
    """采集每小时舆情数据"""
    timestamp = datetime.now().strftime("%Y%m%d_%H00")
    output_file = OUTPUT_DIR / f"{timestamp}.json"
    
    data = {
        "timestamp": datetime.now().isoformat(),
        "keywords": KEYWORDS,
        "sources": SOURCES,
        "data": []
    }
    
    # 注意：实际采集需要调用 web_fetch 或 web_search API
    # 此处为框架代码，实际执行由主会话管理
    
    with open(output_file, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    
    print(f"[{timestamp}] 采集完成，保存至 {output_file}")
    return output_file

def generate_daily_summary():
    """生成每日汇总报告（08:00 执行）"""
    today = datetime.now().strftime("%Y%m%d")
    summary_file = OUTPUT_DIR / f"{today}_summary.json"
    
    # 收集今日所有小时数据
    hourly_files = list(OUTPUT_DIR.glob(f"{today}_*.json"))
    
    summary = {
        "date": today,
        "generated_at": datetime.now().isoformat(),
        "total_collections": len(hourly_files),
        "files": [str(f) for f in hourly_files]
    }
    
    with open(summary_file, 'w', encoding='utf-8') as f:
        json.dump(summary, f, ensure_ascii=False, indent=2)
    
    print(f"[SUMMARY] 生成每日汇总：{summary_file}")
    return summary_file

if __name__ == "__main__":
    import sys
    
    if len(sys.argv) > 1 and sys.argv[1] == "--summary":
        generate_daily_summary()
    else:
        collect_hourly_data()
