#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
正义使命舆情简报 - 小时采集脚本（扩大数据源）
"""

import json
import os
import requests
from datetime import datetime

# 配置
SEARXNG_URL = 'http://localhost:8080/search'
HOURLY_DIR = '/home/admin/.openclaw/workspace/reports/justice-mission/hourly'
LOG_FILE = '/home/admin/.openclaw/workspace/reports/justice-mission/cron.log'

# 扩大采集源 - 国际主流媒体 + 国家主流媒体
QUERIES = [
    # 核心冲突
    '美以伊冲突 最新消息',
    '伊朗 以色列 军事冲突 2026',
    '美国 中东 军事行动',
    
    # 伊方动态
    '伊朗 德黑兰 官方表态',
    '伊朗 革命卫队 军事行动',
    '伊朗 最高领袖 哈梅内伊',
    
    # 美以动态
    '美国 特朗普 中东政策',
    '以色列 内塔尼亚胡 军事行动',
    '美军 中央司令部 中东',
    
    # 地区影响
    '霍尔木兹 海峡 航运',
    '中东 油价 2026',
    
    # 各方反应
    '中国 外交部 中东局势',
    '联合国 安理会 中东',
    '俄罗斯 中东 立场',
    
    # 国际主流媒体（site 搜索）
    'site:reuters.com Israel Iran conflict',
    'site:apnews.com Middle East Israel Iran',
    'site:bbc.com Israel Iran attack',
    'site:scmp.com Israel Iran Middle East',
    'site:xinhuanet.com 中东 以色列 伊朗',
    'site:globaltimes.cn 中东 局势',
    'site:observer.cn 中东 以色列',
    'site:chinanews.com 中东 局势',
]

def log(msg):
    timestamp = datetime.now().strftime('%Y-%m-%d %H:%M:%S')
    line = f'[{timestamp}] {msg}'
    print(line)
    os.makedirs(os.path.dirname(LOG_FILE), exist_ok=True)
    with open(LOG_FILE, 'a', encoding='utf-8') as f:
        f.write(line + '\n')

def main():
    log('==========  hourly 采集启动（扩大数据源） ==========')
    
    hour = datetime.now().strftime('%Y%m%d_%H00')
    output_file = f'{HOURLY_DIR}/{hour}.json'
    os.makedirs(HOURLY_DIR, exist_ok=True)
    
    log(f'开始采集 {hour} 时段数据...（{len(QUERIES)} 个查询）')
    
    all_results = []
    
    for q in QUERIES:
        try:
            params = {'q': q, 'format': 'json', 'language': 'zh', 'engines': 'bing,google,brave'}
            r = requests.get(SEARXNG_URL, params=params, timeout=15)
            if r.status_code == 200:
                data = r.json()
                all_results.append({
                    'query': q,
                    'data': data
                })
                log(f'采集完成：{q}')
        except Exception as e:
            log(f'搜索失败 {q}: {e}')
    
    output = {
        'timestamp': datetime.now().isoformat(),
        'queries': all_results
    }
    
    with open(output_file, 'w', encoding='utf-8') as f:
        json.dump(output, f, ensure_ascii=False, indent=2)
    
    log(f'采集完成，保存 {len(all_results)} 个查询结果至：{output_file}')
    log('========== hourly 采集完成 ==========')

if __name__ == '__main__':
    main()
