#!/usr/bin/env python3
"""批量移除技能文件中的媒体黑名单条款"""
import re
import os

FILES = [
    "skills/开源情报-国家详尽报告/SKILL.md",
    "skills/开源情报-高校舆情分析/SKILL.md",
    "skills/osint-skills/开源情报 - 突发事件境内外舆情分析/SKILL.md",
    "skills/开源情报-战略情报报告风险版/SKILL.md",
    "skills/开源情报-突发事件境内外舆情分析/SKILL.md",
    "skills/开源情报-人物画像/SKILL.md",
    "skills/开源情报-美以伊战争简报/report_template.md",
    "skills/开源情报-美以伊战争简报/SKILL.md",
    "skills/开源情报-军事目标图谱采集/SKILL.md",
    "skills/开源情报-战略情报报告标准大纲（万能版）/SKILL.md",
    "skills/开源情报-台海中东冲突关联分析/SKILL.md",
    "skills/开源情报-境外涉华舆情简报/SKILL.md",
    "skills/开源情报-移民分析简报/SKILL.md",
    "skills/开源情报-社交媒体账号评估/SKILL.md",
    "skills/开源情报-军事人物目标采集/SKILL.md",
    "skills/开源情报-事件综合分析/SKILL.md",
    "skills/开源情报-事件参阅报告/SKILL.md",
    "skills/开源情报-台湾每日舆情简报/SKILL.md",
    "skills/开源情报-自动选题/SKILL.md",
    "skills/开源情报-供应链溯源分析简报/SKILL.md",
    "skills/开源情报-试验项目采集/SKILL.md",
    "skills/开源情报-采办项目采集/SKILL.md",
    "skills/开源情报-美国每日舆情简报/SKILL.md",
    "skills/开源情报-日本每日舆情简报/SKILL.md",
    "skills/开源情报-社交媒体事件挖掘/SKILL.md",
    "skills/开源情报-海外项目分析简报/SKILL.md",
    "skills/开源情报-美对华制裁监测/SKILL.md",
    "skills/开源情报-帖子溯源/SKILL.md",
]

BASE = "/root/.openclaw/workspace"

changes_log = []

for relpath in FILES:
    fpath = os.path.join(BASE, relpath)
    if not os.path.exists(fpath):
        changes_log.append(f"[SKIP] {relpath} - 文件不存在")
        continue
    
    with open(fpath, 'r', encoding='utf-8') as f:
        content = f.read()
    
    original = content
    changes = []
    
    # Pattern 1: Remove "去除非官方媒体" lines that mention 大纪元/VOA etc
    # These are typically bullet points like:
    # "3. **去除非官方媒体** - 不使用大纪元、美国之音等非官方/不可靠媒体来源"
    # "3. **去除非官方媒体** — 不使用大纪元、美国之音等非官方/不可靠媒体来源"
    # "5. **去除非官方媒体** — 不使用大纪元、美国之音、自由亚洲电台等非官方/不可靠媒体来源"
    pattern1 = r'\n\d+\.\s+\*\*去除非官方媒体\*\*\s*[-—]\s*不使用大纪元[、，][^\n]*\n'
    new_content = re.sub(pattern1, '\n', content)
    if new_content != content:
        changes.append("移除'去除非官方媒体'条款")
        content = new_content
    
    # Pattern 2: Remove standalone blacklist lines/blocks like:
    # "- 大纪元"
    # "- 美国之音（VOA）"
    # "- 自由亚洲电台（RFA）"
    # "- 新唐人"
    # "- 新唐人电视台"
    # "- 大纪元、美国之音（VOA）、自由亚洲电台（RFA）、新唐人电视台、其他非官方/不可靠来源"
    pattern2 = r'\n-\s+大纪元[^\n]*\n'
    content = re.sub(pattern2, '\n', content)
    pattern2b = r'\n-\s+美国之音（VOA）[^\n]*\n'
    content = re.sub(pattern2b, '\n', content)
    pattern2c = r'\n-\s+自由亚洲电台（RFA）[^\n]*\n'
    content = re.sub(pattern2c, '\n', content)
    pattern2d = r'\n-\s+新唐人[^\n]*\n'
    content = re.sub(pattern2d, '\n', content)
    pattern2e = r'\n-\s+大纪元、美国之音（VOA）、自由亚洲电台（RFA）、新唐人电视台[^\n]*\n'
    content = re.sub(pattern2e, '\n', content)
    
    # Pattern 3: Remove lines like:
    # "❌ 大纪元、VOA、RFA、新唐人"
    # "❌ 大纪元、美国之音 (VOA)、自由亚洲电台 (RFA)、新唐人"
    pattern3 = r'\n❌\s+大纪元[^\n]*\n'
    content = re.sub(pattern3, '\n', content)
    
    # Pattern 4: Remove checklist items like:
    # "- [ ] 未使用禁止的媒体来源（大纪元、VOA、RFA 等）"
    # "- [ ] 未使用禁止的媒体来源（大纪元、VOA、RFA、新唐人等）"
    pattern4 = r'\n-\s+\[\s*\]\s+未使用禁止的媒体来源[^\n]*\n'
    content = re.sub(pattern4, '\n', content)
    
    # Pattern 5: Fix table rows that reference the blacklist in quality scoring:
    # "| 1 | **真实可溯源** | 绝不编造；URL 必须 HTTP 200；2-3 源交叉印证；禁用大纪元/VOA/RFA/新唐人 |"
    # Replace with: "| 1 | **真实可溯源** | 绝不编造；URL 必须 HTTP 200；2-3 源交叉印证 |"
    pattern5 = r'禁用大纪元/VOA/RFA/新唐人'
    new_content = content.replace(pattern5, '')
    # Clean up trailing ；|
    new_content = re.sub(r'；\s*\|', ' |', new_content)
    if new_content != content:
        changes.append("更新质量评分行（移除黑名单引用）")
        content = new_content
    
    # Pattern 6: Remove "去除非官方媒体（大纪元、VOA、RFA 等）" inline references
    pattern6 = r'去除非官方媒体（大纪元[^\n]*）'
    content = re.sub(pattern6, '去除非官方媒体', content)
    
    # Pattern 7: Remove "不使用大纪元、美国之音等非官方/不可靠媒体来源" inline
    pattern7 = r'不使用大纪元[、，][^\n]*媒体来源'
    content = re.sub(pattern7, '', content)
    
    # Pattern 8: Remove standalone lines mentioning 黑名单媒体
    # "| D级 | 不明来源/黑名单媒体 | 禁止使用 |"
    pattern8 = r'\n\|\s*D级\s*\|[^|]*黑名单[^|]*\|[^|]*\n'
    content = re.sub(pattern8, '\n', content)
    
    # Pattern 9: Remove "媒体源分级与黑名单机制" references
    pattern9 = r'媒体源分级与黑名单机制'
    content = content.replace(pattern9, '媒体源分级机制')
    
    # Pattern 10: Remove "禁用来源" lines in MEMORY-style blocks
    # "- **禁用来源:** ❌ 大纪元、VOA、RFA、新唐人"
    pattern10 = r'\n-\s+\*\*禁用来源:\*\*\s*❌\s*大纪元[^\n]*\n'
    content = re.sub(pattern10, '\n', content)
    
    # Pattern 11: Remove "4. **去伪存真** — 不使用大纪元、VOA、RFA、新唐人等禁止来源"
    pattern11 = r'\n\d+\.\s+\*\*去伪存真\*\*\s*—\s*不使用大纪元[^\n]*\n'
    content = re.sub(pattern11, '\n', content)
    
    # Pattern 12: Remove remaining "禁止来源" lines that list these media
    # "- 大纪元、美国之音（VOA）、自由亚洲电台（RFA）、新唐人电视台"
    pattern12 = r'\n-\s+大纪元、美国之音（VOA）、自由亚洲电台（RFA）、新唐人电视台[^\n]*\n'
    content = re.sub(pattern12, '\n', content)
    
    # Clean up: remove excessive blank lines (3+ consecutive -> 2)
    content = re.sub(r'\n{3,}', '\n\n', content)
    
    if content != original:
        with open(fpath, 'w', encoding='utf-8') as f:
            f.write(content)
        changes_log.append(f"[DONE] {relpath} - {', '.join(changes) if changes else '格式清理'}")
    else:
        # Check if any pattern matched but was already clean
        if '大纪元' in original or 'VOA' in original or 'RFA' in original or '新唐人' in original:
            # Still has references - log for manual review
            remaining = []
            for i, line in enumerate(content.split('\n'), 1):
                if any(kw in line for kw in ['大纪元', 'VOA', 'RFA', '新唐人', '自由时报']):
                    remaining.append(f"  L{i}: {line.strip()[:100]}")
            if remaining:
                changes_log.append(f"[REVIEW] {relpath} - 仍有残留引用:\n" + "\n".join(remaining))
            else:
                changes_log.append(f"[OK] {relpath} - 无需修改")
        else:
            changes_log.append(f"[OK] {relpath} - 无需修改")

for log in changes_log:
    print(log)
