import subprocess, json, re

# 从报告参考文献提取的前50条Google News链接
gn_links = [
    ("1. 最後一輪宏福苑聽證會今舉行（881903.com，07月15日）", "https://news.google.com/rss/articles/CBMiU0FVX3lxTFBNZnFCSzBiNjBkbFF5bUFleVFGcE5FVWwyQ3lRZmpjRWdOLXJhTFY1RHByWXZEdi1xZUg0S2tnWEJIX2o4b0xjVjVnTjcyLTJoVk9B0gFYQVVfeXFMTzdYNkg3WUhSaHRYQW9TS09EZ01QcFVjUlR"),
    ("2. 宏福苑聽證會揭各部門災前收大量投訴（Now新聞，07月13日）", "https://news.google.com/rss/articles/CBMiYkFVX3lxTFBUV19wVGpsZjVpQmE5LVB4ZUN2YXdvYWlubVB6UWhUd043a0xrRi1kNmlNYmJXYVZpS3FiMFdhZXoyQmJEM20zZTRoRVhVbTFhNkNLQXkyeGFmMVo2dDJtVUhR?oc=5"),
    ("3. 宏福苑聽證會｜中華發展︰有火警鐘亦難逃生（Now新聞，07月15日）", "https://news.google.com/rss/articles/CBMiYkFVX3lxTE9idDlaeE1LSlhJWTMxMThpRTJ2NldsUXI5ZWpLcW1MWHItS1hGMFVweEhQUkxhN2JCenN2Q1hYeV9UQTZjbjdJRUlPb2N1aFYyb3RMZmZaLTk2bXBuSTNsaGpR?oc=5"),
]

for label, url in gn_links[:3]:
    print(f"\n=== {label} ===")
    result = subprocess.run(
        ['/usr/local/bin/scrapling', 'extract', 'stealthy-fetch', '--url', url, '--format', 'json'],
        capture_output=True, text=True, timeout=40
    )
    if result.returncode == 0:
        try:
            data = json.loads(result.stdout)
            html = data.get('html', '')
            # 找og:url
            og_match = re.search(r'og:url["\s]+content="([^"]+)"', html)
            canonical_match = re.search(r'<link[^>]+rel="canonical"[^>]+href="([^"]+)"', html)
            # 从URL提取真实域
            url_match = re.search(r'"url":"(https?://[^"]+)"', html)
            print(f"  og:url = {og_match.group(1) if og_match else 'N/A'}")
            print(f"  canonical = {canonical_match.group(1) if canonical_match else 'N/A'}")
            print(f"  json_url = {url_match.group(1)[:100] if url_match else 'N/A'}")
            # 提取来源名称
            source_match = re.search(r'publisher["\s]+content="([^"]+)"', html)
            print(f"  publisher = {source_match.group(1) if source_match else 'N/A'}")
        except:
            print(f"  JSON解析失败: {result.stdout[:200]}")
    else:
        print(f"  错误: {result.stderr[:200]}")
