#!/usr/bin/env python3
"""Search for remaining unresolved articles on media sites using Playwright."""

import asyncio
from playwright.async_api import async_playwright
import re
import json
import time

# Articles that need to be resolved via direct search
failed_articles = [
    (1, "最後一輪宏福苑聽證會今舉行", "881903.com", "07月15日"),
    (10, "宏福苑聽證會揭示圍標問題", "Singtaousa", "07月15日"),
    (11, "宏福苑大火聽證會政府總結陳詞", "Singtaousa", "07月16日"),
    (12, "宏福苑聽證會政府總結陳詞", "Singtaousa", "07月16日"),
    (13, "宏福苑第六輪聽證會月中舉行", "Yahoo", "07月02日"),
    (14, "宏福苑聽證會揭各部門災前收大量投訴", "Yahoo", "07月13日"),
    (15, "宏福苑聽證會｜競委會：有圍標由三合會領導", "Yahoo", "07月15日"),
    (16, "宏福苑聽證會｜置邦：大維修期間沒任何監督角色", "Yahoo", "07月15日"),
    (17, "晨早新聞重點｜宏福苑聽證會", "Yahoo", "07月16日"),
    (18, "宏福苑聽證會｜居民方總結陳詞批有人卸責", "Yahoo", "07月16日"),
    (19, "宏福苑聽證會｜政府指承建商須負首要責任", "Yahoo", "07月16日"),
    (20, "即日焦點｜宏福苑聽證會總結陳詞", "Yahoo", "07月16日"),
    (21, "聽證會｜宏福苑居民指政府有不可推卸責任", "am730", "07月16日"),
    (23, "宏福苑大火獨委會總結陳詞", "on.cc東網", "07月17日"),
    (24, "宏福苑聽證會｜最後一輪聽證會今舉行", "singtao.ca", "07月15日"),
    (25, "宏福苑聽證會｜政府代表總結陳詞", "singtao.ca", "07月16日"),
    (26, "宏福苑聽證會｜黃碧嬌工程前期扮演一定角色", "singtao.ca", "07月16日"),
    (27, "宏福苑聽證會｜物管置邦斥宏泰消防董事供詞信口雌黃", "信報網站", "07月15日"),
    (29, "宏福苑聽證會｜最後一輪最後一場", "信報網站", "07月17日"),
    (32, "宏福苑五級火聽證會｜政府方陳詞", "庭刊", "07月16日"),
    (33, "宏福苑五級火聽證會｜政府方結案陳詞", "庭刊", "07月16日"),
    (34, "宏福苑火災聽證會｜物管置邦", "明報新聞網", "07月15日"),
    (35, "宏福苑火災聽證會｜政府代表稱監管責任", "明報新聞網", "07月16日"),
    (36, "宏福苑火災聽證會｜政府：六大因素促成火災", "明報新聞網", "07月16日"),
    (37, "宏福苑火災聽證會｜政府稱識別5個制度弱點", "明報新聞網", "07月16日"),
    (38, "宏福苑居民：一句「對不起」及認責非奢侈", "明報新聞網", "07月16日"),
    (39, "宏福苑火災聽證會｜居民代表律師", "明報新聞網", "07月16日"),
    (40, "【持續更新】宏福苑火災聽證會｜政府指六大因素引發火災", "明報新聞網", "07月16日"),
    (41, "旁聽居民批淡化責任", "明報新聞網", "07月16日"),
    (42, "鄧國權黃碧嬌無作供 居民陳辭點名", "明報新聞網", "07月16日"),
    (43, "宏福苑火災聽證會｜有居民認為政府部門對火災有責任", "明報新聞網", "07月16日"),
    (44, "宏福苑火災聽證會｜政府：火警規模前所未見", "明報新聞網", "07月16日"),
    (45, "識別5「系統性弱點」", "明報新聞網", "07月16日"),
    (46, "居民總結：環環失誤釀禍", "明報新聞網", "07月16日"),
    (47, "宏福苑聽證會第六輪周三起舉行3場", "星島頭條", "07月13日"),
    (48, "宏福苑聽證會｜政府承認監管制度存五大弱點", "星島頭條", "07月16日"),
    (49, "宏福苑聽證會｜何偉豪留在宏泰閣嘗試協助疏散", "星島頭條", "07月16日"),
    (50, "宏福苑大火最終一輪聽證會", "星島頭條", "07月17日"),
]

# Search URLs for each media source
search_urls = {
    "881903.com": "https://www.881903.com/search?q={keyword}",
    "Singtaousa": "https://www.singtaousa.com/search?q={keyword}",
    "Yahoo": "https://news.search.yahoo.com/search?p={keyword}",
    "am730": "https://www.am730.com.hk/search?q={keyword}",
    "on.cc東網": "https://www.on.cc/news/search.html?q={keyword}",
    "singtao.ca": "https://www.singtao.ca/search?q={keyword}",
    "信報網站": "https://www.hkej.com/search?q={keyword}",
    "庭刊": "https://www.tingchen.com.hk/search?q={keyword}",
    "明報新聞網": "https://news.mingpao.com/search?q={keyword}",
    "星島頭條": "https://www.stheadline.com/search?q={keyword}",
}


async def search_media_site(page, url, keyword):
    """Search on a media site and extract article links."""
    search_url = url.replace("{keyword}", keyword)
    
    try:
        await page.goto(search_url, timeout=15000)
        await page.wait_for_timeout(3000)
        
        # Get all links on the page
        links = await page.eval_on_selector_all('a[href]', '''
            elements => elements.map(e => ({
                href: e.href,
                text: e.textContent.trim().substring(0, 200)
            }))
        ''')
        
        # Filter for relevant links
        relevant = []
        for link in links:
            href = link['href']
            text = link['text']
            if href and len(href) > 20:
                # Check if the link is from the same domain or contains relevant keywords
                if any(domain in href for domain in ['881903', 'singtao', 'am730', 'on.cc', 'hkej', 'tingchen', 'mingpao', 'stheadline', 'yahoo']):
                    if '宏福苑' in text or '聽證' in text or '大火' in text:
                        relevant.append({'url': href, 'text': text})
        
        return relevant
        
    except Exception as e:
        return []


async def resolve_all_failed(failed_articles):
    """Resolve all failed articles via direct search."""
    results = {}
    
    # First load previously successful results
    try:
        with open('/root/.openclaw/workspace/resolved_urls.json', 'r', encoding='utf-8') as f:
            existing = json.load(f)
        for r in existing:
            if r.get('original_url'):
                results[r['num']] = r['original_url']
    except:
        pass
    
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=['--no-sandbox', '--disable-dev-shm-usage'])
        context = await browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
            locale='zh-HK'
        )
        
        for i, (num, title, source, date) in enumerate(failed_articles):
            if num in results:
                print(f"[{i+1}/{len(failed_articles)}] #{num} Already resolved, skipping")
                continue
            
            page = await context.new_page()
            
            # Get the search URL for this source
            search_url = search_urls.get(source, "")
            if not search_url:
                print(f"[{i+1}/{len(failed_articles)}] #{num} No search URL for {source}")
                await page.close()
                continue
            
            # Extract a short keyword for search
            keyword = title[:20]
            
            print(f"[{i+1}/{len(failed_articles)}] #{num} Searching {source} for: {keyword[:30]}...")
            
            try:
                await page.goto(search_url.replace("{keyword}", keyword), timeout=15000)
                await page.wait_for_timeout(3000)
                
                # Get the current URL (after any redirects)
                current_url = page.url
                
                # Look for article links
                links = await page.eval_on_selector_all('a[href]', '''
                    elements => elements.map(e => ({
                        href: e.href,
                        text: e.textContent.trim().substring(0, 200)
                    }))
                ''')
                
                # Find relevant links
                found_url = None
                for link in links:
                    href = link['href']
                    text = link['text']
                    if href and len(href) > 20 and '宏福苑' in (text + href):
                        # Skip Google News links
                        if 'news.google.com' in href:
                            continue
                        found_url = href
                        print(f"  Found: {href[:100]}")
                        break
                
                if found_url:
                    results[num] = found_url
                else:
                    # Try the page title to see if we landed on an article
                    page_title = await page.title()
                    if '宏福苑' in page_title or '聽證' in page_title:
                        results[num] = current_url
                        print(f"  Using current URL: {current_url[:100]}")
                    else:
                        print(f"  NOT FOUND on {source}")
                
            except Exception as e:
                print(f"  Error: {str(e)[:100]}")
            
            await page.close()
            await asyncio.sleep(1)
        
        await browser.close()
    
    return results


async def main():
    print(f"Searching for {len(failed_articles)} failed articles...")
    results = await resolve_all_failed(failed_articles)
    
    # Save updated results
    all_results = []
    try:
        with open('/root/.openclaw/workspace/resolved_urls.json', 'r', encoding='utf-8') as f:
            existing = json.load(f)
        
        # Deduplicate and update
        seen = set()
        for r in existing:
            if r['num'] not in seen:
                seen.add(r['num'])
                if r['num'] in results and r.get('original_url') is None:
                    r['original_url'] = results[r['num']]
                    r['resolved_via'] = 'direct_search'
                all_results.append(r)
                seen.add(r['num'])
        
        # Add any new results
        for num, url in results.items():
            if num not in seen:
                all_results.append({
                    'num': num,
                    'original_url': url,
                    'resolved_via': 'direct_search'
                })
    except Exception as e:
        print(f"Error loading existing results: {e}")
        all_results = [{'num': num, 'original_url': url} for num, url in results.items()]
    
    with open('/root/.openclaw/workspace/resolved_urls_v2.json', 'w', encoding='utf-8') as f:
        json.dump(all_results, f, ensure_ascii=False, indent=2)
    
    print(f"\nResults saved to resolved_urls_v2.json")
    print(f"Total resolved: {len(results)}")
    
    # Print summary
    print("\n=== Summary ===")
    for num, url in sorted(results.items()):
        print(f"#{num}: {url[:100] if url else 'N/A'}")

asyncio.run(main())
