#!/usr/bin/env python3
"""
认知电子战专利批量采集器
高效批量采集，支持增量保存、断点续传、错误恢复
"""

import json
import os
import re
import subprocess
import sys
import time
import urllib.parse

OUTPUT_DIR = "/tmp/patents_cognitive_ew"
PROGRESS_FILE = os.path.join(OUTPUT_DIR, "progress.json")
RESULTS_FILE = os.path.join(OUTPUT_DIR, "patents.json")
SCRAPLING_BIN = "scrapling"

# 17个标准采集字段
PATENT_FIELDS = [
    "中文名称", "英文名称", "摘要", "专利类型", "申请/专利号",
    "专利日期", "公开/公告号", "公开/公告日", "主分类号", "分类号",
    "发明/设计人", "优先权", "法律状态", "专利说明", "专利授权日期",
    "所属国家",
]

# ──────────── 采集任务列表 ────────────

TITLE_TASKS = [
    "Radar Jamming Decision-Making in Cognitive Electronic Warfare: A Review",
    "A Deep Autoencoder Trust Model for Mitigating Jamming Attack in IoT Assisted by Cognitive Radio",
    "Robust secure UAV relay-assisted cognitive communications with resource allocation and cooperative jamming",
    "The Development From Adaptive to Cognitive Radar Resource Management",
    "Covert Communication With Cognitive Jammer",
    "Closing the Loop on Cognitive Radar for Spectrum Sharing",
    "An Intelligent Anti-Jamming Scheme for Cognitive Radio Based on Deep Reinforcement Learning",
    "Survey on cognitive anti-jamming communications",
    "Automatic Jamming Signal Classification in Cognitive UAV Radios",
    "Cognitive Radar Waveform Design and Prototype for Coexistence With Communications",
    "Anti-Jamming Game to Combat Intelligent Jamming for Cognitive Radio Networks",
    "Improving anti-jamming decision-making strategies for cognitive radar via multi-agent deep reinforcement learning",
    "Multifunctional Radar Cognitive Jamming Decision Based on Dueling Double Deep Q-Network",
    "Design of Cognitive Jamming Decision-Making System Against MFR Based on Reinforcement Learning",
    "Evaluation of Real-Time Predictive Spectrum Sharing for Cognitive Radar",
    "Experimental Analysis of Block-Sparsity-Based Spectrum Sensing Techniques for Cognitive Radar",
    "Cognitive Radar Target Tracking Using Intelligent Waveforms Based on Reinforcement Learning",
    "Jamming Recognition of Carrier-Free UWB Cognitive Radar Based on MANet",
    "Radio environment maps for military cognitive networks: density of small-scale sensor network vs. map quality",
    "Assessing Agile Spectrum Management for Cognitive Radar on Measured Data",
    "Joint recognition and parameter estimation of cognitive radar work modes with LSTM-transformer",
    "Microwave Photonic Cognitive Radar With a Subcentimeter Resolution",
    "Memory-enhanced cognitive radar for autonomous navigation",
    "Demonstration of Real-time Cognitive Radar using Spectrally-Notched Random FM Waveforms",
    "Frequency Diverse Array Signal Optimization: From Non-Cognitive to Cognitive Radar",
    "Deep Reinforcement Learning for Cognitive Radar Spectrum Sharing: A Continuous Control Approach",
    "Distributed Online Learning for Coexistence in Cognitive Radar Networks",
    "Practical Aspects of Cognitive Radar",
    "Cognitive radar for waveform diversity utilization",
    "Spectral Prediction and Notching of RF Emitters for Cognitive Radar Coexistence",
    "Next-Generation Cognitive Radar Systems",
    "A Cognitive Jamming Decision-Making Method Based on Heuristic Improved A2C Algorithm",
    "A framework for spectrum sharing in cognitive radio networks for military applications",
    "Cognitive jammers assisted covert communication in cognitive radio networks",
    "Threat Assessment of Cognitive Electronic Warfare to Communication based on Self-organizing Competitive Neural Network",
    "Optimal Jamming Frequency Selection for Cognitive Jammer based on Reinforcement Learning",
    "Cognitive Jammer Time Resource Scheduling With Imperfect Information via Fuzzy Q-Learning",
    'A Systematic Literature Review: Is Military Cognitive Radio System on the Brink of the "Valley of Death"?',
    "THE COGNITIVE ELECTRONIC WARFARE IN THE AGE OF ARTIFICIAL INTELLIGENCE",
    "THE USE OF COGNITIVE RADIO TECHNOLOGY TO IMPROVE THE EFFICIENCY OF WIRELESS DATA TRANSMISSION SYSTEMS IN THE CONDITIONS OF ACTIVE USE OF ELECTRONIC WARFARE",
    "Cognitive Jamming-Aided UAV Multi-User Covert Communication",
    "Throughput-Optimized Spectrum Cognizant Routing for Coded Military Cognitive Ad Hoc Radio Networks",
    "Combined Interference and Communications strategy evaluation as a defense mechanism in typical Cognitive Radio Military Networks",
    "Simulation of Cognitive Electronic Warfare System With Sine and Square Waves",
    "Spectrum-Aware Transitive On-Demand Routing Protocol for Military Cognitive Radio Ad Hoc Networks",
    "COGNITIVE RADIO IN THE ELECTRONIC WARFARE",
    "The method of threat assessment and interference distribution within the framework of cognitive electronic warfare",
    "A Survey on Potential Military Application of Cognitive Radio Networks",
    "The use of cognitive electronic warfare is expanding the boundaries",
    "Jammer Versus Radar in a Cognitive Electronic Warfare Environment",
    "Detection of radar pulses of probing signals and determination of their modulation in cognitive electronic warfare systems",
    "Cognitive Electronic Warfare Application to Emitter Identification Process in Airborne Radar Warning Receiver Systems",
    "Game Informed Reinforcement Learning Framework for Radar-Jammer Strategy Optimization in Cognitive Electronic Warfare",
    "Bayesian Non-Parametric Active Learning for Information Asymmetric ECCM in Cognitive Radar Electronic Warfare",
]

# 关键词搜索任务
KEYWORD_TASKS = [
    {"query": 'TAC=("cognitive radar")', "label": "认知雷达", "max_results": 50},
    {"query": 'TAC=("cognitive electronic warfare")', "label": "认知电子战", "max_results": 50},
    {"query": 'TAC=("cognitive jamming")', "label": "认知干扰", "max_results": 50},
]

# 机构搜索任务
ORG_TASKS = [
    {"query": 'AN=("Lockheed Martin") AND TAC=("cognitive radar" OR "cognitive electronic warfare" OR "cognitive jamming")', "label": "Lockheed Martin", "max_results": 20},
    {"query": 'AN=("Raytheon") AND TAC=("cognitive radar" OR "cognitive electronic warfare" OR "cognitive jamming")', "label": "Raytheon/RTX", "max_results": 20},
    {"query": 'AN=("Northrop Grumman") AND TAC=("cognitive radar" OR "cognitive electronic warfare" OR "cognitive jamming")', "label": "Northrop Grumman", "max_results": 20},
    {"query": 'AN=("Secretary of Defense") AND TAC=("cognitive radar" OR "DARPA")', "label": "DARPA", "max_results": 20},
    {"query": 'AN=("Secretary of the Navy") AND TAC=("cognitive radar" OR "cognitive" OR "electronic warfare")', "label": "US Navy/NRL", "max_results": 20},
    {"query": 'AN=("Secretary of the Air Force") AND TAC=("cognitive radar" OR "cognitive" OR "electronic warfare")', "label": "US Air Force/AFRL", "max_results": 20},
]

# ──────────── 工具函数 ────────────

def load_progress():
    """加载进度"""
    if os.path.exists(PROGRESS_FILE):
        with open(PROGRESS_FILE, "r") as f:
            return json.load(f)
    return {"completed_titles": [], "completed_searches": [], "completed_patent_ids": [], "failed": []}


def save_progress(progress):
    """保存进度"""
    with open(PROGRESS_FILE, "w") as f:
        json.dump(progress, f, ensure_ascii=False, indent=2)


def load_results():
    """加载已有结果"""
    if os.path.exists(RESULTS_FILE):
        with open(RESULTS_FILE, "r") as f:
            return json.load(f)
    return []


def save_results(results):
    """保存结果"""
    with open(RESULTS_FILE, "w") as f:
        json.dump(results, f, ensure_ascii=False, indent=2)


def scrapling_fetch(url, output_file, wait=3000):
    """使用 scrapling stealthy-fetch 获取页面"""
    cmd = [
        SCRAPLING_BIN, "extract", "stealthy-fetch",
        url, output_file,
        "--network-idle",
        f"--wait={wait}",
        "--timeout=60000",
    ]
    try:
        result = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
        if result.returncode == 0 and os.path.exists(output_file) and os.path.getsize(output_file) > 100:
            return True
        return False
    except subprocess.TimeoutExpired:
        print(f"    [超时] 采集超时")
        return False
    except Exception as e:
        print(f"    [异常] {e}")
        return False


def download_pdf(pdf_url, patent_id):
    """下载专利 PDF"""
    pdf_dir = os.path.join(OUTPUT_DIR, "pdfs")
    os.makedirs(pdf_dir, exist_ok=True)
    pdf_file = os.path.join(pdf_dir, f"{patent_id}.pdf")
    if os.path.exists(pdf_file) and os.path.getsize(pdf_file) > 1000:
        return pdf_file
    try:
        subprocess.run(
            ["curl", "-sL", pdf_url, "-o", pdf_file],
            timeout=120, capture_output=True,
        )
        if os.path.exists(pdf_file) and os.path.getsize(pdf_file) > 1000:
            return pdf_file
    except:
        pass
    return None


# ──────────── 解析函数 ────────────

def parse_search_results(md_text):
    """从搜索结果页提取专利号列表"""
    patent_ids = []
    pattern = re.compile(r"\[([A-Z]{2}\d+[A-Z]\d?)\]\(https://patentimages")
    for m in pattern.finditer(md_text):
        pid = m.group(1)
        if pid not in patent_ids:
            patent_ids.append(pid)
    return patent_ids


def parse_patent_detail(md_text, patent_url=""):
    """从专利详情页解析17个字段"""
    record = {f: "" for f in PATENT_FIELDS}
    record["来源URL"] = patent_url

    # 英文名称
    title_match = re.search(
        r"^[A-Z]{2}\d+[A-Z]\d?\s*-\s*(.+?)\s*-\s*Google Patents", md_text, re.M
    )
    if title_match:
        record["英文名称"] = title_match.group(1).strip()
    else:
        for m in re.finditer(r"^(.+?)\n={3,}\s*$", md_text, re.M):
            t = m.group(1).strip()
            if len(t) > 10 and t.lower() not in ("patents", "abstract", "claims", "classifications"):
                record["英文名称"] = t
                break

    # 摘要
    abs_match = re.search(
        r"### Abstract translated from\s*\n+(.*?)(?=\n###|\n##|\nImages|\nClassifications|\nClaims|\Z)",
        md_text, re.S,
    )
    if abs_match:
        record["摘要"] = abs_match.group(1).strip().replace("\n", " ")
    else:
        abs_match2 = re.search(
            r"\bAbstract\b\s*\n+(.*?)(?=\n###|\n##|\nImages|\nClassifications|\nClaims|\Z)",
            md_text, re.S,
        )
        if abs_match2:
            record["摘要"] = abs_match2.group(1).strip().replace("\n", " ")

    # 发明人
    inventors = []
    inv_section = re.search(r"Inventor\s*\n((?:\s*:\s*\[.*?\]\(#\)\s*\n)+)", md_text)
    if inv_section:
        inventors = re.findall(r"\[\s*(.*?)\s*\]\(#\)", inv_section.group(1))
    record["发明/设计人"] = "; ".join(inventors)

    # 受让人
    assignee_match = re.search(r"Current Assignee.*?\n\s*:\s*(.+?)(?:\n|$)", md_text)
    if assignee_match:
        record["专利说明"] = f"受让人: {assignee_match.group(1).strip()}"

    # 申请号/日期/法律状态
    app_pattern = re.compile(
        r"Application number:\s*(.+?)\s*\n\s*Filing date:\s*(\d{4}-\d{2}-\d{2})\s*\n\s*Legal status:\s*(.+?)(?:\n|$)",
        re.M,
    )
    apps = []
    for m in app_pattern.finditer(md_text):
        apps.append({"app_number": m.group(1).strip(), "filing_date": m.group(2).strip(), "legal_status": m.group(3).strip()})

    us_apps = [a for a in apps if "US" in a["app_number"].upper()]
    primary_app = us_apps[0] if us_apps else (apps[0] if apps else {})

    if primary_app:
        record["申请/专利号"] = primary_app["app_number"]
        record["专利日期"] = primary_app["filing_date"]
        record["法律状态"] = primary_app["legal_status"]

    # 优先权
    priority_match = re.search(r"Priority date\s*\n(\d{4}-\d{2}-\d{2})", md_text)
    if priority_match:
        record["优先权"] = priority_match.group(1)
    elif apps:
        record["优先权"] = min(a["filing_date"] for a in apps)

    # 公开/公告号
    url_match = re.search(r"/patent/([A-Z]{2}\d+[A-Z]\d?)", patent_url)
    if url_match:
        record["公开/公告号"] = url_match.group(1)
    else:
        pub_match = re.match(r"([A-Z]{2}\d+[A-Z]\d?)", md_text[:200])
        if pub_match:
            record["公开/公告号"] = pub_match.group(1)

    # 公开/公告日
    pid = record.get("公开/公告号", "")
    if pid:
        pub_date_match = re.search(
            r"(\d{4}-\d{2}-\d{2})\s*\n\s*\[Publication of " + re.escape(pid) + r"\]", md_text
        )
        if pub_date_match:
            record["公开/公告日"] = pub_date_match.group(1)
    if not record["公开/公告日"]:
        grant_pub_match = re.search(r"Application granted\s*\n\s*(\d{4}-\d{2}-\d{2})", md_text)
        if grant_pub_match:
            record["公开/公告日"] = grant_pub_match.group(1)
    if not record["公开/公告日"]:
        pub_date_match = re.search(r"Publication [Dd]ate[:\s]*\n\s*(\d{4}-\d{2}-\d{2})", md_text)
        if pub_date_match:
            record["公开/公告日"] = pub_date_match.group(1)

    # 分类号
    cpc_codes = re.findall(r"\[([A-Z]\d{2}[A-Z]\s*\d+/\d+)\]\(#\)", md_text)
    if cpc_codes:
        record["分类号"] = "; ".join(dict.fromkeys(cpc_codes))
        record["主分类号"] = cpc_codes[0]

    # 专利类型
    patent_id = record.get("公开/公告号", "") or record.get("申请/专利号", "")
    if patent_id:
        if "B1" in patent_id or "B2" in patent_id:
            record["专利类型"] = "授权专利"
        elif "A1" in patent_id:
            record["专利类型"] = "发明专利申请公开"
        elif "A2" in patent_id:
            record["专利类型"] = "专利申请"
        else:
            record["专利类型"] = "专利"

    # 专利授权日期
    grant_date_match = re.search(r"(?:Grant date|Date of Patent|Granted)[:\s]*\n\s*(\d{4}-\d{2}-\d{2})", md_text)
    if grant_date_match:
        record["专利授权日期"] = grant_date_match.group(1)
    elif record["专利类型"] == "授权专利":
        if record["公开/公告日"]:
            record["专利授权日期"] = record["公开/公告日"]
        else:
            grant_match = re.search(r"Application granted\s*\n\s*(\d{4}-\d{2}-\d{2})", md_text)
            if grant_match:
                record["专利授权日期"] = grant_match.group(1)

    # 所属国家
    if patent_id:
        cc = re.match(r"([A-Z]{2})", patent_id)
        if cc:
            cmap = {
                "US": "美国", "EP": "欧洲", "WO": "世界知识产权组织",
                "GB": "英国", "DE": "德国", "FR": "法国", "JP": "日本",
                "CN": "中国", "KR": "韩国", "CA": "加拿大", "AU": "澳大利亚",
            }
            record["所属国家"] = cmap.get(cc.group(1), cc.group(1))
    if not record["所属国家"] and us_apps:
        record["所属国家"] = "美国"

    # PDF URL
    pdf_match = re.search(
        r"https://patentimages\.storage\.googleapis\.com/[a-f0-9/]+/[A-Z]{2}\d+\.pdf", md_text
    )
    if pdf_match:
        record["PDF_URL"] = pdf_match.group(0)

    return record


# ──────────── 采集流程 ────────────

def search_patents(query, max_results=10, page=0):
    """搜索 Google Patents，返回专利号列表"""
    encoded = urllib.parse.quote(query)
    url = f"https://patents.google.com/?q={encoded}&page={page}"
    search_file = os.path.join(OUTPUT_DIR, f"search_{hash(query) % 100000}.md")

    if not scrapling_fetch(url, search_file, wait=5000):
        return []

    with open(search_file, "r", encoding="utf-8") as f:
        md = f.read()

    patent_ids = parse_search_results(md)

    # 清理
    if os.path.exists(search_file):
        os.remove(search_file)

    return patent_ids[:max_results]


def collect_patent_detail(patent_id):
    """采集单个专利详情"""
    url = f"https://patents.google.com/patent/{patent_id}/en"
    detail_file = os.path.join(OUTPUT_DIR, f"detail_{patent_id}.md")

    if not scrapling_fetch(url, detail_file, wait=3000):
        return None

    with open(detail_file, "r", encoding="utf-8") as f:
        md = f.read()

    record = parse_patent_detail(md, url)

    if not record.get("英文名称"):
        if os.path.exists(detail_file):
            os.remove(detail_file)
        return None

    # 下载 PDF
    if record.get("PDF_URL"):
        pdf_path = download_pdf(record["PDF_URL"], patent_id)
        if pdf_path:
            record["PDF本地路径"] = pdf_path

    # 清理临时文件
    if os.path.exists(detail_file):
        os.remove(detail_file)

    return record


def process_title(title, progress, results):
    """处理单个标题搜索"""
    if title in progress["completed_titles"]:
        return

    print(f"\n[标题搜索] {title[:60]}...")

    # 先用 TI= 精确搜索
    patent_ids = search_patents(f'TI=("{title}")', max_results=3)

    # 如果没结果，用 TAC= 扩大搜索
    if not patent_ids:
        # 提取关键短语（去掉冒号后面的副标题）
        short_title = title.split(":")[0].strip()
        if len(short_title) < len(title):
            patent_ids = search_patents(f'TAC=("{short_title}")', max_results=5)

    # 再用关键词组合搜索
    if not patent_ids:
        keywords = re.findall(r'\b(?:cognitive|radar|jamming|electronic warfare|anti-jamming|spectrum)\b', title, re.I)
        if len(keywords) >= 2:
            kw_query = " AND ".join(f'TAC=("{kw}")' for kw in keywords[:3])
            patent_ids = search_patents(kw_query, max_results=5)

    if not patent_ids:
        print(f"  [无结果] 未找到相关专利")
        progress["failed"].append({"type": "title", "query": title, "reason": "无搜索结果"})
        progress["completed_titles"].append(title)
        save_progress(progress)
        return

    print(f"  找到 {len(patent_ids)} 个相关专利: {patent_ids[:3]}")

    # 采集第一个（最相关的）专利详情
    for pid in patent_ids[:1]:  # 只取最相关的
        if pid in progress["completed_patent_ids"]:
            print(f"  [跳过] {pid} 已采集")
            continue

        record = collect_patent_detail(pid)
        if record:
            record["搜索标题"] = title
            record["采集类型"] = "标题搜索"
            results.append(record)
            progress["completed_patent_ids"].append(pid)
            print(f"  [✓] {pid}: {record.get('英文名称', '?')[:50]}")
        else:
            progress["failed"].append({"type": "patent", "patent_id": pid, "reason": "详情采集失败"})

        time.sleep(2)

    progress["completed_titles"].append(title)
    save_progress(progress)
    save_results(results)

    time.sleep(3)


def process_keyword_search(task, progress, results):
    """处理关键词搜索"""
    label = task["label"]
    if label in progress["completed_searches"]:
        return

    print(f"\n[关键词搜索] {label}: {task['query'][:60]}...")

    patent_ids = search_patents(task["query"], max_results=task.get("max_results", 20))

    if not patent_ids:
        print(f"  [无结果]")
        progress["failed"].append({"type": "keyword", "label": label, "reason": "无搜索结果"})
        progress["completed_searches"].append(label)
        save_progress(progress)
        return

    print(f"  找到 {len(patent_ids)} 个专利: {patent_ids[:5]}...")

    for pid in patent_ids:
        if pid in progress["completed_patent_ids"]:
            continue

        record = collect_patent_detail(pid)
        if record:
            record["搜索标签"] = label
            record["采集类型"] = "关键词搜索"
            results.append(record)
            progress["completed_patent_ids"].append(pid)
            print(f"  [✓] {pid}: {record.get('英文名称', '?')[:50]}")
        else:
            progress["failed"].append({"type": "patent", "patent_id": pid, "reason": "详情采集失败"})

        time.sleep(2)
        save_results(results)

    progress["completed_searches"].append(label)
    save_progress(progress)
    time.sleep(3)


def process_org_search(task, progress, results):
    """处理机构搜索"""
    label = task["label"]
    key = f"ORG:{label}"
    if key in progress["completed_searches"]:
        return

    print(f"\n[机构搜索] {label}: {task['query'][:60]}...")

    patent_ids = search_patents(task["query"], max_results=task.get("max_results", 20))

    if not patent_ids:
        print(f"  [无结果]")
        progress["failed"].append({"type": "org", "label": label, "reason": "无搜索结果"})
        progress["completed_searches"].append(key)
        save_progress(progress)
        return

    print(f"  找到 {len(patent_ids)} 个专利: {patent_ids[:5]}...")

    for pid in patent_ids:
        if pid in progress["completed_patent_ids"]:
            continue

        record = collect_patent_detail(pid)
        if record:
            record["搜索标签"] = label
            record["采集类型"] = "机构搜索"
            results.append(record)
            progress["completed_patent_ids"].append(pid)
            print(f"  [✓] {pid}: {record.get('英文名称', '?')[:50]}")
        else:
            progress["failed"].append({"type": "patent", "patent_id": pid, "reason": "详情采集失败"})

        time.sleep(2)
        save_results(results)

    progress["completed_searches"].append(key)
    save_progress(progress)
    time.sleep(3)


def main():
    os.makedirs(OUTPUT_DIR, exist_ok=True)
    os.makedirs(os.path.join(OUTPUT_DIR, "pdfs"), exist_ok=True)

    progress = load_progress()
    results = load_results()

    total = len(TITLE_TASKS) + len(KEYWORD_TASKS) + len(ORG_TASKS)
    done = len(progress["completed_titles"]) + len(progress["completed_searches"])

    print(f"{'='*60}")
    print(f"认知电子战专利批量采集")
    print(f"总任务: {total} | 已完成: {done} | 待处理: {total - done}")
    print(f"已采集专利: {len(results)} | 失败: {len(progress['failed'])}")
    print(f"{'='*60}")

    # Phase 1: 标题搜索（54项）
    print(f"\n{'='*60}")
    print(f"Phase 1: 标题搜索 ({len(TITLE_TASKS)} 项)")
    print(f"{'='*60}")
    for i, title in enumerate(TITLE_TASKS, 1):
        if title in progress["completed_titles"]:
            continue
        print(f"\n--- [{i}/{len(TITLE_TASKS)}] ---")
        process_title(title, progress, results)

    # Phase 2: 关键词搜索（3项）
    print(f"\n{'='*60}")
    print(f"Phase 2: 关键词搜索 ({len(KEYWORD_TASKS)} 项)")
    print(f"{'='*60}")
    for task in KEYWORD_TASKS:
        process_keyword_search(task, progress, results)

    # Phase 3: 机构搜索（6项）
    print(f"\n{'='*60}")
    print(f"Phase 3: 机构搜索 ({len(ORG_TASKS)} 项)")
    print(f"{'='*60}")
    for task in ORG_TASKS:
        process_org_search(task, progress, results)

    # 最终保存
    save_results(results)

    # 生成 CSV
    import csv
    csv_file = os.path.join(OUTPUT_DIR, "patents.csv")
    fields = PATENT_FIELDS + ["来源URL", "PDF_URL", "PDF本地路径", "搜索标题", "搜索标签", "采集类型"]
    with open(csv_file, "w", newline="", encoding="utf-8-sig") as f:
        writer = csv.DictWriter(f, fieldnames=fields)
        writer.writeheader()
        for r in results:
            writer.writerow({k: r.get(k, "") for k in fields})

    # 生成数据护照
    passport = {
        "合规性声明": "所有数据均从 Google Patents 公开页面采集，遵守 robots.txt 协议",
        "采集时间": time.strftime("%Y-%m-%d %H:%M:%S %Z"),
        "数据来源": "Google Patents (patents.google.com)",
        "采集工具": "开源情报-专利采集器 / Scrapling stealthy-fetch",
        "采集范围": f"共采集 {len(results)} 条专利记录",
        "标题搜索": f"{len(TITLE_TASKS)} 项",
        "关键词搜索": f"{len(KEYWORD_TASKS)} 项",
        "机构搜索": f"{len(ORG_TASKS)} 项",
        "失败记录数": len(progress["failed"]),
        "字段覆盖度": {},
    }
    for field in PATENT_FIELDS:
        filled = sum(1 for r in results if r.get(field))
        passport["字段覆盖度"][field] = f"{filled}/{len(results)} ({filled/max(len(results),1)*100:.0f}%)"

    with open(os.path.join(OUTPUT_DIR, "data_passport.json"), "w") as f:
        json.dump(passport, f, ensure_ascii=False, indent=2)

    print(f"\n{'='*60}")
    print(f"采集完成！")
    print(f"  成功: {len(results)} 条专利")
    print(f"  失败: {len(progress['failed'])} 条")
    print(f"  PDF: {len([r for r in results if r.get('PDF本地路径')])} 个")
    print(f"  JSON: {RESULTS_FILE}")
    print(f"  CSV: {csv_file}")
    print(f"{'='*60}")


if __name__ == "__main__":
    main()
