#!/usr/bin/env python3
"""
专利信息采集器
采集全球公开专利信息、专利申请数据
"""

import json
import sys
from datetime import datetime
from typing import Dict, List, Any

class PatentsCollector:
    """专利信息采集"""

    # 公开专利数据库
    PATENT_DATABASES = {
        "USPTO": [
            "patents.google.com",         # Google Patents（聚合多国专利）
            "uspto.gov"                    # 美国专利商标局公开数据库
        ],
        "WIPO": [
            "patentscope.wipo.int"        # WIPO PATENTSCOPE
        ],
        "EPO": [
            "worldwide.espacenet.com"     # 欧洲专利局 Espacenet
        ]
    }

    def __init__(self):
        self.collection_log = {
            "start_time": datetime.now().isoformat(),
            "collector": "PatentsCollector",
            "patents_collected": 0,
            "databases_accessed": []
        }

    async def collect_patents(self,
                              keywords: List[str],
                              countries: List[str] = None,
                              assignees: List[str] = None,
                              year_range: tuple = None,
                              max_results: int = 50) -> Dict[str, Any]:
        """
        采集专利信息

        Args:
            keywords: 技术关键词
            countries: 目标国家代码（如["US", "CN", "JP"]）
            assignees: 专利权人/公司名称
            year_range: 申请年份范围 (start, end)
            max_results: 最大结果数

        Returns:
            结构化专利数据
        """
        self.collection_log["keywords"] = keywords
        self.collection_log["countries"] = countries or []
        self.collection_log["assignees"] = assignees or []
        self.collection_log["year_range"] = year_range

        results = []

        # 使用 Google Patents（最全面的公开来源）
        base_query = " ".join(keywords)

        for keyword in keywords:
            query_parts = [keyword]

            if countries:
                country_filter = " OR ".join([f"({c})" for c in countries])
                query_parts.append(country_filter)

            if assignees:
                assignee_filter = " OR ".join([f"assignee:({a})" for a in assignees])
                query_parts.append(assignee_filter)

            query = " ".join(query_parts)

            self.collection_log["databases_accessed"].append({
                "database": "Google Patents",
                "query": query,
                "timestamp": datetime.now().isoformat()
            })

        self.collection_log["end_time"] = datetime.now().isoformat()
        self.collection_log["patents_collected"] = len(results)

        return {
            "patent_items": results,
            "metadata": {
                "type": "patents",
                "count": len(results),
                "countries": countries or [],
                "assignees": assignees or []
            },
            "collection_log": self.collection_log
        }

    def extract_patent_metadata(self, content: str, url: str) -> Dict[str, Any]:
        """提取专利元数据"""
        return {
            "patent_number": self._extract_patent_number(url),
            "title": self._extract_title(content),
            "abstract": self._extract_abstract(content),
            "assignees": self._extract_assignees(content),
            "inventors": self._extract_inventors(content),
            "filing_date": "",
            "publication_date": "",
            "status": "",
            "url": url
        }

    def _extract_patent_number(self, url: str) -> str:
        """从URL提取专利号"""
        # Google Patents URL 格式：/patent/CN123456789A/en
        if "/patent/" in url:
            start = url.find("/patent/") + 9
            end = url.find("/", start)
            if end > start:
                return url[start:end]
        return "Unknown"

    def _extract_title(self, content: str) -> str:
        """提取标题"""
        if "<title>" in content:
            start = content.find("<title>") + 7
            end = content.find("</title>", start)
            if end > start:
                return content[start:end].strip()
        return "Unknown"

    def _extract_abstract(self, content: str) -> str:
        """提取摘要"""
        # Google Patents 摘要通常在特定标签中
        return content[:500] + "..." if len(content) > 500 else content

    def _extract_assignees(self, content: str) -> List[str]:
        """提取专利权人"""
        return []

    def _extract_inventors(self, content: str) -> List[str]:
        """提取发明人"""
        return []

def main():
    print("PatentsCollector v1.0")
    print("公开数据库：USPTO, WIPO, EPO")
    print("优先使用：Google Patents（聚合源）")

if __name__ == "__main__":
    main()
