#!/bin/bash
# ========================================
# 工具包打包脚本 - 用于离线部署
# 创建时间: 2026-03-31
# 开发者: 瞰宇
# ========================================

set -e

COLOR_GREEN='\033[0;32m'
COLOR_COLOR_YELLOW='\033[1;33m'
COLOR_RED='\033[0;31m'
COLOR_BLUE='\033[0;34m'
COLOR_RESET='\033[0m'

echo -e "${COLOR_BLUE}========================================${COLOR_RESET}"
echo -e "${COLOR_BLUE}  工具包打包程序${COLOR_RESET}"
echo -e "${COLOR_BLUE}========================================${COLOR_RESET}"

WORKSPACE="/root/.openclaw/workspace"
PACKAGE_DIR="${WORKSPACE}/toolkits-deployment-package"
PYTHON_ENV="/root/miniconda3/envs/data-collector"

# 清理并创建打包目录
echo -e "\n${COLOR_YELLOW}[1/7] 清理并创建打包目录...${COLOR_RESET}"
rm -rf "${PACKAGE_DIR}"
mkdir -p "${PACKAGE_DIR}/requirements"
mkdir -p "${PACKAGE_DIR}/tools"
mkdir -p "${PACKAGE_DIR}/scripts"
echo -e "${COLOR_GREEN}✓ 打包目录创建完成${COLOR_RESET}"

# 1. 导出 Python 包为 wheel 文件
echo -e "\n${COLOR_YELLOW}[2/7] 导出 Python 包...${COLOR_RESET}"
${PYTHON_ENV}/bin/pip download \
    -d "${PACKAGE_DIR}/requirements" \
    --no-deps \
    crawl4ai botasaurus twscrape scrapling \
    requests httpx aiohttp lxml beautifulsoup4 \
    pandas openpyxl pyyaml

# 同时下载依赖包
${PYTHON_ENV}/bin/pip download \
    -d "${PACKAGE_DIR}/requirements" \
    crawl4ai botasaurus twscrape scrapling

echo -e "${COLOR_GREEN}✓ Python 包导出完成${COLOR_RESET}"

# 2. 复制 OSINT 工具目录
echo -e "\n${COLOR_YELLOW}[3/7] 复制 OSINT 工具...${COLOR_RESET}"
if [ -d "${WORKSPACE}/tools/OSINT-TOOLS-2025" ]; then
    cp -r "${WORKSPACE}/tools/OSINT-TOOLS-2025" "${PACKAGE_DIR}/tools/"
    echo -e "${COLOR_GREEN}✓ OSINT-TOOLS-2025 已复制${COLOR_RESET}"
else
    echo -e "${COLOR_RED}✗ OSINT-TOOLS-2025 未找到${COLOR_RESET}"
    exit 1
fi

if [ -d "${WORKSPACE}/tools/OSINT-for-countries-V2.0" ]; then
    cp -r "${WORKSPACE}/tools/OSINT-for-countries-V2.0" "${PACKAGE_DIR}/tools/"
    echo -e "${COLOR_GREEN}✓ OSINT-for-countries-V2.0 已复制${COLOR_RESET}"
else
    echo -e "${COLOR_RED}✗ OSINT-for-countries-V2.0 未找到${COLOR_RESET}"
    exit 1
fi

# 3. 生成 requirements.txt
echo -e "\n${COLOR_YELLOW}[4/7] 生成 requirements.txt...${COLOR_RESET}"
cat > "${PACKAGE_DIR}/requirements.txt" << 'EOF'
# 核心数据采集工具
crawl4ai
botasaurus
twscrape
scrapling

# 依赖库
requests
httpx
aiohttp
lxml
beautifulsoup4
pandas
openpyxl
pyyaml
EOF
echo -e "${COLOR_GREEN}✓ requirements.txt 已生成${COLOR_RESET}"

# 4. 创建自动化安装脚本
echo -e "\n${COLOR_YELLOW}[5/7] 创建自动化安装脚本...${COLOR_RESET}"
cat > "${PACKAGE_DIR}/install.sh" << 'INSTALL_SCRIPT'
#!/bin/bash
# ========================================
# 工具包自动化安装脚本
# 适用环境: Linux/macOS (需已安装 Miniconda/Anaconda)
# ========================================

set -e

COLOR_GREEN='\033[0;32m'
COLOR_YELLOW='\033[1;33m'
COLOR_RED='\033[0;31m'
COLOR_BLUE='\033[0;34m'
COLOR_RESET='\033[0m'

echo -e "${COLOR_BLUE}========================================${COLOR_RESET}"
echo -e "${COLOR_BLUE}  工具包自动化安装程序${COLOR_RESET}"
echo -e "${COLOR_BLUE}========================================${COLOR_RESET}"

# 检查是否在正确的目录
if [ ! -f "requirements.txt" ] || [ ! -d "tools" ] || [ ! -d "requirements" ]; then
    echo -e "${COLOR_RED}错误: 请在工具包根目录运行此脚本${COLOR_RESET}"
    echo -e "${COLOR_RED}当前目录应包含: requirements.txt, tools/, requirements/${COLOR_RESET}"
    exit 1
fi

# 询问安装路径
echo -e "\n${COLOR_YELLOW}请输入 Miniconda/Anaconda 安装路径 [默认: ${HOME}/miniconda3]:${COLOR_RESET}"
read CONDA_PATH
CONDA_PATH=${CONDA_PATH:-"${HOME}/miniconda3"}

if [ ! -f "${CONDA_PATH}/bin/conda" ]; then
    echo -e "${COLOR_RED}错误: 未找到 conda: ${CONDA_PATH}/bin/conda${COLOR_RESET}"
    echo -e "${COLOR_YELLOW}请确认 Miniconda/Anaconda 已正确安装${COLOR_RESET}"
    exit 1
fi

# 询问环境名称
echo -e "\n${COLOR_YELLOW}请输入 Python 环境名称 [默认: data-collector]:${COLOR_RESET}"
read ENV_NAME
ENV_NAME=${ENV_NAME:-"data-collector"}

# 创建或激活环境
echo -e "\n${COLOR_YELLOW}创建/激活 Python 环境...${COLOR_RESET}"
source "${CONDA_PATH}/bin/activate"

if conda env list | grep -q "^${ENV_NAME} "; then
    echo -e "${COLOR_GREEN}环境已存在: ${ENV_NAME}${COLOR_RESET}"
else
    echo -e "${COLOR_YELLOW}创建新环境: ${ENV_NAME}${COLOR_RESET}"
    conda create -n "${ENV_NAME}" python=3.11 -y
fi

conda activate "${ENV_NAME}"
echo -e "${COLOR_GREEN}✓ 环境已激活: ${ENV_NAME}${COLOR_RESET}"

# 离线安装 Python 包
echo -e "\n${COLOR_YELLOW}离线安装 Python 包...${COLOR_RESET}"
pip install --no-index --find-links=requirements -r requirements.txt

# 显示已安装的包
echo -e "\n${COLOR_GREEN}已安装的包:${COLOR_RESET}"
pip list | grep -E "(crawl4ai|botasaurus|twscrape|scrapling)"

# 复制工具目录
echo -e "\n${COLOR_YELLOW}安装 OSINT 工具...${COLOR_RESET}"
WORKSPACE="${HOME}/.openclaw/workspace"
mkdir -p "${WORKSPACE}/tools"

if [ -d "tools/OSINT-TOOLS-2025" ]; then
    cp -r "tools/OSINT-TOOLS-2025" "${WORKSPACE}/tools/"
    echo -e "${COLOR_GREEN}✓ OSINT-TOOLS-2025 已安装${COLOR_RESET}"
fi

if [ -d "tools/OSINT-for-countries-V2.0" ]; then
    cp -r "tools/OSINT-for-countries-V2.0" "${WORKSPACE}/tools/"
    echo -e "${COLOR_GREEN}✓ OSINT-for-countries-V2.0 已安装${COLOR_RESET}"
fi

# 创建快捷命令脚本
echo -e "\n${COLOR_YELLOW}创建快捷命令...${COLOR_RESET}"
mkdir -p "${WORKSPACE}/bin"

# Crawl4AI 快捷脚本
cat > "${WORKSPACE}/bin/crawl4ai-crawl" << 'CRAWL4AI_SCRIPT'
#!/bin/bash
source "${HOME}/miniconda3/bin/activate"
conda activate data-collector
python -m crawl4ai "$@"
CRAWL4AI_SCRIPT
chmod +x "${WORKSPACE}/bin/crawl4ai-crawl"

# Botasaurus 快捷脚本
cat > "${WORKSPACE}/bin/botasaurus" << 'BOTASAURUS_SCRIPT'
#!/bin/bash
source "${HOME}/miniconda3/bin/activate"
conda activate data-collector
python -c "from botasaurus import *; print('Botasaurus 已就绪')" "$@"
BOTASAURUS_SCRIPT
chmod +x "${WORKSPACE}/bin/botasaurus"

# Twscrape 快捷脚本
cat > "${WORKSPACE}/bin/twscrape" << 'TWSCRAPE_SCRIPT'
#!/bin/bash
source "${HOME}/miniconda3/bin/activate"
conda activate data-collector
python -m twscrape "$@"
TWSCRAPE_SCRIPT
chmod +x "${WORKSPACE}/bin/twscrape"

# Scrapling 快捷脚本
cat > "${WORKSPACE}/bin/scrapling" << 'SCRAPLING_SCRIPT'
#!/bin/bash
source "${HOME}/miniconda3/bin/activate"
conda activate data-collector
scrapling "$@"
SCRAPLING_SCRIPT
chmod +x "${WORKSPACE}/bin/scrapling"

# 添加到 PATH
if ! grep -q "${WORKSPACE}/bin" ~/.bashrc 2>/dev/null; then
    echo "export PATH=\"${WORKSPACE}/bin:\$PATH\"" >> ~/.bashrc
    echo -e "${COLOR_GREEN}✓ 已添加到 PATH (需重新加载 ~/.bashrc)${COLOR_RESET}"
fi

echo -e "\n${COLOR_GREEN}========================================${COLOR_RESET}"
echo -e "${COLOR_GREEN}  安装完成！${COLOR_RESET}"
echo -e "${COLOR_GREEN}========================================${COLOR_RESET}"
echo -e "${COLOR_YELLOW}使用方法:${COLOR_RESET}"
echo -e "  ${COLOR_GREEN}crawl4ai-crawl${COLOR_RESET}     - Crawl4AI 命令"
echo -e "  ${COLOR_GREEN}twscrape${COLOR_RESET}           - Twscrape 命令"
echo -e "  ${COLOR_GREEN}scrapling${COLOR_RESET}          - Scrapling 命令"
echo -e "\n${COLOR_YELLOW}Python 调用示例:${COLOR_RESET}"
echo -e "  ${COLOR_BLUE}conda activate ${ENV_NAME}${COLOR_RESET}"
echo -e "  ${COLOR_BLUE}python -c 'import crawl4ai; print(crawl4ai.__version__) '${COLOR_RESET}"
echo -e "  ${COLOR_BLUE}python -c 'import botasaurus; print(botasaurus.__version__) '${COLOR_RESET}"
echo -e "  ${COLOR_BLUE}python -c 'import twscrape; print(twscrape.__version__) '${COLOR_RESET}"
echo -e "  ${COLOR_BLUE}python -c 'import scrapling; print(scrapling.__version__) '${COLOR_RESET}"
INSTALL_SCRIPT

chmod +x "${PACKAGE_DIR}/install.sh"
echo -e "${COLOR_GREEN}✓ 安装脚本已创建${COLOR_RESET}"

# 5. 创建 README 文档
echo -e "\n${COLOR_YELLOW}[6/7] 创建 README 文档...${COLOR_RESET}"
cat > "${PACKAGE_DIR}/README.md" << 'README_EOF'
# 全球数据采集工具包 - 离线部署包

> **创建时间:** 2026-03-31
> **开发者:** 瞰宇
> **版本:** 1.0.0

---

## 包含工具

### 1. Crawl4AI (0.8.6+)
- **用途:** AI 驱动的网页爬虫，LLM 友好的 Markdown 输出
- **核心特性:** 自动反机器人检测、Shadow DOM 展平、专用 RAG/知识库
- **Python 调用:** `import crawl4ai`

### 2. Botasaurus (4.0.97)
- **用途:** 高级反检测爬虫，绕过 Cloudflare WAF/Turnstile
- **核心特性:** 真实人类鼠标模拟、自动代理轮换
- **Python 调用:** `from botasaurus import *`

### 3. Twscrape (0.17.0)
- **用途:** Twitter/X 爬虫，支持 GraphQL + Search API
- **核心特性:** 异步并行、自动账户切换、速率限制管理
- **Python 调用:** `import twscrape`

### 4. Scrapling (0.4.2)
- **用途:** 综合反爬虫框架，支持反机器人绕过
- **核心特性:** 自适应抓取、爬虫框架、MCP 服务器
- **Python 调用:** `import scrapling`
- **CLI 命令:** `scrapling get / fetch / stealthy-fetch`

### 5. OSINT-TOOLS-2025
- **用途:** 开源情报工具集合
- **包含:** 域名分析、身份搜索、地理定位、图像分析、社交媒体分析

### 6. OSINT-for-countries-V2.0
- **用途:** 国家标准化 OSINT 资源库
- **覆盖:** 28+ 国家/地区的标准化资源

---

## 快速安装

### 前置要求
- Linux/macOS 系统
- 已安装 Miniconda 或 Anaconda

### 一键安装

1. **解压工具包**

```bash
tar -xzf toolkits-deployment-package.tar.gz
cd toolkits-deployment-package
```

2. **运行安装脚本**

```bash
chmod +x install.sh
./install.sh
```

3. **激活环境**

```bash
source ~/miniconda3/bin/activate
conda activate data-collector
```

4. **验证安装**

```bash
python -c "import crawl4ai; print('Crawl4AI:', crawl4ai.__version__)"
python -c "import botasaurus; print('Botasaurus:', botasaurus.__version__)"
python -c "import twscrape; print('Twscrape:', twscrape.__version__)"
python -c "import scrapling; print('Scrapling:', scrapling.__version__)"
```

---

## 使用示例

### Crawl4AI - 爬取网页为 Markdown

```python
import asyncio
from crawl4ai import AsyncWebCrawler

async def crawl_example():
    async with AsyncWebCrawler() as crawler:
        result = await crawler.arun(url="https://example.com")
        print(result.markdown)

asyncio.run(crawl_example())
```

### Botasaurus - 绕过 Cloudflare

```python
from botasaurus import *

@browser(
    headless=True,
    proxy="http://user:pass@proxy:port"
)
def scrape_data(driver):
    driver.get("https://example.com")
    return driver.page_source

result = scrape_data()
```

### Twscrape - Twitter 数据采集

```bash
# 添加账户
twscrape accounts add

# 搜索推文
twscrape search "关键词" --limit 100

# 异步 Python 调用
import asyncio
from twscrape import AccountsPool, Scraper

async def twitter_example():
    pool = AccountsPool()
    pool.load_from_db()
    scraper = Scraper(pool)
    async for tweet in scraper.search("关键词"):
        print(tweet)

asyncio.run(twitter_example())
```

### Scrapling - CLI 快速抓取

```bash
# 简单 GET 请求
scrapling extract get https://example.com

# 浏览器动态获取
scrapling extract fetch https://example.com

# 隐蔽模式（绕过 Cloudflare）
scrapling extract stealthy-fetch https://example.com --css-selector ".content"
```

---

## 目录结构

```
toolkits-deployment-package/
├── requirements/           # 离线 wheel 包
├── requirements.txt       # 依赖列表
├── tools/                 # OSINT 工具目录
│   ├── OSINT-TOOLS-2025/
│   └── OSINT-for-countries-V2.0/
├── install.sh             # 自动化安装脚本
└── README.md              # 本文档
```

---

## 技术支持

如遇问题，请检查：
1. Python 版本：建议 3.11+
2. 网络连接：首次安装可能需要下载浏览器驱动
3. 依赖冲突：如遇冲突，建议在全新环境中安装

---

**开发者:** 瞻宇 (Kàn Yǔ)
**定位:** 全球数据采集师 · 认知战研究专家 · 合规之眼
README_EOF

echo -e "${COLOR_GREEN}✓ README.md 已创建${COLOR_RESET}"

# 6. 打包为 tar.gz
echo -e "\n${COLOR_YELLOW}[7/7] 打包为 tar.gz...${COLOR_RESET}"
tar -czf "${WORKSPACE}/toolkits-deployment-package.tar.gz" -C "${WORKSPACE}" toolkits-deployment-package

PACKAGE_SIZE=$(du -h "${WORKSPACE}/toolkits-deployment-package.tar.gz" | cut -f1)
echo -e "${COLOR_GREEN}✓ 打包完成: ${WORKSPACE}/toolkits-deployment-package.tar.gz (${PACKAGE_SIZE})${COLOR_RESET}"

# 完成
echo -e "\n${COLOR_GREEN}========================================${COLOR_RESET}"
echo -e "${COLOR_GREEN}  打包完成！${COLOR_RESET}"
echo -e "${COLOR_GREEN}========================================${COLOR_RESET}"
echo -e "\n${COLOR_YELLOW}文件位置:${COLOR_RESET} ${WORKSPACE}/toolkits-deployment-package.tar.gz"
echo -e "${COLOR_YELLOW}文件大小:${COLOR_RESET} ${PACKAGE_SIZE}"
echo -e "\n${COLOR_YELLOW}部署方法:${COLOR_RESET}"
echo -e "  1. 复制文件到目标机器"
echo -e "  2. 解压: tar -xzf toolkits-deployment-package.tar.gz"
echo -e "  3. 进入目录: cd toolkits-deployment-package"
echo -e "  4. 运行安装: ./install.sh"
echo -e "\n${COLOR_GREEN}所有工具已准备就绪，可离线部署！${COLOR_RESET}"
