#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
公文段落格式统一修复脚本（通用版）
=====================================

适用范围：所有报告输出类技能产出的 docx 文档

修复规则（严格按《公文段落格式统一规范》）：
1. 全文正文段落：left_indent=None, first_line_indent=406400 EMU (2字符)
2. 去除段首项目符号：· • ・ - * 等
3. 清除 numPr (Word 列表样式标记)
4. 保留参考文献的悬挂缩进（仅"参考文献"/"附录"章节内）
5. 章节标题（"第N章 XXX"）保持原样

用法：
    python fix_paragraph_format.py <docx文件路径> [文件路径2 ...]
    python fix_paragraph_format.py <目录路径>  # 处理目录下所有 docx

返回：每个文件修复的段落数
"""
import os
import re
import sys
import glob
from docx import Document
from docx.shared import Emu

TARGET_FIRST_LINE_INDENT = Emu(406400)  # 2 字符首行缩进
WORDML_NS = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"

# 项目符号检测正则
BULLET_RE = re.compile(r"^[·•・\u2022\u2027\-*]+\s*")
# 章节标题检测
CHAPTER_TITLE_RE = re.compile(r"^第[一二三四五六七八九十]+章")
# 附录/参考文献起点
APPENDIX_RE = re.compile(r"^(参考文献|附录|附件)")


def is_appendix_start(text):
    return bool(APPENDIX_RE.match(text.strip()))


def is_chapter_title(text):
    return bool(CHAPTER_TITLE_RE.match(text.strip()))


def fix_paragraph(p):
    """修复单个段落。返回 True 表示有修改。"""
    if not p.runs:
        return False
    changed = False

    # 1) 去掉第一个非空 run 起始的 bullet 符号
    for run in p.runs:
        if run.text and run.text.strip():
            new_text = BULLET_RE.sub("", run.text)
            if new_text != run.text:
                run.text = new_text
                changed = True
            break

    # 2) 重置段落缩进
    pf = p.paragraph_format
    if pf.left_indent is not None:
        pf.left_indent = None
        changed = True
    if pf.first_line_indent != TARGET_FIRST_LINE_INDENT:
        pf.first_line_indent = TARGET_FIRST_LINE_INDENT
        changed = True

    # 3) 清理 numPr（列表样式标记）
    pPr = p._p.find(f"{WORDML_NS}pPr")
    if pPr is not None:
        numPr = pPr.find(f"{WORDML_NS}numPr")
        if numPr is not None:
            pPr.remove(numPr)
            changed = True
    return changed


def process_file(path):
    """处理单个 docx 文件，返回修复段落数。"""
    try:
        doc = Document(path)
    except Exception as e:
        print(f"  ❌ {os.path.basename(path)} - 打开失败: {e}")
        return 0

    in_appendix = False
    fixed_count = 0
    for p in doc.paragraphs:
        t = p.text.strip()
        if not t:
            continue

        # 检测进入附录区
        if is_appendix_start(t):
            in_appendix = True
            continue

        # 附录内（参考文献等）保持原格式
        if in_appendix:
            continue

        # 跳过章节标题本身
        if is_chapter_title(t):
            continue

        # 修复正文段落
        if fix_paragraph(p):
            fixed_count += 1

    doc.save(path)
    return fixed_count


def collect_files(paths):
    """收集所有 docx 文件路径。"""
    files = []
    for path in paths:
        if os.path.isdir(path):
            files.extend(sorted(glob.glob(os.path.join(path, "**", "*.docx"),
                                          recursive=True)))
        elif os.path.isfile(path) and path.endswith(".docx"):
            files.append(path)
        else:
            print(f"  ⚠️ 跳过无效路径: {path}")
    return files


def main():
    if len(sys.argv) < 2:
        print(__doc__)
        sys.exit(1)

    files = collect_files(sys.argv[1:])
    if not files:
        print("未找到任何 docx 文件")
        sys.exit(1)

    print(f"找到 {len(files)} 个文件待修复\n")
    total = 0
    for f in files:
        n = process_file(f)
        print(f"  ✅ {os.path.basename(f)} - 修复 {n} 个段落")
        total += n
    print(f"\n全部完成，合计修复 {total} 个段落")


if __name__ == "__main__":
    main()
