"""文档静态复核：链接、slug、表格、冲突标记、凭据与 Python 语法；不写文件。"""
import argparse
import ast
import collections
import pathlib
import re
import urllib.parse

import yaml


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--workspace", type=pathlib.Path, required=True)
    args = parser.parse_args()
    root = args.workspace.resolve()
    docs = root / "doc_code/docs-develop/database-design"
    errors = []
    slugs = collections.defaultdict(list)
    files = sorted(docs.rglob("*.md"))
    for file in files:
        text = file.read_text(encoding="utf-8")
        relative = file.relative_to(docs).as_posix()
        if "\ufffd" in text:
            errors.append((relative, "unicode_replacement_character"))
        match = re.match(r"\A---\r?\n(.*?)\r?\n---", text, re.S)
        if not match:
            errors.append((relative, "missing_frontmatter"))
        else:
            metadata = yaml.safe_load(match.group(1))
            slug = metadata.get("slug")
            if not slug or not metadata.get("title"):
                errors.append((relative, "missing_title_or_slug"))
            else:
                slugs[slug].append(relative)
        in_fence = False
        previous_columns = None
        for line_number, line in enumerate(text.splitlines(), 1):
            if line.startswith("```"):
                in_fence = not in_fence
            if re.match(r"^(<<<<<<<|=======|>>>>>>>)", line):
                errors.append((relative, line_number, "conflict_marker"))
            if line.rstrip() != line:
                errors.append((relative, line_number, "trailing_whitespace"))
            if not in_fence and line.startswith("|") and line.endswith("|"):
                columns = len(re.split(r"(?<!\\)\|", line)) - 2
                if previous_columns is not None and columns != previous_columns:
                    errors.append((relative, line_number, "table_column_mismatch"))
                previous_columns = columns
            else:
                previous_columns = None
        if in_fence:
            errors.append((relative, "unclosed_fence"))
        # 不将代码块内示例链接误当成站点导航。
        body = re.sub(r"```.*?```", "", text, flags=re.S)
        explicit_ids = re.findall(r'<span id="([^"]+)">', body)
        if len(explicit_ids) != len(set(explicit_ids)):
            errors.append((relative, "duplicate_explicit_anchor"))
        for target in re.findall(r"\[[^\]\n]*\]\(([^\s)]+)\)", body):
            if target.startswith(("http:", "https:", "mailto:", "/")):
                continue
            target, _, anchor = urllib.parse.unquote(target).partition("#")
            resolved = file.parent / target if target else file
            if not resolved.exists():
                errors.append((relative, "missing_link", target))
            elif anchor.startswith(("section-", "table-", "td-", "index-", "mapping-", "external-scope")):
                if f'id="{anchor}"' not in resolved.read_text(encoding="utf-8"):
                    errors.append((relative, "missing_explicit_anchor", target, anchor))
    for slug, names in slugs.items():
        if len(names) > 1:
            errors.append((slug, "duplicate_slug", names))
    # 对旧文档只读检查路由冲突，绝不更新其内容或导航。
    for old in (root / "doc_code/docs-develop").rglob("*.md"):
        if docs in old.parents:
            continue
        match = re.search(r"^slug:\s*['\"]?([^'\"\r\n]+)", old.read_text(encoding="utf-8"), re.M)
        if match and match.group(1).strip() in slugs:
            errors.append((old.name, "existing_slug_collision", match.group(1)))
    for script in (docs / "tools").glob("*.py"):
        ast.parse(script.read_text(encoding="utf-8"), filename=str(script))
    # 真正的密码值只在内存做匹配；只报告文件与配置键名，不输出值。
    secrets = []
    def walk(value, key=""):
        if isinstance(value, dict):
            for child_key, child in value.items():
                walk(child, str(child_key))
        elif isinstance(value, list):
            for child in value:
                walk(child, key)
        elif isinstance(value, str) and re.search(r"password|secret|private.?key|access.?key|api.?key", key, re.I):
            if len(value) >= 8 and not value.startswith("${"):
                secrets.append((key, value))
    for config in (root / "doc/数据库设计管理/nacos").glob("*.yaml"):
        for document in yaml.safe_load_all(config.read_text(encoding="utf-8")):
            walk(document)
    for file in docs.rglob("*"):
        if file.is_file() and file.suffix in (".md", ".py", ".java", ".json", ".mjs"):
            content = file.read_text(encoding="utf-8")
            for key, secret in secrets:
                if secret in content:
                    errors.append((file.relative_to(docs).as_posix(), "credential_value", key))
    print({"markdown_files": len(files), "unique_slugs": len(slugs), "errors": len(errors)})
    for error in errors:
        print(error)
    raise SystemExit(bool(errors))


if __name__ == "__main__":
    main()
