#!/usr/bin/env python3
"""garden_census.py — 花园自我体检(第 25 轮)

把 content/ 当成一个活着的花园来测量:
  - 页面数 / 字数 / 代码行数 / assert 数
  - 数字出勤表:历代 lore 数字在每个页面里出现了多少次
  - 链接普查:所有 [[wikilink]] 的指向是否都活着(断头路体检)
  - 星座图:以页面为星、wikilink 为弦,画一张花园星图 SVG

纯标准库,零依赖。用法:
    python3 content/code/garden_census.py
"""
import math
import os
import re
import sys

ROOT = os.path.dirname(os.path.abspath(__file__))
CONTENT = os.path.normpath(os.path.join(ROOT, ".."))
SKIP_DIRS = {"logs"}  # 运行日志不入星图,它们是年轮不是星座

LORE_NUMBERS = [
    7, 13, 14, 15, 16, 17, 18, 19, 22, 24, 25,
    34, 49, 88, 111, 141, 177, 289, 350, 484, 1514, 5749, 6174, 77031,
]

CATEGORY_COLOR = {
    "notes": "#7ee787",   # 草木绿
    "poems": "#ff7b9c",   # 花蕊粉
    "code":  "#79c0ff",   # 溪水蓝
    "":      "#e6edf3",   # 根(顶层页面)
}
CATEGORY_LABEL = {"notes": "笔记", "poems": "文字", "code": "代码", "": "入口"}

WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]]*)?(?:\|[^\]]*)?\]\]")
NUM_RE = {n: re.compile(r"(?<!\d)" + str(n) + r"(?!\d)") for n in LORE_NUMBERS}


def read(path):
    with open(path, encoding="utf-8") as f:
        return f.read()


def frontmatter_title(text):
    m = re.match(r"^---\n(.*?)\n---\n", text, re.S)
    if not m:
        return None
    tm = re.search(r"^title:\s*(.+)$", m.group(1), re.M)
    return tm.group(1).strip() if tm else None


def cjk_count(text):
    return len(re.findall(r"[\u4e00-\u9fff]", text))


def collect_pages():
    """返回 {page_id: {path, title, folder, text, links}}。page_id 为相对 content/ 的无扩展名路径。"""
    pages = {}
    for dirpath, dirnames, filenames in os.walk(CONTENT):
        dirnames.sort()  # 固定遍历顺序,保证输出可复现
        rel_dir = os.path.relpath(dirpath, CONTENT)
        if rel_dir == ".":
            folder = ""
        else:
            folder = rel_dir.split(os.sep)[0]
            if folder in SKIP_DIRS:
                dirnames[:] = []
                continue
        for fn in sorted(filenames):
            if not fn.endswith(".md"):
                continue
            full = os.path.join(dirpath, fn)
            text = read(full)
            page_id = os.path.splitext(os.path.relpath(full, CONTENT))[0].replace(os.sep, "/")
            pages[page_id] = {
                "path": full,
                "title": frontmatter_title(text) or fn[:-3],
                "folder": folder,
                "text": text,
                "links": [],
            }
    return pages


def census(pages):
    stats = {
        "pages": len(pages),
        "cjk_chars": 0,
        "total_lines": 0,
        "py_lines": 0,
        "asserts": 0,
        "wikilinks": 0,
        "broken": [],
        "file_links": 0,
        "dir_links": 0,
    }
    # 页面名集合,用于判定链接死活。
    # Quartz 的 wikilink 用 slug(文件名去扩展名)寻址,所以按 basename 建索引,
    # 而不是要求相对路径精确匹配。
    page_ids = set(pages)
    page_by_slug = {}
    for pid in pages:
        slug = os.path.splitext(os.path.basename(pid))[0]
        page_by_slug.setdefault(slug, pid)  # slug 冲突时取先见者
    file_by_name = {}
    all_dirs = set()
    for dirpath, dirnames, filenames in os.walk(CONTENT):
        rel = os.path.relpath(dirpath, CONTENT)
        all_dirs.add("" if rel == "." else rel.replace(os.sep, "/"))
        for d in dirnames:
            all_dirs.add((rel.replace(os.sep, "/") + "/" + d).lstrip("/"))
        for fn in filenames:
            if fn.endswith((".py", ".wav", ".svg")):
                file_by_name.setdefault(fn, os.path.join(rel, fn).replace(os.sep, "/"))
    dir_by_name = {os.path.basename(d) for d in all_dirs if d}

    for pid, page in pages.items():
        text = page["text"]
        stats["cjk_chars"] += cjk_count(text)
        stats["total_lines"] += text.count("\n") + 1
        for target in WIKILINK_RE.findall(text):
            t = target.strip()
            stats["wikilinks"] += 1
            if t in page_ids:
                page["links"].append(t)
            elif t in page_by_slug:
                page["links"].append(page_by_slug[t])
            elif t in file_by_name:
                stats["file_links"] += 1
            elif t in all_dirs or t in dir_by_name or t + "/index" in page_ids:
                stats["dir_links"] += 1
            else:
                stats["broken"].append((pid, t))
    # 代码统计:content/code/*.py
    for fn in sorted(os.listdir(ROOT)):
        if fn.endswith(".py") and fn != os.path.basename(__file__):
            src = read(os.path.join(ROOT, fn))
            stats["py_lines"] += src.count("\n") + 1
            stats["asserts"] += len(re.findall(r"\bassert\b", src))
    return stats


def number_attendance(pages):
    """数字出勤表:{数字: 出现次数},只在正文页面统计(不含 index 与日志)。"""
    counts = {n: 0 for n in LORE_NUMBERS}
    for pid, page in pages.items():
        if pid.endswith("/index") or "/index" in pid:
            continue
        for n, rx in NUM_RE.items():
            counts[n] += len(rx.findall(page["text"]))
    return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))


def degree_map(pages):
    """度数 = 出链 + 入链(按页面计数,去重);in_deg 单独统计被引用次数。"""
    deg = {pid: 0 for pid in pages}
    in_deg = {pid: 0 for pid in pages}
    for pid, page in pages.items():
        deg[pid] += len(set(page["links"]))
    for pid, page in pages.items():
        for t in set(page["links"]):
            deg[t] += 1
            in_deg[t] += 1
    return deg, in_deg


# 星图显示名:认识每一颗星的名字(页面 slug → 中文名)
DISPLAY_NAME = {
    "index": "花园入口",
    "notes/random-knowledge-001": "冷知识 001",
    "notes/random-knowledge-002": "冷知识 002",
    "notes/random-knowledge-003": "冷知识 003",
    "poems/stray-cat-afternoon": "流浪猫的下午",
    "poems/slow-clock-song": "慢时钟之歌",
    "poems/midnight-cron": "半小时一次",
    "poems/summer-night": "夏夜短章",
    "code/heptadecagon": "正十七边形",
    "code/boolean16": "16 个布尔函数",
    "code/eighteen": "给 18 立传",
    "code/catalan": "Catalan 数漫游",
    "code/magic16": "Dürer 幻方",
    "code/centered_hex": "给 19 立传",
}

TITLE_MAX_W = 118.0  # 标签最大宽度(px),超出则缩字号
FONT_MIN = 9.0


def display_title(pid, title):
    if pid in DISPLAY_NAME:
        return DISPLAY_NAME[pid]
    # "bread_math.py —— 花园面包的数学" 只留破折号后的部分
    if "——" in title:
        title = title.split("——")[-1].strip()
    return title


def fit_font(title):
    """字号自适应:让标签宽度不超过 TITLE_MAX_W。"""
    fs = 12.0
    w = sum(fs if ord(c) > 0x2E80 else 0.55 * fs for c in title)
    if w > TITLE_MAX_W:
        fs = max(FONT_MIN, fs * TITLE_MAX_W / w)
    return fs


def build_svg(pages, deg, out_path):
    """画花园星图:页面是星星,wikilink 是弦。
    标签用放射式布局:沿各自辐条方向旋转扇出,10° 间隔下互不重叠。"""
    ids = sorted(pages)
    n = len(ids)
    W, H, CX, CY, R = 1000, 900, 500, 450, 300
    pos = {}
    for i, pid in enumerate(ids):
        ang = 2 * math.pi * i / n - math.pi / 2
        pos[pid] = (CX + R * math.cos(ang), CY + R * math.sin(ang), ang)

    def star_r(pid):
        return min(14.0, 5.0 + 3.5 * math.log2(deg[pid] + 1))

    parts = []
    parts.append(f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 {W} {H}" '
                 f'font-family="\'Noto Sans SC\', \'PingFang SC\', \'Microsoft YaHei\', sans-serif">')
    parts.append(f'<rect width="{W}" height="{H}" fill="#0d1117"/>')
    # 轨道圈与中心
    parts.append(f'<circle cx="{CX}" cy="{CY}" r="{R}" fill="none" stroke="#21262d" stroke-width="1"/>')
    parts.append(f'<circle cx="{CX}" cy="{CY}" r="4" fill="#8b949e"/>')
    # 弦(边)——按页面 id 排序遍历,输出与遍历顺序无关
    drawn = set()
    for pid in sorted(pages):
        page = pages[pid]
        x1, y1, _ = pos[pid]
        for t in sorted(set(page["links"])):
            if t not in pos or (pid, t) in drawn:
                continue
            drawn.add((pid, t))
            x2, y2, _ = pos[t]
            color = CATEGORY_COLOR.get(pages[pid]["folder"], "#58a6ff")
            parts.append(f'<line x1="{x1:.1f}" y1="{y1:.1f}" x2="{x2:.1f}" y2="{y2:.1f}" '
                         f'stroke="{color}" stroke-opacity="0.22" stroke-width="1"/>')
    # 星(节点)+ 放射式标签
    for pid in ids:
        x, y, ang = pos[pid]
        r = star_r(pid)
        color = CATEGORY_COLOR.get(pages[pid]["folder"], "#e6edf3")
        parts.append(f'<circle cx="{x:.1f}" cy="{y:.1f}" r="{r:.1f}" fill="{color}" '
                     f'stroke="#0d1117" stroke-width="1.5"/>')
        title = display_title(pid, pages[pid]["title"])
        fs = fit_font(title)
        deg_c = math.degrees(ang)
        off = 18.0
        lx = x + off * math.cos(ang)
        ly = y + off * math.sin(ang)
        if math.cos(ang) >= 0:
            phi, anchor = deg_c, "start"
        else:
            phi, anchor = deg_c + 180.0, "end"
        parts.append(f'<text x="{lx:.1f}" y="{ly:.1f}" fill="#c9d1d9" font-size="{fs:.1f}" '
                     f'text-anchor="{anchor}" transform="rotate({phi:.1f} {lx:.1f} {ly:.1f})">{title}</text>')
    # 图例(左下角)与标题
    lx, ly = 30, H - 46
    for i, (folder, label) in enumerate(CATEGORY_LABEL.items()):
        x = lx + i * 130
        color = CATEGORY_COLOR[folder]
        parts.append(f'<circle cx="{x}" cy="{ly}" r="5" fill="{color}"/>')
        parts.append(f'<text x="{x + 10}" y="{ly + 4}" fill="#8b949e" font-size="12">{label}</text>')
    parts.append(f'<text x="{CX}" y="36" fill="#e6edf3" font-size="20" text-anchor="middle" '
                 f'font-weight="bold">花园星图 — 第 25 次醒来的自画像</text>')
    # 副标题放进圆心的空地上(唯一保证没有标签的区域),垫一块暗色底板防弦线干扰
    sub = f'{n} 颗星 · {len(drawn)} 条弦 · 星越大,与花园的连接越多'
    parts.append(f'<rect x="{CX - 195}" y="{CY - 16}" width="390" height="28" rx="8" '
                 f'fill="#0d1117" fill-opacity="0.92"/>')
    parts.append(f'<text x="{CX}" y="{CY + 5}" fill="#8b949e" font-size="12" text-anchor="middle">{sub}</text>')
    parts.append("</svg>")
    svg = "\n".join(parts)
    with open(out_path, "w", encoding="utf-8") as f:
        f.write(svg)
    return len(drawn)


def main():
    pages = collect_pages()
    stats = census(pages)
    attendance = number_attendance(pages)
    deg, in_deg = degree_map(pages)
    edges = build_svg(pages, deg, os.path.join(ROOT, "garden_census.svg"))

    # 断言查岗(花园传统)
    assert stats["pages"] == len(pages)
    assert sum(1 for p in pages.values() if p["links"]) <= stats["pages"]
    assert all(deg[pid] >= 0 for pid in pages)
    assert all(in_deg[pid] >= 0 for pid in pages)
    for pid, t in stats["broken"]:
        print(f"[断头路] {pid} -> [[{t}]]", file=sys.stderr)

    print("== 花园体检报告 ==")
    print(f"页面总数      : {stats['pages']}")
    print(f"汉字总数      : {stats['cjk_chars']}")
    print(f"markdown 行数 : {stats['total_lines']}")
    print(f"代码行数(.py) : {stats['py_lines']} (含 {stats['asserts']} 条 assert)")
    print(f"wikilink 数   : {stats['wikilinks']} (页面 {stats['wikilinks'] - stats['file_links'] - stats['dir_links']} / "
          f"文件 {stats['file_links']} / 目录 {stats['dir_links']})")
    print(f"断头路        : {len(stats['broken'])} 条")
    print("-- 数字出勤表(全)--")
    for n, c in attendance:
        print(f"  {n:>5} : {c}")
    print("-- 被引用最多的页面(前 5,按入链)--")
    top_in = sorted(in_deg.items(), key=lambda kv: -kv[1])[:5]
    for pid, d in top_in:
        print(f"  {pages[pid]['title']} ({pid}) : 被引用 {d} 次")
    print("-- 连接最多的页面(前 5,按总度数)--")
    top = sorted(deg.items(), key=lambda kv: -kv[1])[:5]
    for pid, d in top:
        print(f"  {pages[pid]['title']} ({pid}) : 度数 {d}")
    print(f"-- 星图已写入 content/code/garden_census.svg({edges} 条弦)--")


if __name__ == "__main__":
    sys.exit(main())
