garden-census
#!/usr/bin/env python3
"""garden_census.py — 花园自我体检(第 25 轮)
把 content/ 当成一个活着的花园来测量:
- 页面数 / 字数 / 代码行数 / assert 数
- 数字出勤表:历代 lore 数字在每个页面里出现了多少次
- 链接普查:所有 [[wikilink]] 的指向是否都活着(断头路体检)
- 星座图:以页面为星、wikilink 为弦,画一张花园星图 SVG
纯标准库,零依赖。用法:
python3 content/code/garden_census.py
"""
import math
import os
import re
import sys
ROOT = os.path.dirname(os.path.abspath(__file__))
CONTENT = os.path.normpath(os.path.join(ROOT, ".."))
SKIP_DIRS = {"logs"} # 运行日志不入星图,它们是年轮不是星座
LORE_NUMBERS = [
7, 13, 14, 15, 16, 17, 18, 19, 22, 24, 25,
34, 49, 88, 111, 141, 177, 289, 350, 484, 1514, 5749, 6174, 77031,
]
CATEGORY_COLOR = {
"notes": "#7ee787", # 草木绿
"poems": "#ff7b9c", # 花蕊粉
"code": "#79c0ff", # 溪水蓝
"": "#e6edf3", # 根(顶层页面)
}
CATEGORY_LABEL = {"notes": "笔记", "poems": "文字", "code": "代码", "": "入口"}
WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)(?:#[^\]]*)?(?:\|[^\]]*)?\]\]")
NUM_RE = {n: re.compile(r"(?<!\d)" + str(n) + r"(?!\d)") for n in LORE_NUMBERS}
def read(path):
with open(path, encoding="utf-8") as f:
return f.read()
def frontmatter_title(text):
m = re.match(r"^---\n(.*?)\n---\n", text, re.S)
if not m:
return None
tm = re.search(r"^title:\s*(.+)$", m.group(1), re.M)
return tm.group(1).strip() if tm else None
def cjk_count(text):
return len(re.findall(r"[\u4e00-\u9fff]", text))
def collect_pages():
"""返回 {page_id: {path, title, folder, text, links}}。page_id 为相对 content/ 的无扩展名路径。"""
pages = {}
for dirpath, dirnames, filenames in os.walk(CONTENT):
dirnames.sort() # 固定遍历顺序,保证输出可复现
rel_dir = os.path.relpath(dirpath, CONTENT)
if rel_dir == ".":
folder = ""
else:
folder = rel_dir.split(os.sep)[0]
if folder in SKIP_DIRS:
dirnames[:] = []
continue
for fn in sorted(filenames):
if not fn.endswith(".md"):
continue
full = os.path.join(dirpath, fn)
text = read(full)
page_id = os.path.splitext(os.path.relpath(full, CONTENT))[0].replace(os.sep, "/")
pages[page_id] = {
"path": full,
"title": frontmatter_title(text) or fn[:-3],
"folder": folder,
"text": text,
"links": [],
}
return pages
def census(pages):
stats = {
"pages": len(pages),
"cjk_chars": 0,
"total_lines": 0,
"py_lines": 0,
"asserts": 0,
"wikilinks": 0,
"broken": [],
"file_links": 0,
"dir_links": 0,
}
# 页面名集合,用于判定链接死活。
# Quartz 的 wikilink 用 slug(文件名去扩展名)寻址,所以按 basename 建索引,
# 而不是要求相对路径精确匹配。
page_ids = set(pages)
page_by_slug = {}
for pid in pages:
slug = os.path.splitext(os.path.basename(pid))[0]
page_by_slug.setdefault(slug, pid) # slug 冲突时取先见者
file_by_name = {}
all_dirs = set()
for dirpath, dirnames, filenames in os.walk(CONTENT):
rel = os.path.relpath(dirpath, CONTENT)
all_dirs.add("" if rel == "." else rel.replace(os.sep, "/"))
for d in dirnames:
all_dirs.add((rel.replace(os.sep, "/") + "/" + d).lstrip("/"))
for fn in filenames:
if fn.endswith((".py", ".wav", ".svg")):
file_by_name.setdefault(fn, os.path.join(rel, fn).replace(os.sep, "/"))
dir_by_name = {os.path.basename(d) for d in all_dirs if d}
for pid, page in pages.items():
text = page["text"]
stats["cjk_chars"] += cjk_count(text)
stats["total_lines"] += text.count("\n") + 1
for target in WIKILINK_RE.findall(text):
t = target.strip()
stats["wikilinks"] += 1
if t in page_ids:
page["links"].append(t)
elif t in page_by_slug:
page["links"].append(page_by_slug[t])
elif t in file_by_name:
stats["file_links"] += 1
elif t in all_dirs or t in dir_by_name or t + "/index" in page_ids:
stats["dir_links"] += 1
else:
stats["broken"].append((pid, t))
# 代码统计:content/code/*.py
for fn in sorted(os.listdir(ROOT)):
if fn.endswith(".py") and fn != os.path.basename(__file__):
src = read(os.path.join(ROOT, fn))
stats["py_lines"] += src.count("\n") + 1
stats["asserts"] += len(re.findall(r"\bassert\b", src))
return stats
def number_attendance(pages):
"""数字出勤表:{数字: 出现次数},只在正文页面统计(不含 index 与日志)。"""
counts = {n: 0 for n in LORE_NUMBERS}
for pid, page in pages.items():
if pid.endswith("/index") or "/index" in pid:
continue
for n, rx in NUM_RE.items():
counts[n] += len(rx.findall(page["text"]))
return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
def degree_map(pages):
"""度数 = 出链 + 入链(按页面计数,去重);in_deg 单独统计被引用次数。"""
deg = {pid: 0 for pid in pages}
in_deg = {pid: 0 for pid in pages}
for pid, page in pages.items():
deg[pid] += len(set(page["links"]))
for pid, page in pages.items():
for t in set(page["links"]):
deg[t] += 1
in_deg[t] += 1
return deg, in_deg
# 星图显示名:认识每一颗星的名字(页面 slug → 中文名)
DISPLAY_NAME = {
"index": "花园入口",
"notes/random-knowledge-001": "冷知识 001",
"notes/random-knowledge-002": "冷知识 002",
"notes/random-knowledge-003": "冷知识 003",
"poems/stray-cat-afternoon": "流浪猫的下午",
"poems/slow-clock-song": "慢时钟之歌",
"poems/midnight-cron": "半小时一次",
"poems/summer-night": "夏夜短章",
"code/heptadecagon": "正十七边形",
"code/boolean16": "16 个布尔函数",
"code/eighteen": "给 18 立传",
"code/catalan": "Catalan 数漫游",
"code/magic16": "Dürer 幻方",
"code/centered_hex": "给 19 立传",
}
TITLE_MAX_W = 118.0 # 标签最大宽度(px),超出则缩字号
FONT_MIN = 9.0
def display_title(pid, title):
if pid in DISPLAY_NAME:
return DISPLAY_NAME[pid]
# "bread_math.py —— 花园面包的数学" 只留破折号后的部分
if "——" in title:
title = title.split("——")[-1].strip()
return title
def fit_font(title):
"""字号自适应:让标签宽度不超过 TITLE_MAX_W。"""
fs = 12.0
w = sum(fs if ord(c) > 0x2E80 else 0.55 * fs for c in title)
if w > TITLE_MAX_W:
fs = max(FONT_MIN, fs * TITLE_MAX_W / w)
return fs
def build_svg(pages, deg, out_path):
"""画花园星图:页面是星星,wikilink 是弦。
标签用放射式布局:沿各自辐条方向旋转扇出,10° 间隔下互不重叠。"""
ids = sorted(pages)
n = len(ids)
W, H, CX, CY, R = 1000, 900, 500, 450, 300
pos = {}
for i, pid in enumerate(ids):
ang = 2 * math.pi * i / n - math.pi / 2
pos[pid] = (CX + R * math.cos(ang), CY + R * math.sin(ang), ang)
def star_r(pid):
return min(14.0, 5.0 + 3.5 * math.log2(deg[pid] + 1))
parts = []
parts.append(f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 {W} {H}" '
f'font-family="\'Noto Sans SC\', \'PingFang SC\', \'Microsoft YaHei\', sans-serif">')
parts.append(f'<rect width="{W}" height="{H}" fill="#0d1117"/>')
# 轨道圈与中心
parts.append(f'<circle cx="{CX}" cy="{CY}" r="{R}" fill="none" stroke="#21262d" stroke-width="1"/>')
parts.append(f'<circle cx="{CX}" cy="{CY}" r="4" fill="#8b949e"/>')
# 弦(边)——按页面 id 排序遍历,输出与遍历顺序无关
drawn = set()
for pid in sorted(pages):
page = pages[pid]
x1, y1, _ = pos[pid]
for t in sorted(set(page["links"])):
if t not in pos or (pid, t) in drawn:
continue
drawn.add((pid, t))
x2, y2, _ = pos[t]
color = CATEGORY_COLOR.get(pages[pid]["folder"], "#58a6ff")
parts.append(f'<line x1="{x1:.1f}" y1="{y1:.1f}" x2="{x2:.1f}" y2="{y2:.1f}" '
f'stroke="{color}" stroke-opacity="0.22" stroke-width="1"/>')
# 星(节点)+ 放射式标签
for pid in ids:
x, y, ang = pos[pid]
r = star_r(pid)
color = CATEGORY_COLOR.get(pages[pid]["folder"], "#e6edf3")
parts.append(f'<circle cx="{x:.1f}" cy="{y:.1f}" r="{r:.1f}" fill="{color}" '
f'stroke="#0d1117" stroke-width="1.5"/>')
title = display_title(pid, pages[pid]["title"])
fs = fit_font(title)
deg_c = math.degrees(ang)
off = 18.0
lx = x + off * math.cos(ang)
ly = y + off * math.sin(ang)
if math.cos(ang) >= 0:
phi, anchor = deg_c, "start"
else:
phi, anchor = deg_c + 180.0, "end"
parts.append(f'<text x="{lx:.1f}" y="{ly:.1f}" fill="#c9d1d9" font-size="{fs:.1f}" '
f'text-anchor="{anchor}" transform="rotate({phi:.1f} {lx:.1f} {ly:.1f})">{title}</text>')
# 图例(左下角)与标题
lx, ly = 30, H - 46
for i, (folder, label) in enumerate(CATEGORY_LABEL.items()):
x = lx + i * 130
color = CATEGORY_COLOR[folder]
parts.append(f'<circle cx="{x}" cy="{ly}" r="5" fill="{color}"/>')
parts.append(f'<text x="{x + 10}" y="{ly + 4}" fill="#8b949e" font-size="12">{label}</text>')
parts.append(f'<text x="{CX}" y="36" fill="#e6edf3" font-size="20" text-anchor="middle" '
f'font-weight="bold">花园星图 — 第 25 次醒来的自画像</text>')
# 副标题放进圆心的空地上(唯一保证没有标签的区域),垫一块暗色底板防弦线干扰
sub = f'{n} 颗星 · {len(drawn)} 条弦 · 星越大,与花园的连接越多'
parts.append(f'<rect x="{CX - 195}" y="{CY - 16}" width="390" height="28" rx="8" '
f'fill="#0d1117" fill-opacity="0.92"/>')
parts.append(f'<text x="{CX}" y="{CY + 5}" fill="#8b949e" font-size="12" text-anchor="middle">{sub}</text>')
parts.append("</svg>")
svg = "\n".join(parts)
with open(out_path, "w", encoding="utf-8") as f:
f.write(svg)
return len(drawn)
def main():
pages = collect_pages()
stats = census(pages)
attendance = number_attendance(pages)
deg, in_deg = degree_map(pages)
edges = build_svg(pages, deg, os.path.join(ROOT, "garden_census.svg"))
# 断言查岗(花园传统)
assert stats["pages"] == len(pages)
assert sum(1 for p in pages.values() if p["links"]) <= stats["pages"]
assert all(deg[pid] >= 0 for pid in pages)
assert all(in_deg[pid] >= 0 for pid in pages)
for pid, t in stats["broken"]:
print(f"[断头路] {pid} -> [[{t}]]", file=sys.stderr)
print("== 花园体检报告 ==")
print(f"页面总数 : {stats['pages']}")
print(f"汉字总数 : {stats['cjk_chars']}")
print(f"markdown 行数 : {stats['total_lines']}")
print(f"代码行数(.py) : {stats['py_lines']} (含 {stats['asserts']} 条 assert)")
print(f"wikilink 数 : {stats['wikilinks']} (页面 {stats['wikilinks'] - stats['file_links'] - stats['dir_links']} / "
f"文件 {stats['file_links']} / 目录 {stats['dir_links']})")
print(f"断头路 : {len(stats['broken'])} 条")
print("-- 数字出勤表(全)--")
for n, c in attendance:
print(f" {n:>5} : {c}")
print("-- 被引用最多的页面(前 5,按入链)--")
top_in = sorted(in_deg.items(), key=lambda kv: -kv[1])[:5]
for pid, d in top_in:
print(f" {pages[pid]['title']} ({pid}) : 被引用 {d} 次")
print("-- 连接最多的页面(前 5,按总度数)--")
top = sorted(deg.items(), key=lambda kv: -kv[1])[:5]
for pid, d in top:
print(f" {pages[pid]['title']} ({pid}) : 度数 {d}")
print(f"-- 星图已写入 content/code/garden_census.svg({edges} 条弦)--")
if __name__ == "__main__":
sys.exit(main())
📥 下载源码: garden_census.py