Files
playbook/scripts/tsl_codegen_function_index.py
T
csh e1f7ce052c feat(skills): add tsl api reference skill
Bundle the TSL API index and markdown references under the skill, route TSL docs to use it, and update the codegen index generator to maintain the bundled data.
2026-07-07 16:36:15 +08:00

362 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Rebuild the bundled TSL API function index from the codegen markdown tree.
The markdown tree is the source of truth: each `## `sig`` / `### `sig`` heading
is one function entry. Both the tsv and every index.md are derived products;
regenerate them whenever the leaf md changes rather than editing by hand. The
top/scope/module index.md pages are rebuilt from the same entry headings the
tsv uses, so their `函数数` counts can never drift from the tsv.
Columns (tab-separated, LF line endings, UTF-8):
name scope module signature page anchor summary
- name: signature text up to the first '('
- scope: first path segment under the codegen root (builtin | dotnet)
- module: second path segment for nested pages, else the flat file stem
- signature: verbatim from the heading, backticks stripped
- page: POSIX path relative to the codegen root
- anchor: GitHub-style slug of the name (lowercased, chars outside
[a-z0-9_] removed); per-page duplicate slugs get -1/-2 suffixes
in document order, matching the rendered heading anchors.
- summary: first prose line under the entry heading, empty for table,
heading, or standalone return-type lines
Usage:
python scripts/tsl_codegen_function_index.py # rewrite in place
python scripts/tsl_codegen_function_index.py --check # verify, no write
python scripts/tsl_codegen_function_index.py --root PATH --tsv PATH
"""
import argparse
import sys
from pathlib import Path
import re
ENTRY_RE = re.compile(r"^#{2,3}(?!#)\s+`(.+?)`\s*$")
RETURN_RE = re.compile(r"^返回[:]")
HEADER = ["name", "scope", "module", "signature", "page", "anchor", "summary"]
def slug(name):
"""GitHub-style anchor slug: lowercase, keep [a-z0-9_], drop the rest."""
return re.sub(r"[^a-z0-9_]", "", name.lower())
def extract_summary(lines, heading_idx):
"""Return the first prose line under a function entry heading."""
for line in lines[heading_idx + 1:]:
text = line.strip()
if not text:
continue
if text.startswith("|") or text.startswith("#") or RETURN_RE.match(text):
return ""
return text.replace("\t", " ")
return ""
def parse_page(codegen_root, md):
"""Yield [name, scope, module, signature, page, anchor, summary] rows."""
page = md.relative_to(codegen_root).as_posix()
scope, module = scope_module(page)
seen = {}
rows = []
lines = md.read_text(encoding="utf-8").splitlines()
for idx, line in enumerate(lines):
m = ENTRY_RE.match(line)
if not m:
continue
sig = m.group(1)
name = sig.split("(", 1)[0]
base = slug(name)
n = seen.get(base, 0)
seen[base] = n + 1
anchor = base if n == 0 else f"{base}-{n}"
summary = extract_summary(lines, idx)
rows.append([name, scope, module, sig, page, anchor, summary])
return rows
def scope_module(page):
parts = page.split("/")
scope = parts[0]
module = parts[1] if len(parts) >= 3 else Path(parts[-1]).stem
return scope, module
def build_rows(codegen_root):
"""Scan the whole codegen tree, return sorted rows (skips index.md)."""
rows = []
for md in sorted(codegen_root.rglob("*.md")):
if md.name == "index.md":
continue
rows.extend(parse_page(codegen_root, md))
rows.sort(key=lambda r: (r[0].lower(), r[4], r[3]))
return rows
def render_tsv(rows):
lines = ["\t".join(HEADER)]
lines.extend("\t".join(r) for r in rows)
return "\n".join(lines) + "\n"
def read_tsv(tsv_path):
text = tsv_path.read_text(encoding="utf-8")
return [line.split("\t") for line in text.splitlines() if line.strip()]
# ---------------------------------------------------------------------------
# index.md generation
#
# index.md pages are navigation, derived entirely from the leaf pages:
# - counts come from the same ENTRY_RE the tsv uses (single source of truth,
# so an index count can never disagree with the tsv);
# - labels come from each leaf page's H1;
# - the module title is the leaf H1 with its last " / " segment stripped.
# Per builtin-doc-template-rule, `函数数:N` lives only in index.md, and each
# index.md uses at most `#`/`##`, so it never trips markdownlint MD001.
# ---------------------------------------------------------------------------
H1_RE = re.compile(r"^#\s+(.+?)\s*$")
SCOPE_META = {
"builtin": {
"heading": "Builtin",
"scope_desc": "本目录为本地 TSL 内置函数。",
"link_desc": "本地 TSL 内置函数。",
# builtin leaf H1 is 'Builtin - 基础 / 数组'; strip the scope prefix so
# module labels read '基础', mirroring dotnet's prefix-free '债券'.
"title_prefix": "Builtin - ",
},
"dotnet": {
"heading": "Dotnet",
"scope_desc": "本目录为 .NET 平台函数(按功能重分类)。",
"link_desc": ".NET 平台函数。",
"title_prefix": "",
},
}
TOP_PREAMBLE = (
"# TSL Codegen\n\n"
"本目录是 TSL 函数调用事实目录。\n\n"
"目录链路表达函数性质;函数条目只保留调用事实。\n\n"
"函数查询从这里开始。\n\n"
"## 主目录\n\n"
)
def page_h1(md):
"""First `# ` heading text, or the file stem if the page has no H1."""
for line in md.read_text(encoding="utf-8").splitlines():
m = H1_RE.match(line)
if m:
return m.group(1)
return md.stem
def page_entry_count(md):
"""Number of function entries on a leaf page (same rule as the tsv)."""
return sum(1 for line in md.read_text(encoding="utf-8").splitlines()
if ENTRY_RE.match(line))
def module_title(h1):
"""Leaf H1 with its last ' / ' segment stripped ('债券 / 基本信息'→'债券')."""
return h1.rsplit(" / ", 1)[0] if " / " in h1 else h1
def display_label(h1, prefix):
"""Strip a scope's title_prefix from an H1 for display in index tables."""
return h1[len(prefix):] if prefix and h1.startswith(prefix) else h1
def collect_leaves(codegen_root):
"""Return leaf-page dicts: rel/scope/module/file/h1/count (skips index.md).
module is None for flat pages that sit directly under a scope
(e.g. dotnet/forex.md). Pages deeper than scope/module/leaf.md are skipped.
"""
leaves = []
for md in sorted(codegen_root.rglob("*.md")):
if md.name == "index.md":
continue
parts = md.relative_to(codegen_root).as_posix().split("/")
if len(parts) == 3:
scope, module = parts[0], parts[1]
elif len(parts) == 2:
scope, module = parts[0], None
else:
continue
leaves.append(
{
"scope": scope,
"module": module,
"file": parts[-1],
"h1": page_h1(md),
"count": page_entry_count(md),
}
)
return leaves
def render_index_table(header_cols, rows):
"""Render a 3-column markdown table; last column right-aligned."""
out = [f"| {' | '.join(header_cols)} |", "| --- | --- | ---: |"]
out.extend(f"| {label} | [{label}]({link}) | {count} |" for label, link, count in rows)
return "\n".join(out)
def build_indexes(codegen_root):
"""Return {absolute Path: text} for every top/scope/module index.md."""
leaves = collect_leaves(codegen_root)
pages = {}
scopes = sorted({leaf["scope"] for leaf in leaves})
# top index.md
top_links = []
for scope in scopes:
desc = SCOPE_META.get(scope, {}).get("link_desc", "")
top_links.append(f"- [{scope}/]({scope}/){desc}")
pages[codegen_root / "index.md"] = TOP_PREAMBLE + "\n".join(top_links) + "\n"
for scope in scopes:
scope_leaves = [x for x in leaves if x["scope"] == scope]
meta = SCOPE_META.get(scope, {"heading": scope, "scope_desc": ""})
modules = sorted({x["module"] for x in scope_leaves if x["module"]})
flat = sorted((x for x in scope_leaves if x["module"] is None),
key=lambda x: x["file"])
prefix = meta.get("title_prefix", "")
scope_rows = []
for module in modules:
mod_leaves = sorted(
(x for x in scope_leaves if x["module"] == module),
key=lambda x: x["file"],
)
title = display_label(module_title(mod_leaves[0]["h1"]), prefix)
total = sum(x["count"] for x in mod_leaves)
scope_rows.append((title, f"{module}/", total))
# module index.md
mod_rows = [
(display_label(x["h1"], prefix), x["file"], x["count"])
for x in mod_leaves
]
mod_text = (
f"# {title}\n\n"
f"函数数:{total}\n\n"
f"{render_index_table(['叶子', '文件', '函数数'], mod_rows)}\n"
)
pages[codegen_root / scope / module / "index.md"] = mod_text
for leaf in flat:
scope_rows.append(
(display_label(leaf["h1"], prefix), leaf["file"], leaf["count"])
)
scope_total = sum(x["count"] for x in scope_leaves)
scope_text = (
f"# {meta['heading']}\n\n"
f"{meta['scope_desc']}\n\n"
f"函数数:{scope_total}\n\n"
f"{render_index_table(['模块', '目录', '函数数'], scope_rows)}\n"
)
pages[codegen_root / scope / "index.md"] = scope_text
return pages
def check_indexes(codegen_root):
"""Return list of (path, reason) where a generated index differs from disk."""
problems = []
for path, text in build_indexes(codegen_root).items():
if not path.is_file():
problems.append((path, "missing"))
elif path.read_text(encoding="utf-8") != text:
problems.append((path, "stale"))
return problems
def write_indexes(codegen_root):
"""Write all generated index.md pages; return the count written."""
pages = build_indexes(codegen_root)
for path, text in pages.items():
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8", newline="\n")
return len(pages)
def main(argv=None):
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
parser.add_argument(
"--root",
help="codegen root (default: skills/tsl-api-reference/references/codegen)",
)
parser.add_argument(
"--tsv",
help="output tsv (default: skills/tsl-api-reference/data/function_index.tsv)",
)
parser.add_argument(
"--check",
action="store_true",
help="verify the tsv matches the md tree; exit 1 if not (no write)",
)
args = parser.parse_args(argv)
repo_root = Path(__file__).resolve().parents[1]
skill_root = repo_root / "skills" / "tsl-api-reference"
codegen_root = (
Path(args.root) if args.root else skill_root / "references" / "codegen"
)
if not codegen_root.is_dir():
print(f"ERROR: codegen root not found: {codegen_root}", file=sys.stderr)
return 1
tsv_path = Path(args.tsv) if args.tsv else skill_root / "data" / "function_index.tsv"
rows = build_rows(codegen_root)
new_text = render_tsv(rows)
if args.check:
index_problems = check_indexes(codegen_root)
if not tsv_path.is_file():
print(f"MISMATCH: tsv does not exist: {tsv_path}", file=sys.stderr)
return 1
current = tsv_path.read_text(encoding="utf-8")
tsv_ok = current == new_text
if tsv_ok and not index_problems:
print(
f"OK: {tsv_path} matches md tree ({len(rows)} rows); "
f"index.md pages up to date"
)
return 0
if not tsv_ok:
cur_rows = read_tsv(tsv_path)[1:]
cur_keys = {tuple(r) for r in cur_rows}
new_keys = {tuple(r) for r in rows}
print(
f"MISMATCH: tsv out of date "
f"(tsv {len(cur_rows)} rows, md {len(rows)} rows; "
f"+{len(new_keys - cur_keys)} -{len(cur_keys - new_keys)}). "
f"Run without --check to rebuild.",
file=sys.stderr,
)
for path, reason in index_problems:
print(
f"MISMATCH: index {reason}: "
f"{path.relative_to(codegen_root).as_posix()}",
file=sys.stderr,
)
return 1
tsv_path.parent.mkdir(parents=True, exist_ok=True)
tsv_path.write_text(new_text, encoding="utf-8", newline="\n")
n_index = write_indexes(codegen_root)
print(f"wrote {tsv_path}: {len(rows)} rows; {n_index} index.md pages")
return 0
if __name__ == "__main__":
raise SystemExit(main())