Files
playbook/tools/tsl-codegen/scripts/build_index.py
T
csh 9edc8fc868 feat(tsl-codegen): support unified API declarations
Add TSF conversion and shared Markdown recognition for mixed
function, class, and unit declarations.

Extend generation, linting, indexing, examples, and tests around
the unified declaration model.
2026-07-29 15:45:50 +08:00

297 lines
9.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Rebuild the bundled TSL API index from the codegen markdown tree.
The markdown tree is the source of truth. The TSV is a derived product;
regenerate it whenever the leaf Markdown changes rather than editing it by
hand.
Columns (tab-separated, LF line endings, UTF-8):
name scope module signature page anchor tags summary
kind binding visibility owner qualified_name
- name: signature text up to the first '('
- scope: first path segment under the codegen root (for example project)
- module: second path segment for nested pages, else the flat file stem
- signature: verbatim from the heading, backticks stripped
- page: POSIX path relative to the codegen root
- anchor: top-level declarations use the historic name slug; typed members
slug the complete visible API title. Per-page duplicate slugs get
-1/-2 suffixes in document order.
- tags: space-separated keywords from `<!-- tags: ... -->`
- summary: first prose line under the entry heading
- kind: function/class/method/property/field/constant/unit/variable
- binding: instance/class/static/unit, or empty when not applicable
- visibility: public/protected for class members; public for unit interface
- owner: dot-separated containing API path, excluding the entry name
- qualified_name: dot-separated stable API identity
Usage (run from repo root; --skill-dir is required):
SKILL=skills/tsl-api-reference
python tools/tsl-codegen/scripts/build_index.py \
--skill-dir "$SKILL" # rewrite the TSV in place
python tools/tsl-codegen/scripts/build_index.py \
--skill-dir "$SKILL" --check # verify against md tree, no write
"""
import argparse
import sys
from pathlib import Path
import re
SCRIPT_DIR = Path(__file__).resolve().parent
if str(SCRIPT_DIR) not in sys.path:
sys.path.insert(0, str(SCRIPT_DIR))
from api_markdown import DECLARATION_LINE_RE, iter_api_entries, slug
RETURN_RE = re.compile(r"^返回[:]")
DECLARATION_TYPE_RE = re.compile(
r"^类型[:]\s*(function|class|unit)\s*$", re.IGNORECASE
)
TAGS_RE = re.compile(r"^<!--\s*tags:\s*(.*?)\s*-->$")
VISIBILITY_RE = re.compile(
r"^可见性[:]\s*`?(public|protected|private)`?\s*$",
re.IGNORECASE,
)
HEADER = [
"name",
"scope",
"module",
"signature",
"page",
"anchor",
"tags",
"summary",
"kind",
"binding",
"visibility",
"owner",
"qualified_name",
]
def extract_metadata(lines, heading_idx, end_idx=None):
"""Return tags and the first prose line under one API heading."""
tags = ""
summary = ""
summary_open = True
for line in lines[heading_idx + 1:end_idx]:
text = line.strip()
if not text:
continue
if DECLARATION_LINE_RE.fullmatch(text):
continue
tag_match = TAGS_RE.match(text)
if tag_match:
tags = " ".join(tag_match.group(1).split()).replace("\t", " ")
continue
if summary:
continue
if (
text.startswith("|")
or text.startswith("#")
or RETURN_RE.match(text)
or DECLARATION_TYPE_RE.match(text)
):
summary_open = False
continue
if summary_open:
summary = text.replace("\t", " ")
return tags, summary
def extract_visibility(lines, start, end):
for line in lines[start + 1:end]:
match = VISIBILITY_RE.match(line.strip())
if match:
return match.group(1).casefold()
return ""
def next_anchor(base, seen):
count = seen.get(base, 0)
seen[base] = count + 1
return base if count == 0 else f"{base}-{count}"
def parse_page(codegen_root, md):
"""Yield index rows for one Markdown page."""
page = md.relative_to(codegen_root).as_posix()
scope, module = scope_module(page)
seen = {}
rows = []
lines = md.read_text(encoding="utf-8").splitlines()
entries = list(iter_api_entries(lines))
root_kind = ""
root_name = ""
unit_class_owner = ""
for entry in entries:
heading = entry.heading
if not heading.valid:
continue
if heading.level == 2:
root_kind = heading.kind
root_name = heading.name
unit_class_owner = ""
owner = ""
anchor_base = slug(heading.name)
elif root_kind == "class":
owner = root_name
anchor_base = slug(heading.visible_title)
elif root_kind == "unit" and heading.level == 3:
anchor_base = slug(heading.visible_title)
owner = root_name
unit_class_owner = (
f"{root_name}.{heading.name}"
if heading.kind == "class"
else ""
)
elif root_kind == "unit" and heading.level == 4:
owner = unit_class_owner
anchor_base = slug(heading.visible_title)
else:
continue
qualified_name = (
f"{owner}.{heading.name}" if owner else heading.name
)
if root_kind == "unit" and heading.level == 3:
visibility = "public"
elif (
root_kind == "class" and heading.level == 3
) or (root_kind == "unit" and heading.level == 4):
visibility = extract_visibility(lines, entry.start, entry.end)
else:
visibility = ""
tags, summary = extract_metadata(lines, entry.start, entry.end)
rows.append(
[
heading.name,
scope,
module,
heading.signature,
page,
next_anchor(anchor_base, seen),
tags,
summary,
heading.kind,
heading.binding,
visibility,
owner,
qualified_name,
]
)
return rows
def scope_module(page):
parts = page.split("/")
scope = parts[0]
module = parts[1] if len(parts) >= 3 else Path(parts[-1]).stem
return scope, module
def build_rows(codegen_root):
"""Scan the whole codegen tree and return sorted rows."""
rows = []
for md in sorted(codegen_root.rglob("*.md")):
rows.extend(parse_page(codegen_root, md))
rows.sort(key=lambda r: (r[0].lower(), r[4], r[3]))
return rows
def render_tsv(rows):
lines = ["\t".join(HEADER)]
lines.extend("\t".join(r) for r in rows)
return "\n".join(lines) + "\n"
def read_tsv(tsv_path):
text = tsv_path.read_text(encoding="utf-8")
return [line.split("\t") for line in text.splitlines() if line.strip()]
def main(argv=None):
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
parser = argparse.ArgumentParser(
description=__doc__.splitlines()[0],
add_help=False,
allow_abbrev=False,
)
parser.add_argument(
"--help",
action="help",
help="show this help message and exit (no -h short option)",
)
parser.add_argument(
"--skill-dir",
required=True,
help="tsl-api-reference skill dir; codegen root defaults to "
"<skill-dir>/references/codegen and tsv to "
"<skill-dir>/data/function_index.tsv",
)
parser.add_argument(
"--root",
help="explicit codegen root, overriding the one derived from --skill-dir",
)
parser.add_argument(
"--tsv",
help="explicit output tsv, overriding the one derived from --skill-dir",
)
parser.add_argument(
"--check",
action="store_true",
help="verify the tsv matches the md tree; exit 1 if not (no write)",
)
args = parser.parse_args(argv)
skill_root = Path(args.skill_dir)
codegen_root = (
Path(args.root) if args.root else skill_root / "references" / "codegen"
)
if not codegen_root.is_dir():
print(f"ERROR: codegen root not found: {codegen_root}", file=sys.stderr)
return 1
tsv_path = (
Path(args.tsv)
if args.tsv
else skill_root / "data" / "function_index.tsv"
)
rows = build_rows(codegen_root)
new_text = render_tsv(rows)
if args.check:
if not tsv_path.is_file():
print(f"MISMATCH: tsv does not exist: {tsv_path}", file=sys.stderr)
return 1
current = tsv_path.read_text(encoding="utf-8")
tsv_ok = current == new_text
if tsv_ok:
print(f"OK: {tsv_path} matches md tree ({len(rows)} rows)")
return 0
if not tsv_ok:
cur_rows = read_tsv(tsv_path)[1:]
cur_keys = {tuple(r) for r in cur_rows}
new_keys = {tuple(r) for r in rows}
print(
f"MISMATCH: tsv out of date "
f"(tsv {len(cur_rows)} rows, md {len(rows)} rows; "
f"+{len(new_keys - cur_keys)} -{len(cur_keys - new_keys)}). "
f"Run without --check to rebuild.",
file=sys.stderr,
)
return 1
tsv_path.parent.mkdir(parents=True, exist_ok=True)
tsv_path.write_text(new_text, encoding="utf-8", newline="\n")
print(f"wrote {tsv_path}: {len(rows)} rows")
return 0
if __name__ == "__main__":
raise SystemExit(main())