Files
playbook/skills/tsl-syntax-reference/scripts/lookup.py
T
csh c6895fe41a 🐛 fix(tsl-syntax-reference): require exact long options in lookup cli
- drop the -h short option and disable long-option abbreviation: `--che`
  used to run `--check` and exit 0, so a misspelled argument silently
  performed a different action instead of failing
- document the lexicon/page alias validation under --check, now the only
  guard against a renamed page silently losing its natural language entry
- state that --query takes terms rather than the user's own words, and
  describe the weak-hit marker and its rc=2 in both --query help and epilog
2026-07-29 15:45:49 +08:00

1271 lines
46 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
import argparse
import difflib
import json
import re
import sys
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from urllib.parse import unquote
SKILL_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_REFERENCES_DIR = SKILL_ROOT / "references"
DEFAULT_LEXICON_PATH = SKILL_ROOT / "data" / "lexicon.json"
HEADING_RE = re.compile(r"^(#{1,6})(?!#)\s+(.+?)\s*$")
FENCE_RE = re.compile(r"^```([^`]*)$")
MARKDOWN_LINK_RE = re.compile(r"!?\[[^\]]*\]\(([^)]+)\)")
ASCII_TOKEN_RE = re.compile(r"[a-z_][a-z0-9_.$:+-]*", re.IGNORECASE)
CHINESE_RUN_RE = re.compile(r"[\u3400-\u9fff]+")
_WHITESPACE_RE = re.compile(r"\s+")
INLINE_CODE_RE = re.compile(r"`([^`\n]+)`")
FENCED_CODE_RE = re.compile(r"```[^\n]*\n(.*?)```", re.DOTALL)
# 章节级检索标签:写在标题下方的 HTML 注释里,渲染后不可见。
# 用途是补用户侧说法与正文词面之间的缺口(正文写 `var` / `operator[]`
# 用户说"传引用" / "中括号")。逗号或顿号分隔,只影响检索,不是事实正文。
SECTION_TAG_RE = re.compile(r"<!--\s*tags?\s*:\s*(.*?)\s*-->", re.DOTALL | re.IGNORECASE)
TAG_SEPARATOR_RE = re.compile(r"[,,、]\s*")
IDENTITY_PREFIX = "代码块身份:"
BLOCK_DESCRIPTION_PREFIX = "代码块说明:"
ALLOWED_IDENTITIES = {
"可直接照写示例",
"反例 / 不可照写",
"输出片段",
"配置片段 / 概念骨架",
"仅服务端可执行示例",
}
ROUTER_PHRASES = ("路由中心", "选择一个主专题", "候选页继续判断")
EXCLUDED_REFERENCE_FILES = {"index.md"}
DUTY_HEADING = "本篇职责"
# 这些标题在多页重复出现,不承载单一事实,不参与「必须有 tag」的约束。
GENERIC_HEADINGS = frozenset(
{
"本篇职责",
"核心规则",
"禁止项",
"可直接照写示例",
"默认生成模板",
"本页不生成的范围",
"示例与行为",
"决策边界和禁止项",
"文件模型示例",
"术语对照",
}
)
SUSPICIOUS_FENCE_RE = re.compile(r"^(?:\s+`{3}|`{4,})")
WRITE_PRELUDE_ANCHORS = (
("02_core_model.md", "文件模型核心规则"),
("01_quickstart.md", "语言核心事实速查"),
)
HEADING_TOKEN_SCORE = 12
HEADING_EXACT_SCORE = 20
# 标签是人工策展的检索意图,权重与标识符同级:足以在页内区分章节,
# 但压不过标题精确命中,避免标签写宽了就绑架整页。
TAG_TOKEN_SCORE = 10
TAG_EXACT_SCORE = 16
# 标签按「被查询覆盖的比例」判命中,而不是逐 token 累加。中文按 2-gram 切分,
# 逐 token 累加会让「参数」「函数」这类泛化词命中一整页的标签,把整页抬起来;
# 要求覆盖过半,则「只读参数」不会被「临时改系统参数」点亮,而单 token 的
# 精确标签(lambda)仍然 100% 覆盖、照常命中。
TAG_COVERAGE_THRESHOLD = 0.5
TERM_TOKEN_SCORE = 10
TERM_EXACT_SCORE = 16
PAGE_TITLE_TOKEN_SCORE = 5
PAGE_TITLE_EXACT_SCORE = 8
BODY_TOKEN_SCORE = 3
BODY_EXACT_SCORE = 4
DIRECT_EXAMPLE_BOOST = 8
COUNTEREXAMPLE_BOOST = 30
EXACT_ERROR_BOOST = 14
MIXED_QUERY_MIN_SCORE = 10
PAGE_INTENT_SCORE = 80
# 单页最多贡献几条候选。上限保证候选跨页分散,但页级 intent 命中(+80)会把
# 整页抬起来,页内只剩十几分的词法差异在排序;上限过小时正确章节会被同页
# 邻居挤掉,且加大 --limit 也救不回来。3 是实测下节级准确率与跨页分散的平衡点。
PAGE_MATCH_CAP = 3
CHINESE_STOP_TOKENS = {
"一个",
"为什",
"什么",
"怎么",
"怎样",
"是否",
"能不",
"不能",
"帮我",
"想要",
"然后",
"里面",
}
ASCII_FILTER_STOP_TOKENS = {
"debug",
"please",
"program",
"tinysoft",
"tsl",
"tsf",
}
QUERY_SYNONYMS: dict[str, tuple[str, ...]]
PAGE_INTENT_ALIASES: dict[str, tuple[str, ...]]
def _load_lexicon(
path: Path = DEFAULT_LEXICON_PATH,
) -> tuple[dict[str, tuple[str, ...]], dict[str, tuple[str, ...]]]:
"""加载策展词表(口语同义词与页级意图短语)。
词表是持续生长的策展数据,与检索引擎分离维护在 data/lexicon.json
策展纪律见 data/README.md。加载失败必须响亮报错——静默回退为空表
会让全部自然语言入口消失而检索仍然"正常"返回。
"""
try:
data = json.loads(path.read_text(encoding="utf-8"))
except FileNotFoundError:
raise SystemExit(f"词表文件缺失:{path};检查 skill 安装是否完整")
except json.JSONDecodeError as error:
raise SystemExit(f"词表文件不是合法 JSON{path}{error}")
def _table(name: str) -> dict[str, tuple[str, ...]]:
table = data.get(name)
if not isinstance(table, dict):
raise SystemExit(f"{path}: 缺少 {name} 表或不是对象")
result: dict[str, tuple[str, ...]] = {}
for key, values in table.items():
if (
not isinstance(values, list)
or not values
or not all(isinstance(item, str) and item.strip() for item in values)
):
raise SystemExit(f"{path}: {name}[{key!r}] 必须是非空字符串数组")
result[key] = tuple(values)
return result
return _table("query_synonyms"), _table("page_intent_aliases")
QUERY_SYNONYMS, PAGE_INTENT_ALIASES = _load_lexicon()
@dataclass(frozen=True)
class Section:
id: str
page: Path
page_title: str
heading_path: tuple[str, ...]
body: str
local_body: str
identities: tuple[str, ...]
tags: tuple[str, ...]
searchable_text: str
@dataclass(frozen=True)
class ValidationProblem:
page: Path
line: int
message: str
@dataclass(frozen=True)
class QueryMatch:
section: Section
score: int
priority: tuple[int, ...] = ()
reasons: tuple[str, ...] = ()
weak: bool = False
@dataclass(frozen=True)
class ScoreBreakdown:
intent: int
heading_path: int
exact_term: int
tag: int
page_title: int
body: int
mode_boost: int
synonym_hits: int
diagnose_priority: bool = False
@property
def lexical_total(self) -> int:
return (
self.intent
+ self.heading_path
+ self.exact_term
+ self.tag
+ self.page_title
+ self.body
)
@property
def weak(self) -> bool:
# 没有任何强字段命中(意图短语、标题、标识符、标签),只靠正文 /
# 页标题的低分撞词进入候选。实测这是"查询根本不在本 skill 事实域"
# 时的典型形态(如通用词二字撞上正文),而策展用例从不落进来;
# 全部候选皆弱时按无匹配处理。
return self.intent + self.heading_path + self.exact_term + self.tag == 0
@property
def total(self) -> int:
return self.lexical_total + self.mode_boost
@property
def priority(self) -> tuple[int, ...]:
if self.diagnose_priority:
return (
self.intent,
self.mode_boost,
self.heading_path,
self.exact_term,
self.tag,
self.page_title,
self.body,
)
return (
self.intent,
self.heading_path,
self.exact_term,
self.tag,
self.page_title,
self.mode_boost,
self.body,
)
@dataclass
class QueryResult:
query: str
mode: str
matches: list[QueryMatch]
prelude: list[Section]
limit: int
def normalize(value: str) -> str:
return unicodedata.normalize("NFKC", value).casefold()
SYMBOL_SLUG_REPLACEMENTS = (
("**", " double-star "),
("[]", " index "),
("::", " double-colon "),
(":.", " colon-dot "),
("*", " star "),
)
def _slug(value: str) -> str:
normalized = normalize(value)
for symbol, replacement in SYMBOL_SLUG_REPLACEMENTS:
normalized = normalized.replace(symbol, replacement)
slug = re.sub(r"[^\w]+", "-", normalized, flags=re.UNICODE).strip("-_")
return slug or "section"
def section_id(relative_page: Path | str, heading_path: tuple[str, ...]) -> str:
page = Path(relative_page)
parts = [_slug(page.with_suffix("").as_posix()), *(_slug(item) for item in heading_path)]
return "--".join(parts)
def _reference_pages(references_dir: Path) -> list[Path]:
return [
page
for page in sorted(references_dir.glob("*.md"), key=lambda item: item.name)
if page.name not in EXCLUDED_REFERENCE_FILES
]
def _heading_records(lines: list[str]) -> tuple[str, list[tuple[int, int, str]]]:
page_title = ""
records: list[tuple[int, int, str]] = []
in_fence = False
for index, line in enumerate(lines):
if FENCE_RE.match(line.rstrip("\r\n")):
in_fence = not in_fence
continue
if in_fence:
continue
match = HEADING_RE.match(line.rstrip("\r\n"))
if not match:
continue
level = len(match.group(1))
title = match.group(2)
if level == 1 and not page_title:
page_title = title
elif level in (2, 3, 4):
records.append((index, level, title))
return page_title, records
def _associated_identity(lines: list[str], opening_fence: int) -> str | None:
previous = opening_fence - 1
while previous >= 0 and not lines[previous].strip():
previous -= 1
while previous >= 0 and lines[previous].strip().startswith(BLOCK_DESCRIPTION_PREFIX):
previous -= 1
while previous >= 0 and not lines[previous].strip():
previous -= 1
if previous < 0:
return None
metadata = lines[previous].strip()
if not metadata.startswith(IDENTITY_PREFIX):
return None
identity = metadata[len(IDENTITY_PREFIX) :].strip()
earlier = previous - 1
while earlier >= 0 and not lines[earlier].strip():
earlier -= 1
if earlier >= 0 and lines[earlier].strip().startswith(IDENTITY_PREFIX):
return None
return identity if identity in ALLOWED_IDENTITIES else None
def _identities(body: str) -> tuple[str, ...]:
identities: list[str] = []
lines = body.splitlines()
in_fence = False
for index, line in enumerate(lines):
if not FENCE_RE.match(line):
continue
if not in_fence:
identity = _associated_identity(lines, index)
if identity is not None:
identities.append(identity)
in_fence = not in_fence
return tuple(identities)
def _section_tags(body: str) -> tuple[str, ...]:
tags: list[str] = []
for block in SECTION_TAG_RE.findall(body):
for tag in TAG_SEPARATOR_RE.split(block.replace("\n", " ")):
tag = tag.strip()
if tag and tag not in tags:
tags.append(tag)
return tuple(tags)
def load_sections(references_dir: Path = DEFAULT_REFERENCES_DIR) -> list[Section]:
sections: list[Section] = []
for page in _reference_pages(Path(references_dir)):
text = page.read_text(encoding="utf-8")
lines = text.splitlines(keepends=True)
page_title, headings = _heading_records(lines)
heading_stack: dict[int, str] = {}
for position, (start, level, title) in enumerate(headings):
for stacked_level in tuple(heading_stack):
if stacked_level >= level:
del heading_stack[stacked_level]
heading_stack[level] = title
heading_path = tuple(
heading_stack[item]
for item in range(2, level + 1)
if item in heading_stack
)
end = len(lines)
for next_start, next_level, _ in headings[position + 1 :]:
if next_level <= level:
end = next_start
break
body = "".join(lines[start:end])
local_end = (
headings[position + 1][0]
if position + 1 < len(headings)
else len(lines)
)
local_body = "".join(lines[start:local_end])
base_id = section_id(page.relative_to(references_dir), heading_path)
tags = _section_tags(local_body)
searchable_text = normalize(
"\n".join((page.stem, page_title, *heading_path, *tags, local_body))
)
sections.append(
Section(
id=base_id,
page=page,
page_title=page_title,
heading_path=heading_path,
body=body,
local_body=local_body,
identities=_identities(local_body),
tags=tags,
searchable_text=searchable_text,
)
)
return sections
def _identity_problems(page: Path, lines: list[str]) -> list[ValidationProblem]:
problems: list[ValidationProblem] = []
for index, line in enumerate(lines, start=1):
stripped = line.strip()
if stripped.startswith(IDENTITY_PREFIX):
identity = stripped[len(IDENTITY_PREFIX) :].strip()
if identity not in ALLOWED_IDENTITIES:
problems.append(ValidationProblem(page, index, f"未知身份:{identity}"))
in_fence = False
for index, line in enumerate(lines):
if FENCE_RE.match(line):
if in_fence:
in_fence = False
else:
if _associated_identity(lines, index) is None:
problems.append(
ValidationProblem(
page, index + 1, "每个代码围栏必须关联恰好一个代码块身份"
)
)
in_fence = True
continue
if not in_fence and SUSPICIOUS_FENCE_RE.match(line):
problems.append(
ValidationProblem(
page, index + 1, "不支持的代码围栏形态(缩进围栏或四个及以上反引号)"
)
)
if in_fence:
problems.append(ValidationProblem(page, len(lines), "代码围栏未闭合"))
return problems
def _local_link_problems(
page: Path, text: str, references_dir: Path
) -> list[ValidationProblem]:
problems: list[ValidationProblem] = []
# Blank out fenced/inline code without changing line offsets, so reported
# line numbers line up with the original file.
without_fences = re.sub(
r"```.*?```",
lambda match: re.sub(r"[^\n]", " ", match.group(0)),
text,
flags=re.DOTALL,
)
searchable_markdown = re.sub(
r"`[^`\n]*`", lambda match: " " * len(match.group(0)), without_fences
)
for match in MARKDOWN_LINK_RE.finditer(searchable_markdown):
target = match.group(1).strip().split(maxsplit=1)[0].strip("<>")
if target.startswith(("#", "http://", "https://", "mailto:")):
continue
target_path = unquote(target.split("#", 1)[0].replace("\\", "/"))
resolved = (page.parent / target_path).resolve()
try:
resolved.relative_to(references_dir.resolve())
except ValueError:
exists = False
else:
exists = resolved.is_file()
if not exists:
line = searchable_markdown.count("\n", 0, match.start()) + 1
problems.append(ValidationProblem(page, line, f"本地链接不存在:{target}"))
return problems
def _tag_problems(sections: list[Section]) -> list[ValidationProblem]:
"""章节 tag 的结构校验。
tag 是页内区分章节的主要信号,写空、漏写或页内重复都会静默削弱检索,
而其余校验一概发现不了。含代码围栏的具体章节是事实落点,必须可被
自然语言命中;纯交接说明(正文只指向别页、没有围栏)反而不该有 tag,
否则会和真正拥有事实的那一页抢候选。
"""
problems: list[ValidationProblem] = []
seen_per_page: dict[str, dict[str, str]] = {}
for section in sections:
if not section.heading_path:
continue
heading = section.heading_path[-1]
has_fence = bool(FENCED_CODE_RE.search(section.local_body))
if has_fence and heading not in GENERIC_HEADINGS and not section.tags:
problems.append(
ValidationProblem(
section.page, 1, f"含代码围栏的章节缺少检索 tag:{heading}"
)
)
for tag in section.tags:
if not tag.strip():
problems.append(
ValidationProblem(section.page, 1, f"空 tag{heading}")
)
continue
owners = seen_per_page.setdefault(section.page.name, {})
if tag in owners:
problems.append(
ValidationProblem(
section.page,
1,
f"页内 tag 重复:「{tag}」同时属于「{owners[tag]}」和「{heading}」",
)
)
else:
owners[tag] = heading
return problems
def validate_references(
references_dir: Path = DEFAULT_REFERENCES_DIR,
) -> list[ValidationProblem]:
references_dir = Path(references_dir)
problems: list[ValidationProblem] = []
# 逐页校验在零页时全部静默通过;参考页缺失属于安装/路径错误,必须报错,
# 否则 --check 会为一个空目录返回成功。
if not _reference_pages(references_dir):
problems.append(
ValidationProblem(
references_dir, 1, "references 中没有可校验的参考页;检查路径或重新安装 skill"
)
)
return problems
index_page = references_dir / "index.md"
if index_page.exists():
problems.append(ValidationProblem(index_page, 1, "references 中不得保留 index.md"))
for page in sorted(references_dir.glob("*.md"), key=lambda item: item.name):
text = page.read_text(encoding="utf-8")
lines = text.splitlines()
problems.extend(_identity_problems(page, lines))
problems.extend(_local_link_problems(page, text, references_dir))
page_title, headings = _heading_records(text.splitlines(keepends=True))
h1_count = 0
in_fence = False
previous_level = 1
for index, line in enumerate(lines):
if FENCE_RE.match(line):
in_fence = not in_fence
continue
if in_fence:
continue
match = HEADING_RE.match(line)
if match is None:
continue
level = len(match.group(1))
if level == 1:
h1_count += 1
previous_level = 1
continue
if level not in (2, 3, 4):
continue
if level > previous_level + 1:
problems.append(
ValidationProblem(page, index + 1, "H2/H3/H4 标题层级跳跃")
)
previous_level = level
if h1_count != 1 or not page_title:
problems.append(ValidationProblem(page, 1, "每页必须有且仅有一个非空 H1"))
duty_sections = []
for position, (start, level, title) in enumerate(headings):
if level != 2 or title != DUTY_HEADING:
continue
end = len(lines)
for next_start, next_level, _ in headings[position + 1 :]:
if next_level <= level:
end = next_start
break
body = "\n".join(lines[start + 1 : end]).strip()
duty_sections.append(body)
if len(duty_sections) != 1 or not duty_sections[0]:
problems.append(
ValidationProblem(
page, 1, "每页必须有且仅有一个非空「本篇职责」"
)
)
for phrase in ROUTER_PHRASES:
for index, line in enumerate(lines, start=1):
if phrase in line:
problems.append(
ValidationProblem(page, index, f"包含人工路由协议:{phrase}")
)
sections = load_sections(references_dir)
problems.extend(_tag_problems(sections))
ids: dict[str, Section] = {}
for section in sections:
if section.id in ids:
problems.append(ValidationProblem(section.page, 1, f"重复 section ID{section.id}"))
ids[section.id] = section
for page_name, heading in WRITE_PRELUDE_ANCHORS:
page_sections = [
section for section in sections if section.page.name == page_name
]
# Only enforce the anchor when the page is present, so validating a
# synthetic references dir (tests) does not demand the bundled pages.
if page_sections and not any(
heading in section.heading_path for section in page_sections
):
problems.append(
ValidationProblem(
references_dir / page_name,
1,
f"write 模式前置章节缺失:{page_name} 的「{heading}」",
)
)
# 概念地图逐页从「本篇职责」段生成;有该段的页必须产出非空摘要,
# 否则某页职责段被清空/写坏时地图会静默缺页。
mapped_pages = {page_name for page_name, _, _ in build_concept_map(references_dir)}
reference_pages = {page.name for page in _reference_pages(references_dir)}
if mapped_pages != reference_pages:
missing = ", ".join(sorted(reference_pages - mapped_pages)) or "none"
problems.append(
ValidationProblem(
references_dir,
1,
f"概念地图页数与参考页不一致;缺失:{missing}",
)
)
for section in sections:
if section.heading_path != (DUTY_HEADING,):
continue
if section.page.name not in mapped_pages:
problems.append(
ValidationProblem(
section.page,
1,
f"概念地图摘要为空:{section.page.name} 的「{DUTY_HEADING}」",
)
)
# 词表校验只对内置参考目录有意义:alias 键指向的是内置页文件名,
# 用 --references-dir 校验合成目录(测试)时跳过,避免整表误报。
if references_dir.resolve() == DEFAULT_REFERENCES_DIR.resolve():
page_names = {page.name for page in _reference_pages(references_dir)}
for stale in sorted(set(PAGE_INTENT_ALIASES) - page_names):
problems.append(
ValidationProblem(
DEFAULT_LEXICON_PATH,
1,
f"page_intent_aliases 指向不存在的参考页:{stale}",
)
)
for missing in sorted(page_names - set(PAGE_INTENT_ALIASES)):
problems.append(
ValidationProblem(
DEFAULT_LEXICON_PATH,
1,
f"参考页缺少 page_intent_aliases 自然语言入口:{missing}",
)
)
return problems
def _base_query_tokens(text: str) -> set[str]:
normalized = normalize(text)
tokens = set(ASCII_TOKEN_RE.findall(normalized))
for run in CHINESE_RUN_RE.findall(normalized):
tokens.add(run)
tokens.update(run[index : index + 2] for index in range(len(run) - 1))
return {
token
for token in tokens
if token.strip() and token not in CHINESE_STOP_TOKENS
}
def _ascii_token_sequence(text: str) -> list[str]:
return [
token.rstrip(".$:+-")
for token in ASCII_TOKEN_RE.findall(normalize(text))
if token.rstrip(".$:+-")
]
def _query_contains_phrase(query: str, phrase: str) -> bool:
normalized_phrase = normalize(phrase)
if CHINESE_RUN_RE.search(normalized_phrase):
# 含中文的短语按去空白后的串比较。SKILL.md 要求智能体传「术语」而不是
# 用户原话,术语常以空格分隔(「数组 下标 起点」),逐字子串匹配会
# 整条落空;去空白后 phrase 仍要求连续出现,不放宽词序。
return _WHITESPACE_RE.sub("", normalized_phrase) in _WHITESPACE_RE.sub(
"", normalize(query)
)
phrase_tokens = _ascii_token_sequence(phrase)
query_tokens_in_order = _ascii_token_sequence(query)
if not phrase_tokens:
return False
width = len(phrase_tokens)
return any(
query_tokens_in_order[index : index + width] == phrase_tokens
for index in range(len(query_tokens_in_order) - width + 1)
)
def _synonym_tokens(text: str) -> set[str]:
tokens: set[str] = set()
for phrase, synonyms in QUERY_SYNONYMS.items():
if not _query_contains_phrase(text, phrase):
continue
for synonym in synonyms:
tokens.update(_base_query_tokens(synonym))
return tokens
def query_tokens(text: str) -> set[str]:
return _base_query_tokens(text) | _synonym_tokens(text)
def _text_contains_token(text: str, token: str) -> bool:
normalized_text = normalize(text)
if ASCII_TOKEN_RE.fullmatch(token):
text_tokens = _bare_tokens(set(ASCII_TOKEN_RE.findall(normalized_text)))
return token.rstrip(".$:+-") in text_tokens
return token in normalized_text
def _text_contains_exact_query(text: str, query: str) -> bool:
normalized_query = normalize(query).strip()
if ASCII_TOKEN_RE.fullmatch(normalized_query):
return _text_contains_token(text, normalized_query)
return bool(normalized_query and normalized_query in normalize(text))
def _intent_score(section: Section, query: str) -> int:
aliases = PAGE_INTENT_ALIASES.get(section.page.name, ())
return PAGE_INTENT_SCORE * sum(
_query_contains_phrase(query, alias) for alias in aliases
)
def _has_chinese_context(section: Section, query: str) -> bool:
runs = CHINESE_RUN_RE.findall(normalize(query))
for run in runs:
tokens = (
[run]
if len(run) < 2
else [run[index : index + 2] for index in range(len(run) - 1)]
)
matched = sum(token in section.searchable_text for token in tokens)
if matched >= (len(tokens) + 1) // 2:
return True
return not runs
def _tag_matched_tokens(tags: tuple[str, ...], query_token_set: set[str]) -> int:
"""标签命中的 token 数;覆盖率不过门槛的标签整条不计分。
覆盖率只做门控,计分仍按命中 token 数——否则一条深度吻合的标签
(命中 4 个 token)和一条勉强擦边的标签得分相同,信号被抹平。
"""
matched = 0
for tag in tags:
tag_tokens = _base_query_tokens(tag)
if not tag_tokens:
continue
covered = sum(1 for token in tag_tokens if token in query_token_set)
if covered / len(tag_tokens) >= TAG_COVERAGE_THRESHOLD:
matched += covered
return matched
def _code_text(body: str) -> str:
inline = INLINE_CODE_RE.findall(body)
fenced = FENCED_CODE_RE.findall(body)
return normalize("\n".join((*inline, *fenced)))
def _score_section(section: Section, query: str, mode: str) -> ScoreBreakdown:
normalized_query = normalize(query).strip()
tokens = query_tokens(query)
heading_text = normalize("\n".join(section.heading_path))
page_title_text = normalize(section.page_title)
body_text = normalize(SECTION_TAG_RE.sub(" ", section.local_body))
tag_text = normalize("\n".join(section.tags))
term_text = _code_text(section.local_body)
expanded_only_tokens = _synonym_tokens(query) - _base_query_tokens(query)
synonym_hits = sum(
any(
_text_contains_token(text, token)
for text in (heading_text, term_text, tag_text, page_title_text, body_text)
)
for token in expanded_only_tokens
)
heading_score = 0
term_score = 0
tag_score = TAG_TOKEN_SCORE * _tag_matched_tokens(section.tags, tokens)
page_title_score = 0
body_score = 0
for token in tokens:
if _text_contains_token(heading_text, token):
heading_score += HEADING_TOKEN_SCORE
if _text_contains_token(term_text, token):
term_score += TERM_TOKEN_SCORE
if _text_contains_token(page_title_text, token):
page_title_score += PAGE_TITLE_TOKEN_SCORE
if _text_contains_token(body_text, token):
body_score += BODY_TOKEN_SCORE
if _text_contains_exact_query(heading_text, query):
heading_score += HEADING_EXACT_SCORE
if _text_contains_exact_query(term_text, query):
term_score += TERM_EXACT_SCORE
if section.tags and any(
normalize(tag) == normalize(query).strip() for tag in section.tags
):
tag_score += TAG_EXACT_SCORE
if _text_contains_exact_query(page_title_text, query):
page_title_score += PAGE_TITLE_EXACT_SCORE
if _text_contains_exact_query(body_text, query):
body_score += BODY_EXACT_SCORE
mode_boost = 0
if mode == "write" and "可直接照写示例" in section.identities:
mode_boost += DIRECT_EXAMPLE_BOOST
if mode == "diagnose":
# 反例按代码块身份加分,不按页名。反例分散在各专题页里,
# 没有一页专门收口它们。
if "反例 / 不可照写" in section.identities:
mode_boost += COUNTEREXAMPLE_BOOST
if normalized_query and normalized_query in section.searchable_text:
mode_boost += EXACT_ERROR_BOOST
return ScoreBreakdown(
intent=_intent_score(section, query),
heading_path=heading_score,
exact_term=term_score,
tag=tag_score,
page_title=page_title_score,
body=body_score,
mode_boost=mode_boost,
synonym_hits=synonym_hits,
diagnose_priority=mode == "diagnose",
)
def _score_reasons(score: ScoreBreakdown) -> tuple[str, ...]:
components = (
("intent", score.intent),
("heading", score.heading_path),
("identifier", score.exact_term),
("tag", score.tag),
("page_title", score.page_title),
("body", score.body),
("mode", score.mode_boost),
)
reasons = tuple(f"{name}={value}" for name, value in components if value)
if score.synonym_hits:
reasons += (f"synonym={score.synonym_hits}",)
return reasons
def _write_prelude(sections: list[Section]) -> list[Section]:
prelude: list[Section] = []
for page_name, heading in WRITE_PRELUDE_ANCHORS:
match = next(
(
section
for section in sections
if section.page.name == page_name and heading in section.heading_path
),
None,
)
if match is not None:
prelude.append(match)
return prelude
STRING_LITERAL_RE = re.compile(r'"[^"\n]*"|\'[^\'\n]*\'')
def _bare_tokens(tokens: set[str]) -> set[str]:
# ASCII_TOKEN_RE 的连续字符类会把尾缀符号吞进 token(如编译开关
# `{$varByRef-}` 产出 `varbyref-`);按剥掉尾部符号的裸形态比较。
return {token.rstrip(".$:+-") for token in tokens} - {""}
def _ascii_anchor(section: Section, ascii_tokens: set[str]) -> bool:
# 只有足够长的标识符按标识符位置命中(标题、行内代码或剥离字符串
# 字面量后的围栏代码)才豁免中文门控;短 token 与字符串样例数据
# (如 "XYZ"、"xyz*")不算点名,避免样例值驱动召回。
anchor_tokens = {
token for token in _bare_tokens(ascii_tokens) if len(token) >= 4
}
if not anchor_tokens:
return False
heading_tokens = set(
ASCII_TOKEN_RE.findall(normalize("\n".join(section.heading_path)))
)
code_text = "\n".join(
(
*INLINE_CODE_RE.findall(section.local_body),
*FENCED_CODE_RE.findall(section.local_body),
)
)
code_tokens = set(
ASCII_TOKEN_RE.findall(normalize(STRING_LITERAL_RE.sub(" ", code_text)))
)
return bool(anchor_tokens & _bare_tokens(heading_tokens | code_tokens))
def _related_sections(a: Section, b: Section) -> bool:
if a.page != b.page or len(a.heading_path) == len(b.heading_path):
return False
shorter, longer = sorted((a.heading_path, b.heading_path), key=len)
return longer[: len(shorter)] == shorter
def query_sections(
query: str,
mode: str,
limit: int = 5,
references_dir: Path = DEFAULT_REFERENCES_DIR,
) -> QueryResult:
if mode not in {"write", "diagnose", "explain"}:
raise ValueError(f"unsupported mode: {mode}")
if not 1 <= limit <= 10:
raise ValueError("limit must be between 1 and 10")
all_sections = load_sections(references_dir)
sections = all_sections
prelude = _write_prelude(all_sections) if mode == "write" else []
required_ids = {section.id for section in prelude}
raw_ascii_tokens = {
token
for token in _base_query_tokens(query)
if ASCII_TOKEN_RE.fullmatch(token)
and token not in ASCII_FILTER_STOP_TOKENS
}
ascii_tokens = {
token
for token in raw_ascii_tokens
if any(_ascii_anchor(section, {token}) for section in all_sections)
}
if ascii_tokens:
sections = [
section
for section in sections
if any(
_text_contains_token(section.searchable_text, token)
for token in ascii_tokens
)
]
has_chinese = bool(CHINESE_RUN_RE.search(normalize(query)))
minimum_score = MIXED_QUERY_MIN_SCORE if ascii_tokens and has_chinese else 1
ranked: list[QueryMatch] = []
for section in sections:
# 中文上下文门控只裁剪正文级 ASCII 噪声;标识符在标题或代码里
# 精确命中的 section 不因中文措辞不同而被丢弃。
if (
ascii_tokens
and has_chinese
and not _ascii_anchor(section, ascii_tokens)
and not _has_chinese_context(section, query)
):
continue
score = _score_section(section, query, mode)
if score.lexical_total < minimum_score:
continue
ranked.append(
QueryMatch(
section,
score.total,
score.priority,
_score_reasons(score),
score.weak,
)
)
ranked.sort(
key=lambda match: (
-match.score,
*(-value for value in match.priority),
match.section.page.as_posix(),
match.section.id,
)
)
# H2 聚合 section 的正文逐字包含其 H3 子节;父子同时入选时只保留
# 排名更高的一个,避免同一内容重复返回。
matches: list[QueryMatch] = []
page_counts: dict[Path, int] = {}
for match in ranked:
if match.section.id in required_ids:
continue
if any(
_related_sections(match.section, kept.section) for kept in matches
):
continue
if page_counts.get(match.section.page, 0) >= PAGE_MATCH_CAP:
continue
matches.append(match)
page_counts[match.section.page] = page_counts.get(match.section.page, 0) + 1
if len(matches) == limit:
break
return QueryResult(
query=query,
mode=mode,
matches=matches,
prelude=prelude,
limit=limit,
)
def _logical_source(section: Section) -> str:
return f"references/{section.page.name}"
def _safe_json_string(value: str) -> str:
encoded = json.dumps(value, ensure_ascii=False)
for separator in ("\u0085", "\u2028", "\u2029"):
encoded = encoded.replace(separator, f"\\u{ord(separator):04x}")
return encoded
def _plain_text_summary(body: str, limit: int = 180) -> str:
# 标签是检索元数据,不是事实正文;不能泄进候选摘要。
without_tags = SECTION_TAG_RE.sub(" ", body)
without_fences = FENCED_CODE_RE.sub(" ", without_tags)
without_links = re.sub(
r"!?\[([^\]]*)\]\([^)]+\)", lambda match: match.group(1), without_fences
)
without_inline_code = INLINE_CODE_RE.sub(lambda match: match.group(1), without_links)
content_lines = []
for line in without_inline_code.splitlines():
stripped = line.strip()
if not stripped or stripped.startswith(("#", IDENTITY_PREFIX, BLOCK_DESCRIPTION_PREFIX)):
continue
content_lines.append(stripped.lstrip("-* "))
summary = re.sub(r"\s+", " ", " ".join(content_lines)).strip()
if len(summary) <= limit:
return summary
return summary[: limit - 1].rstrip() + "…"
def render_candidates(result: QueryResult) -> str:
lines = [
"# TSL Syntax Candidates",
"",
f"Mode: `{result.mode}`",
f"Query: {_safe_json_string(result.query)}",
]
candidates = [
(section, 0, True, ("required=1",), False) for section in result.prelude
] + [
(match.section, match.score, False, match.reasons, match.weak)
for match in result.matches
]
# --limit is the budget for query matches only; write 模式的前置章节额外附加,
# 否则 limit 小于前置章节数时会一条真实候选都不返回。
budget = result.limit + len(result.prelude)
for index, (section, score, required, reasons, weak) in enumerate(
candidates[:budget], start=1
):
lines.extend(
[
"",
f"## Candidate {index}",
"",
f"Score: {score}",
f"Required: {'yes' if required else 'no'}",
]
)
if weak:
# 只在弱命中时输出该行:没有 Weak 行即为强命中。
lines.append("Weak: yes")
lines.extend(
[
f"Section ID: `{section.id}`",
f"Source: `{_logical_source(section)}`",
f"Heading: `{' > '.join(section.heading_path)}`",
f"Why: `{', '.join(reasons) or 'lexical=1'}`",
f"Summary: {_plain_text_summary(section.local_body)}",
]
)
return "\n".join(lines).rstrip() + "\n"
def render_section(section: Section) -> str:
lines = [
"# TSL Syntax Section",
"",
f"Section ID: `{section.id}`",
f"Source: `{_logical_source(section)}`",
"",
section.body.rstrip(),
]
return "\n".join(lines).rstrip() + "\n"
def build_concept_map(references_dir: Path = DEFAULT_REFERENCES_DIR) -> list[tuple[str, str, str]]:
entries: list[tuple[str, str, str]] = []
seen_pages: set[str] = set()
for section in load_sections(references_dir):
page_name = section.page.name
if page_name in seen_pages:
continue
if section.heading_path != (DUTY_HEADING,):
continue
summary = _plain_text_summary(section.local_body, limit=600)
if not summary:
continue
seen_pages.add(page_name)
entries.append((page_name, section.page_title, summary))
entries.sort(key=lambda item: item[0])
return entries
def render_concept_map(entries: list[tuple[str, str, str]]) -> str:
lines = [
"# TSL 概念地图",
"",
"把自然语言需求映射到该查哪个 TSL 概念,随后仍用 --query 获取候选、用 --section 取回事实正文。",
"本清单不含可照写语法,也不替代精确章节取回。",
]
for page_name, page_title, summary in entries:
lines.extend(["", f"## {page_title}", "", summary])
return "\n".join(lines).rstrip() + "\n"
def _configure_utf8() -> None:
for stream in (sys.stdout, sys.stderr):
reconfigure = getattr(stream, "reconfigure", None)
if reconfigure is not None:
reconfigure(encoding="utf-8")
HELP_EPILOG = """\
检索分两步,缺一步都不算取回事实:
1. 先取候选(只有摘要和 Section ID,不含事实正文)
lookup.py --query "命名参数 默认参数" --mode write
2. 再按候选里的 Section ID 取回正文;多个要素各自跑完第 1 步后,
选定的 Section ID 可以合并成一次取回
lookup.py --section "05_functions_and_calls--可直接照写示例--基础函数-过程骨架" \\
"02_core_model--文件模型核心规则"
查询词无从下手时先 --map 把需求映射到 TSL 概念;改动参考页或 data/ 词表后用 --check 校验。
弱命中(候选标 Weak: yes)没有意图/标题/标识符/标签命中,只靠正文低分撞词;
全部候选皆弱时视同无匹配并返回 rc=2,应改进查询词重试而不是从弱候选里挑。
"""
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="检索 TSL 语法参考页,取回可照写的语法事实。",
epilog=HELP_EPILOG,
formatter_class=argparse.RawDescriptionHelpFormatter,
# 只接受完整拼写的长选项:不提供 -h 短选项(add_help=False 后手工挂
# --help),也不接受缩写。缩写会让打错的参数静默命中另一个动作——
# `--che` 曾等同 `--check` 并以 rc=0 返回,看不出参数写错了。
add_help=False,
allow_abbrev=False,
)
parser.add_argument(
"--help",
action="help",
help="显示本帮助并退出(不提供 -h 短选项)",
)
action = parser.add_argument_group("动作(必选其一)").add_mutually_exclusive_group(
required=True
)
action.add_argument(
"--query",
help="按术语、报错原文或语法要素名检索候选章节(不要传用户原话);需配合 "
"--mode。候选标 `Weak: yes` 表示只有正文低分撞词,全部候选皆弱时退出码为 2",
)
action.add_argument(
"--section",
nargs="+",
metavar="SECTION_ID",
help="按 Section ID 取回章节正文,可一次传多个 ID 批量取回"
"ID 抄自 --query 输出);这是唯一的事实来源",
)
action.add_argument(
"--map",
dest="show_map",
action="store_true",
help="输出各专题的职责摘要,用于把自然语言需求映射到 TSL 概念;不含可照写事实",
)
action.add_argument(
"--check",
action="store_true",
help="校验参考页的结构、代码块身份、本地链接,以及 data/lexicon.json 的页级"
"意图短语与参考页是否一一对应;发现问题时退出码为 1",
)
parser.add_argument(
"--mode",
choices=("write", "diagnose", "explain"),
help="检索意图,仅用于 --query"
"write 编写或修改代码,额外附加文件模型与核心事实速查;"
"diagnose 定位语法错误,优先易错点与反例;"
"explain 解释语言规则或代码含义",
)
parser.add_argument(
"--limit",
type=int,
default=5,
help="--query 返回的候选条数上限,取值 1..10(默认 %(default)s);"
"write 模式的前置章节不占该预算",
)
parser.add_argument(
"--references-dir",
type=Path,
default=DEFAULT_REFERENCES_DIR,
metavar="DIR",
help="参考页目录(默认为本 skill 内置的 references/",
)
return parser
def _nearest_section_ids(
requested: str, sections: list[Section], limit: int = 5
) -> list[str]:
ranked = sorted(
sections,
key=lambda section: (
-difflib.SequenceMatcher(None, requested, section.id).ratio(),
section.id,
),
)
return [section.id for section in ranked[:limit]]
def main(argv: list[str] | None = None) -> int:
_configure_utf8()
parser = _parser()
args = parser.parse_args(argv)
if not 1 <= args.limit <= 10:
parser.error("--limit 取值必须在 1..10 之间")
if args.query is not None and args.mode is None:
parser.error("--query 必须同时指定 --mode")
if args.mode is not None and args.query is None:
parser.error("--mode 仅用于 --query")
if args.show_map:
print(render_concept_map(build_concept_map(args.references_dir)), end="")
return 0
if args.check:
problems = validate_references(args.references_dir)
for problem in problems:
print(f"{problem.page}:{problem.line}: {problem.message}", file=sys.stderr)
return 1 if problems else 0
if args.section is not None:
sections = load_sections(args.references_dir)
by_id = {item.id: item for item in sections}
requested = list(dict.fromkeys(args.section))
missing = [item for item in requested if item not in by_id]
if missing:
# 原子失败:只要有一个 ID 不存在就不输出任何正文,
# 避免智能体把"部分取回"误当作全部要素已取回。
for requested_id in missing:
print(f"section not found: {requested_id}", file=sys.stderr)
print("Nearest section IDs:", file=sys.stderr)
for candidate in _nearest_section_ids(requested_id, sections):
print(f"- {candidate}", file=sys.stderr)
return 2
print(
"\n".join(render_section(by_id[item]) for item in requested),
end="",
)
return 0
result = query_sections(args.query, args.mode, args.limit, args.references_dir)
print(render_candidates(result), end="")
if not result.matches:
print("no matching sections", file=sys.stderr)
return 2
if all(match.weak for match in result.matches):
print(
"only weak candidates (no intent/heading/identifier/tag hit); "
"视同无匹配,改进查询词后重试",
file=sys.stderr,
)
return 2
return 0
if __name__ == "__main__":
raise SystemExit(main())