Files
playbook/tools/tsl-codegen/scripts/convert_tsf.py
T
csh aa8a3e73a8 🐛 fix(tsl-codegen): make TSF conversion best effort
Recover recognizable function, procedure, class, and unit structure from
nonstandard TSF input while preserving strict parsing for compliant files.

Keep STANDARD.md limited to published format requirements and cover
recovery paths with regression tests.
2026-07-30 09:20:18 +08:00

2519 lines
83 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Convert TSF function, class, and unit files to declaration drafts."""
import argparse
import json
import os
import re
import sys
import tempfile
from dataclasses import dataclass
from pathlib import Path
STRUCTURE_KEYWORDS = {"function", "procedure", "type", "unit"}
PARAM_MODIFIERS = {"const", "var"}
RECOVERY_DECLARATION_KEYWORDS = {
"begin",
"class",
"const",
"constructor",
"destructor",
"end",
"function",
"finalization",
"implementation",
"initialization",
"interface",
"private",
"procedure",
"property",
"protected",
"public",
"static",
"type",
"unit",
"uses",
"var",
}
RESERVED_IDENTIFIER_KEYWORDS = RECOVERY_DECLARATION_KEYWORDS | {
"overload",
"override",
"virtual",
}
SIGNATURE_BOUNDARY_KEYWORDS = RECOVERY_DECLARATION_KEYWORDS - {"const", "var"}
DOCUMENT_LINE_RE = re.compile(r"^\s*///(.*)$")
DIRECTIVE_RE = re.compile(r"^@([a-z]+):(.*)$")
class ConversionError(Exception):
def __init__(self, path, line, message):
self.path = Path(path)
self.line = line
self.message = message
super().__init__(str(self))
def __str__(self):
location = str(self.path)
if self.line is not None:
location += f":{self.line}"
return f"{location}: {self.message}"
@dataclass(frozen=True)
class Token:
value: str
start: int
end: int
line: int
kind: str
@dataclass(frozen=True)
class Parameter:
name: str
type: str
optional: bool
line: int
@dataclass(frozen=True)
class DocumentLine:
text: str
line: int
class ChineseHelpFormatter(argparse.HelpFormatter):
def add_usage(self, usage, actions, groups, prefix=None):
super().add_usage(usage, actions, groups, prefix or "用法:")
def format_help(self):
return (
super()
.format_help()
.replace("位置参数:\n", "位置参数:\n")
.replace("选项:\n", "选项:\n")
)
def fail(path, line, message):
raise ConversionError(path, line, message)
def is_valid_identifier_token(token):
return (
token.kind == "identifier"
and token.value.casefold() not in RESERVED_IDENTIFIER_KEYWORDS
)
def is_identifier_start(char):
return char == "_" or char.isalpha()
def is_identifier_part(char):
return char == "_" or char.isalnum()
def tokenize(source):
tokens = []
index = 0
line = 1
length = len(source)
while index < length:
char = source[index]
if char.isspace():
if char == "\n":
line += 1
index += 1
continue
if source.startswith("//", index):
newline = source.find("\n", index + 2)
if newline == -1:
break
index = newline
continue
if source.startswith("(*", index):
end = source.find("*)", index + 2)
if end == -1:
end = length - 2
line += source.count("\n", index, end + 2)
index = end + 2
continue
if char == "{":
end = source.find("}", index + 1)
if end == -1:
end = length - 1
line += source.count("\n", index, end + 1)
index = end + 1
continue
if char in {"'", '"'}:
quote = char
start = index
token_line = line
index += 1
while index < length:
if source[index] == "\n":
line += 1
if source[index] == quote:
if index + 1 < length and source[index + 1] == quote:
index += 2
continue
index += 1
break
if source[index] == "\\" and index + 1 < length:
index += 2
continue
index += 1
tokens.append(
Token(source[start:index], start, index, token_line, "string")
)
continue
if is_identifier_start(char):
start = index
token_line = line
index += 1
while index < length and is_identifier_part(source[index]):
index += 1
tokens.append(
Token(source[start:index], start, index, token_line, "identifier")
)
continue
start = index
if source.startswith("...", index):
index += 3
else:
index += 1
tokens.append(Token(source[start:index], start, index, line, "symbol"))
return tokens
def first_structure_token(tokens):
for token in tokens:
if token.kind != "identifier":
continue
keyword = token.value.casefold()
if keyword not in STRUCTURE_KEYWORDS:
continue
return token
return None
def matching_parenthesis(tokens, open_index, path):
depth = 0
for index in range(open_index, len(tokens)):
value = tokens[index].value
if value == "(":
depth += 1
elif value == ")":
depth -= 1
if depth == 0:
return index
fail(path, tokens[open_index].line, "函数参数列表缺少右括号")
def declaration_end(tokens, start_index, path):
depth = 0
for index in range(start_index, len(tokens)):
value = tokens[index].value
if value in {"(", "["}:
depth += 1
elif value in {
")",
"]",
}:
depth -= 1
elif value == ";" and depth == 0:
return index
line = (
tokens[start_index].line
if start_index < len(tokens)
else tokens[-1].line
if tokens
else 1
)
fail(path, line, "function 声明缺少分号")
def split_parameter_ranges(tokens, start, end):
ranges = []
range_start = start
depth = 0
for token in tokens:
if token.start < start or token.end > end:
continue
if token.value in {"(", "["}:
depth += 1
elif token.value in {
")",
"]",
}:
depth -= 1
elif token.value in {",", ";"} and depth == 0:
ranges.append((range_start, token.start))
range_start = token.end
ranges.append((range_start, end))
return ranges
def top_level_separator(tokens, start, end, value):
depth = 0
for token in tokens:
if token.start < start or token.end > end:
continue
if token.value in {"(", "["}:
depth += 1
elif token.value in {
")",
"]",
}:
depth -= 1
elif token.value == value and depth == 0:
return token
return None
def tokens_in_range(tokens, start, end):
return [token for token in tokens if token.start >= start and token.end <= end]
def parse_parameter(source, tokens, start, end, path, fallback_line):
if not source[start:end].strip():
fail(path, fallback_line, "参数声明不能为空")
equals = top_level_separator(tokens, start, end, "=")
declaration_end_offset = equals.start if equals else end
colon = top_level_separator(tokens, start, declaration_end_offset, ":")
name_end = colon.start if colon else declaration_end_offset
name_tokens = tokens_in_range(tokens, start, name_end)
if name_tokens and name_tokens[0].value.casefold() in PARAM_MODIFIERS:
name_tokens = name_tokens[1:]
if len(name_tokens) != 1:
fail(path, fallback_line, "参数声明必须包含一个参数名")
name_token = name_tokens[0]
if not is_valid_identifier_token(name_token) and name_token.value != "...":
fail(path, name_token.line, "参数名不是有效标识符")
param_type = ""
if colon:
param_type = source[colon.end : declaration_end_offset].strip()
return Parameter(name_token.value, param_type, equals is not None, name_token.line)
def parse_function_signature(source, tokens, function_index, path):
function_token = tokens[function_index]
if function_index + 1 >= len(tokens):
fail(path, function_token.line, "function 声明缺少函数名")
name_token = tokens[function_index + 1]
if not is_valid_identifier_token(name_token):
fail(path, name_token.line, "function 声明缺少有效函数名")
end_index = declaration_end(tokens, function_index + 2, path)
if any(
token.line > function_token.line
and token.kind == "identifier"
and token.value.casefold() in SIGNATURE_BOUNDARY_KEYWORDS
for token in tokens[function_index + 2 : end_index]
):
fail(path, function_token.line, "function 声明缺少分号")
end_token = tokens[end_index]
between = tokens[function_index + 2 : end_index]
open_index = next(
(
function_index + 2 + index
for index, token in enumerate(between)
if token.value == "("
),
None,
)
parameters = []
return_search_start = name_token.end
if open_index is not None:
close_index = matching_parenthesis(tokens, open_index, path)
if close_index >= end_index:
fail(path, tokens[open_index].line, "函数参数列表没有在声明分号前结束")
open_token = tokens[open_index]
close_token = tokens[close_index]
if source[open_token.end : close_token.start].strip():
for start, end in split_parameter_ranges(
tokens, open_token.end, close_token.start
):
parameters.append(
parse_parameter(
source,
tokens,
start,
end,
path,
open_token.line,
)
)
return_search_start = close_token.end
seen_names = set()
for parameter in parameters:
folded = parameter.name.casefold()
if folded in seen_names:
fail(path, parameter.line, f"参数名重复:{parameter.name}")
seen_names.add(folded)
return_colon = top_level_separator(
tokens, return_search_start, end_token.start, ":"
)
return_type = ""
if return_colon:
return_type = source[return_colon.end : end_token.start].strip()
return name_token.value, parameters, return_type, end_index
def function_begin_before_next_structure(tokens, start_index):
for token in tokens[start_index:]:
if token.kind != "identifier":
continue
lowered = token.value.casefold()
if lowered == "begin":
return token
if lowered in STRUCTURE_KEYWORDS:
return None
return None
def parse_declaration(source, tokens, function_token, path):
function_index = tokens.index(function_token)
name, parameters, return_type, end_index = parse_function_signature(
source, tokens, function_index, path
)
end_token = tokens[end_index]
begin_token = function_begin_before_next_structure(tokens, end_index + 1)
if begin_token is None:
fail(path, end_token.line, "function 缺少 begin")
return name, parameters, return_type, begin_token
def extract_document_lines(source, begin_token):
physical_lines = source.splitlines()
begin_index = begin_token.line - 1
line_end = source.find("\n", begin_token.end)
if line_end == -1:
line_end = len(source)
if source[begin_token.end : line_end].strip():
return []
first_content = None
for index in range(begin_index + 1, len(physical_lines)):
if physical_lines[index].strip():
first_content = index
break
if first_content is None:
return []
if not DOCUMENT_LINE_RE.match(physical_lines[first_content]):
return []
document = []
for index in range(first_content, len(physical_lines)):
match = DOCUMENT_LINE_RE.match(physical_lines[index])
if not match:
break
text = match.group(1)
if text.startswith(" "):
text = text[1:]
document.append(DocumentLine(text, index + 1))
return document
def trim_blank_lines(lines):
start = 0
end = len(lines)
while start < end and not lines[start]:
start += 1
while end > start and not lines[end - 1]:
end -= 1
return lines[start:end]
def parse_directive(line, path):
match = DIRECTIVE_RE.match(line.text)
if not match:
fail(path, line.line, "未知或格式错误的文档指令")
return match.group(1), match.group(2).strip()
def collect_indented(lines, start, path, label):
collected = []
index = start
while index < len(lines) and not lines[index].text.startswith("@"):
item = lines[index]
if item.text:
if not item.text.startswith(" "):
fail(path, item.line, f"{label}必须在 /// 后额外缩进两个空格")
collected.append(item.text[2:])
else:
collected.append("")
index += 1
return trim_blank_lines(collected), index
def parse_param_directive(argument, line, path):
parts = argument.split(None, 1)
if len(parts) != 2 or not parts[1].strip():
fail(path, line, "@param: 必须包含参数名和说明")
name = parts[0]
remainder = parts[1].strip()
documented_type = ""
if remainder.startswith("{"):
close = remainder.find("}", 1)
if close == -1:
fail(path, line, "@param: 参数类型缺少右花括号")
documented_type = remainder[1:close].strip()
if not documented_type:
fail(path, line, "@param: 参数类型不能为空")
remainder = remainder[close + 1 :].strip()
if not remainder:
fail(path, line, "@param: 参数说明不能为空")
return name, documented_type, remainder
def parse_enum_item(text, path, line):
decoder = json.JSONDecoder()
stripped = text.lstrip()
try:
value, end = decoder.raw_decode(stripped)
except json.JSONDecodeError:
fail(path, line, "枚举值必须使用 json 标量写法")
if isinstance(value, (list, dict)) or value is None:
fail(path, line, "枚举值只允许数字、字符串或布尔值")
remainder = stripped[end:].lstrip()
if not remainder.startswith(":") or not remainder[1:].strip():
fail(path, line, "枚举项必须写成“值: 说明”")
return value, remainder[1:].strip()
def enum_value_key(value):
return type(value).__name__, json.dumps(value, ensure_ascii=False, sort_keys=True)
def normalized_type_tokens(value):
normalized = []
index = 0
while index < len(value):
char = value[index]
if char.isspace():
index += 1
continue
if is_identifier_start(char):
end = index + 1
while end < len(value) and is_identifier_part(value[end]):
end += 1
normalized.append(("identifier", value[index:end].casefold()))
index = end
continue
normalized.append(("symbol", char))
index += 1
return normalized
def parameter_index(parameters, name):
folded = name.casefold()
for index, parameter in enumerate(parameters):
if parameter.name.casefold() == folded:
return index
return None
def parse_document(document, parameters, declared_return, path):
description_lines = []
index = 0
while index < len(document) and not document[index].text.startswith("@"):
description_lines.append(document[index].text)
index += 1
description_lines = trim_blank_lines(description_lines)
if not any(line.strip() for line in description_lines):
fail(path, document[0].line, "函数描述不能为空")
tags = None
param_docs = {}
param_types = {}
param_values = {}
documented_param_indexes = []
documented_return = None
documented_return_line = None
examples = []
phase = "tags"
last_param = None
while index < len(document):
line = document[index]
directive, argument = parse_directive(line, path)
if directive == "tags":
if phase != "tags" or tags is not None:
fail(path, line.line, "@tags: 重复或顺序错误")
tags = argument.split()
if not tags:
fail(path, line.line, "@tags: 至少需要一个标签")
index += 1
continue
if directive == "param":
if phase == "examples":
fail(path, line.line, "示例组之后不能再写函数级指令")
if phase == "returns":
fail(path, line.line, "@param: 必须写在 @returns: 之前")
phase = "params"
name, documented_type, desc = parse_param_directive(
argument, line.line, path
)
param_idx = parameter_index(parameters, name)
if param_idx is None:
fail(path, line.line, f"@param: 引用了不存在的参数 {name}")
if documented_param_indexes and param_idx <= documented_param_indexes[-1]:
fail(path, line.line, "@param: 必须按函数声明中的参数顺序书写")
documented_param_indexes.append(param_idx)
canonical_name = parameters[param_idx].name
key = canonical_name.casefold()
declared_type = parameters[param_idx].type
if (
declared_type
and documented_type
and normalized_type_tokens(declared_type)
!= normalized_type_tokens(documented_type)
):
fail(
path,
line.line,
f"参数 {canonical_name} 的类型与函数声明不一致:"
f"注释为 {documented_type},声明为 {declared_type}",
)
param_docs[key] = desc
if documented_type:
param_types[key] = documented_type
last_param = canonical_name
index += 1
continue
if directive == "values":
if phase == "examples":
fail(path, line.line, "示例组之后不能再写函数级指令")
if phase != "params" or last_param is None:
fail(path, line.line, "@values: 必须紧跟对应的 @param:")
if not argument or len(argument.split()) != 1:
fail(path, line.line, "@values: 只能包含一个参数名")
param_idx = parameter_index(parameters, argument)
if param_idx is None:
fail(path, line.line, f"@values: 引用了不存在的参数 {argument}")
canonical_name = parameters[param_idx].name
if canonical_name.casefold() != last_param.casefold():
fail(path, line.line, "@values: 必须紧跟对应的 @param:")
key = canonical_name.casefold()
if key in param_values:
fail(path, line.line, f"参数 {canonical_name} 的 @values: 重复")
raw_items, index = collect_indented(document, index + 1, path, "枚举项")
if not any(item.strip() for item in raw_items):
fail(path, line.line, "@values: 必须包含至少一个枚举项")
values = []
seen = set()
item_line_index = line.line + 1
for raw_item in raw_items:
if not raw_item.strip():
item_line_index += 1
continue
value, desc = parse_enum_item(raw_item, path, item_line_index)
value_key = enum_value_key(value)
if value_key in seen:
fail(path, item_line_index, f"枚举值重复:{value!r}")
seen.add(value_key)
values.append({"value": value, "desc": desc})
item_line_index += 1
param_values[key] = values
last_param = None
continue
if directive == "returns":
if phase == "examples":
fail(path, line.line, "示例组之后不能再写函数级指令")
if documented_return is not None:
fail(path, line.line, "@returns: 不能重复")
if not argument:
fail(path, line.line, "@returns: 类型不能为空")
phase = "returns"
documented_return = argument
documented_return_line = line.line
last_param = None
index += 1
continue
if directive == "example":
if not argument:
fail(path, line.line, "@example: 示例说明不能为空")
phase = "examples"
last_param = None
code, next_index = collect_indented(document, index + 1, path, "示例代码")
if not any(code_line.strip() for code_line in code):
fail(path, line.line, "@example: 必须包含示例代码")
if any(code_line.lstrip().startswith("// 输出:") for code_line in code):
fail(path, line.line, "示例代码不能包含 // 输出:,请改用 @output:")
example = {"desc": argument, "code": "\n".join(code)}
index = next_index
if index < len(document):
next_directive, next_argument = parse_directive(document[index], path)
if next_directive == "output":
if next_argument:
fail(path, document[index].line, "@output: 后不能写行内内容")
output, index = collect_indented(
document, index + 1, path, "示例输出"
)
if not any(output_line.strip() for output_line in output):
fail(path, document[index - 1].line, "@output: 不能为空")
if any(
output_line.lstrip().startswith("//") for output_line in output
):
fail(
path,
document[index - 1].line,
"@output: 中不写 // 注释标记",
)
example["output"] = "\n".join(output)
examples.append(example)
continue
if directive == "output":
fail(path, line.line, "@output: 必须位于 @example: 的代码之后")
fail(path, line.line, f"未知文档指令:@{directive}:")
if declared_return and documented_return:
if normalized_type_tokens(declared_return) != normalized_type_tokens(
documented_return
):
fail(
path,
documented_return_line,
"@returns: 返回类型与函数声明不一致:"
f"注释为 {documented_return},声明为 {declared_return}",
)
return_type = declared_return or documented_return or ""
function = {"desc": "\n".join(description_lines)}
if tags:
function["tags"] = tags
if parameters:
converted_params = []
for parameter in parameters:
key = parameter.name.casefold()
converted = {
"name": parameter.name,
"type": parameter.type or param_types.get(key, ""),
}
if parameter.optional:
converted["optional"] = True
converted["desc"] = param_docs.get(key, "")
if key in param_values:
converted["values"] = param_values[key]
converted_params.append(converted)
function["params"] = converted_params
function["returns"] = return_type
if examples:
function["examples"] = examples
return function
def keyword_at(tokens, index, value):
return (
index < len(tokens)
and tokens[index].kind == "identifier"
and tokens[index].value.casefold() == value
)
def document_before(source, declaration_line, lower_bound_line=1):
"""Return the /// block immediately before a declaration, allowing blanks."""
lines = source.splitlines()
index = declaration_line - 2
while index >= lower_bound_line - 1 and not lines[index].strip():
index -= 1
if index < lower_bound_line - 1 or not DOCUMENT_LINE_RE.match(lines[index]):
return []
end = index
while index >= lower_bound_line - 1 and DOCUMENT_LINE_RE.match(lines[index]):
index -= 1
document = []
for line_index in range(index + 1, end + 1):
match = DOCUMENT_LINE_RE.match(lines[line_index])
text = match.group(1)
if text.startswith(" "):
text = text[1:]
document.append(DocumentLine(text, line_index + 1))
return document
def trailing_line_comment(source, declaration_end_token):
line_end = source.find("\n", declaration_end_token.end)
if line_end == -1:
line_end = len(source)
remainder = source[declaration_end_token.end : line_end].lstrip()
if not remainder.startswith("//") or remainder.startswith("///"):
return ""
return remainder[2:].strip()
def parse_simple_document(document, path, label):
if not document:
return {"desc": ""}
description = []
index = 0
while index < len(document) and not document[index].text.startswith("@"):
description.append(document[index].text)
index += 1
description = trim_blank_lines(description)
result = {"desc": "\n".join(description)}
if index < len(document):
directive, argument = parse_directive(document[index], path)
if directive != "tags" or not argument:
fail(path, document[index].line, f"{label}只支持 @tags: 指令")
result["tags"] = argument.split()
index += 1
if index != len(document):
fail(path, document[index].line, f"{label}文档包含不支持的指令")
return result
def converted_parameters(parameters, descriptions=None, values=None):
descriptions = descriptions or {}
values = values or {}
result = []
for parameter in parameters:
key = parameter.name.casefold()
item = {
"name": parameter.name,
"type": parameter.type,
}
if parameter.optional:
item["optional"] = True
item["desc"] = descriptions.get(key, "")
if key in values:
item["values"] = values[key]
result.append(item)
return result
def best_effort_function_details(document, parameters, declared_return, path):
if document:
try:
return parse_document(document, parameters, declared_return, path)
except ConversionError:
pass
return {
"desc": "",
"params": converted_parameters(parameters),
"returns": declared_return,
}
def skip_begin_block(tokens, begin_index, path):
depth = 0
for index in range(begin_index, len(tokens)):
if keyword_at(tokens, index, "begin"):
depth += 1
elif keyword_at(tokens, index, "end"):
depth -= 1
if depth == 0:
if index + 1 < len(tokens) and tokens[index + 1].value == ";":
return index + 2
return index + 1
fail(path, tokens[begin_index].line, "内联方法缺少 end")
def parse_method_member(source, tokens, function_index, path, visibility, binding):
name, parameters, declared_return, end_index = parse_function_signature(
source, tokens, function_index, path
)
document = (
document_before(source, tokens[function_index].line)
if visibility != "private"
else []
)
if document:
details = parse_document(document, parameters, declared_return, path)
else:
details = {"desc": ""}
if parameters:
details["params"] = converted_parameters(parameters)
if declared_return:
details["returns"] = declared_return
if not details.get("returns"):
details.pop("returns", None)
modifiers = []
next_index = end_index + 1
declaration_end_token = tokens[end_index]
while next_index + 1 < len(tokens):
modifier = tokens[next_index].value.casefold()
if modifier not in {"overload", "virtual", "override"}:
break
if tokens[next_index + 1].value != ";":
break
if modifier in modifiers:
fail(path, tokens[next_index].line, f"方法修饰符重复:{modifier}")
modifiers.append(modifier)
declaration_end_token = tokens[next_index + 1]
next_index += 2
if visibility != "private" and not details["desc"]:
details["desc"] = trailing_line_comment(source, declaration_end_token)
if keyword_at(tokens, next_index, "begin"):
next_index = skip_begin_block(tokens, next_index, path)
names = ", ".join(parameter.name for parameter in parameters)
member = {
"kind": "method",
"name": name,
"visibility": visibility,
"binding": binding,
"signature": f"{name}({names})",
**details,
}
if modifiers:
member["modifiers"] = modifiers
return member, next_index
def parse_property_member(source, tokens, property_index, path, visibility):
if property_index + 1 >= len(tokens):
fail(path, tokens[property_index].line, "property 声明缺少名称")
name_token = tokens[property_index + 1]
if not is_valid_identifier_token(name_token):
fail(path, name_token.line, "property 声明缺少有效名称")
end_index = declaration_end(tokens, property_index + 2, path)
end_token = tokens[end_index]
cursor = property_index + 2
parameters = []
type_start = name_token.end
if cursor < end_index and tokens[cursor].value == "(":
close_index = matching_parenthesis(tokens, cursor, path)
if close_index >= end_index:
fail(path, tokens[cursor].line, "property 参数列表没有在分号前结束")
if source[tokens[cursor].end : tokens[close_index].start].strip():
for start, end in split_parameter_ranges(
tokens, tokens[cursor].end, tokens[close_index].start
):
parameters.append(
parse_parameter(
source, tokens, start, end, path, tokens[cursor].line
)
)
cursor = close_index + 1
type_start = tokens[close_index].end
read_indexes = [
index for index in range(cursor, end_index) if keyword_at(tokens, index, "read")
]
write_indexes = [
index
for index in range(cursor, end_index)
if keyword_at(tokens, index, "write")
]
if not read_indexes and not write_indexes:
fail(path, name_token.line, "property 必须包含 read 或 write")
accessor_index = min(read_indexes + write_indexes)
colon_index = next(
(
index
for index in range(cursor, accessor_index)
if tokens[index].value == ":"
),
None,
)
property_type = ""
if colon_index is not None:
property_type = source[
tokens[colon_index].end : tokens[accessor_index].start
].strip()
document = (
document_before(source, tokens[property_index].line)
if visibility != "private"
else []
)
if document:
details = parse_document(document, parameters, "", path)
if details.get("returns"):
fail(path, document[0].line, "property 文档不支持 @returns:")
if details.get("examples"):
fail(path, document[0].line, "property 文档不支持示例")
details.pop("returns", None)
else:
details = {"desc": ""}
if parameters:
details["params"] = converted_parameters(parameters)
access = "readwrite" if read_indexes and write_indexes else (
"read" if read_indexes else "write"
)
member = {
"kind": "property",
"name": name_token.value,
"visibility": visibility,
**details,
"access": access,
}
if property_type:
member["type"] = property_type
return member, end_index + 1
def skip_attributes(tokens, index, end_index, path):
while index < end_index and tokens[index].value == "[":
depth = 1
index += 1
while index < end_index and depth:
if tokens[index].value == "[":
depth += 1
elif tokens[index].value == "]":
depth -= 1
index += 1
if depth:
fail(path, tokens[index - 1].line, "成员属性缺少右方括号")
return index
def parse_field_member(source, tokens, start_index, path, visibility, is_static):
end_index = declaration_end(tokens, start_index, path)
cursor = start_index + (1 if is_static else 0)
cursor = skip_attributes(tokens, cursor, end_index, path)
if cursor >= end_index or not is_valid_identifier_token(tokens[cursor]):
fail(path, tokens[start_index].line, "字段声明缺少有效名称")
name_token = tokens[cursor]
colon_index = next(
(
index
for index in range(cursor + 1, end_index)
if tokens[index].value == ":"
),
None,
)
comma_limit = colon_index if colon_index is not None else end_index
if visibility != "private" and any(
tokens[index].value == "," for index in range(cursor + 1, comma_limit)
):
fail(path, name_token.line, "对外字段必须一项一条声明")
equals_index = next(
(
index
for index in range(cursor + 1, end_index)
if tokens[index].value == "="
),
None,
)
field_type = ""
if colon_index is not None:
type_end = equals_index if equals_index is not None else end_index
field_type = source[
tokens[colon_index].end : tokens[type_end].start
].strip()
details = (
parse_simple_document(
document_before(source, tokens[start_index].line), path, "字段"
)
if visibility != "private"
else {"desc": ""}
)
member = {
"kind": "field",
"name": name_token.value,
"visibility": visibility,
**details,
"type": field_type,
}
if is_static:
member["static"] = True
return member, end_index + 1
def parse_constant_member(source, tokens, start_index, path, visibility, is_static):
const_index = start_index + (1 if is_static else 0)
end_index = declaration_end(tokens, const_index + 1, path)
if (
const_index + 1 >= end_index
or not is_valid_identifier_token(tokens[const_index + 1])
):
fail(path, tokens[const_index].line, "const 声明缺少有效名称")
name_token = tokens[const_index + 1]
equals_index = next(
(
index
for index in range(const_index + 2, end_index)
if tokens[index].value == "="
),
None,
)
if equals_index is None:
fail(path, name_token.line, "const 声明缺少值")
colon_index = next(
(
index
for index in range(const_index + 2, equals_index)
if tokens[index].value == ":"
),
None,
)
name_end_index = colon_index if colon_index is not None else equals_index
if visibility != "private" and any(
tokens[index].value == ","
for index in range(const_index + 2, name_end_index)
):
fail(path, name_token.line, "对外常量必须一项一条声明")
constant_type = ""
if colon_index is not None:
constant_type = source[
tokens[colon_index].end : tokens[equals_index].start
].strip()
value = source[tokens[equals_index].end : tokens[end_index].start].strip()
details = (
parse_simple_document(
document_before(source, name_token.line), path, "常量"
)
if visibility != "private"
else {"desc": ""}
)
member = {
"kind": "constant",
"name": name_token.value,
"visibility": visibility,
**details,
"value": value,
}
if constant_type:
member["type"] = constant_type
if is_static:
member["static"] = True
return member, end_index + 1
def parse_bases(source, tokens, class_index, path):
if class_index + 1 >= len(tokens) or tokens[class_index + 1].value != "(":
return [], class_index + 1, tokens[class_index]
open_index = class_index + 1
close_index = matching_parenthesis(tokens, open_index, path)
raw_start = tokens[open_index].end
raw_end = tokens[close_index].start
bases = []
for start, end in split_parameter_ranges(tokens, raw_start, raw_end):
base = source[start:end].strip()
if base:
bases.append(base)
return bases, close_index + 1, tokens[close_index]
def first_unbound_document_line(source, start_line, end_line, allowed_lines):
for line_number, text in enumerate(
source.splitlines()[start_line - 1 : end_line], start=start_line
):
if DOCUMENT_LINE_RE.match(text) and line_number not in allowed_lines:
return line_number
return None
def parse_class(source, tokens, type_index, path, *, require_filename_match=True):
if type_index + 3 >= len(tokens):
fail(path, tokens[type_index].line, "class 声明不完整")
name_token = tokens[type_index + 1]
if (
not is_valid_identifier_token(name_token)
or tokens[type_index + 2].value != "="
):
fail(path, tokens[type_index].line, "type class 声明格式错误")
class_index = type_index + 3
if not keyword_at(tokens, class_index, "class"):
fail(path, tokens[type_index].line, "目前只支持 class type")
if (
require_filename_match
and name_token.value.casefold() != Path(path).stem.casefold()
):
fail(path, name_token.line, "对外 class 名称必须与文件名一致")
bases, index, header_end = parse_bases(source, tokens, class_index, path)
class_document = (
document_before(source, tokens[index].line, header_end.line + 1)
if index < len(tokens)
else []
)
class_details = parse_simple_document(class_document, path, "class")
used_class_doc_lines = {line.line for line in class_document}
allowed_document_lines = set(used_class_doc_lines)
visibility = "public"
members = []
while index < len(tokens):
token = tokens[index]
lowered = token.value.casefold() if token.kind == "identifier" else ""
if lowered == "end":
if index + 1 >= len(tokens) or tokens[index + 1].value != ";":
fail(path, token.line, "class 结尾必须是 end;")
unbound_line = first_unbound_document_line(
source,
header_end.line + 1,
token.line,
allowed_document_lines,
)
if unbound_line is not None:
fail(path, unbound_line, "class 文档块无法绑定")
result = {
"name": name_token.value,
**class_details,
"members": members,
}
if bases:
result["bases"] = bases
return result, index + 2
if lowered in {"public", "protected", "private"}:
visibility = lowered
index += 1
continue
if lowered == "uses":
index = declaration_end(tokens, index + 1, path) + 1
continue
declaration_line = token.line
document = document_before(source, declaration_line)
if document and any(line.line in used_class_doc_lines for line in document):
document = []
allowed_document_lines.update(line.line for line in document)
if lowered == "class" and keyword_at(tokens, index + 1, "function"):
member, index = parse_method_member(
source, tokens, index + 1, path, visibility, "class"
)
elif lowered == "function":
member, index = parse_method_member(
source, tokens, index, path, visibility, "instance"
)
elif lowered == "procedure":
fail(path, token.line, "class 暂不支持 procedure")
elif lowered == "property":
member, index = parse_property_member(
source, tokens, index, path, visibility
)
elif lowered == "static":
if keyword_at(tokens, index + 1, "function"):
fail(path, token.line, "不存在 static function,请使用 class function")
if keyword_at(tokens, index + 1, "const"):
member, index = parse_constant_member(
source, tokens, index, path, visibility, True
)
elif any(
keyword_at(tokens, index + 1, keyword)
for keyword in {"class", "procedure", "property", "type", "var"}
):
fail(path, token.line, f"class 不支持成员声明:static {tokens[index + 1].value}")
else:
member, index = parse_field_member(
source, tokens, index, path, visibility, True
)
elif lowered == "const":
member, index = parse_constant_member(
source, tokens, index, path, visibility, False
)
elif lowered in {
"class",
"constructor",
"destructor",
"finalization",
"implementation",
"initialization",
"interface",
"type",
"unit",
"var",
}:
fail(path, token.line, f"class 不支持成员声明:{token.value}")
else:
member, index = parse_field_member(
source, tokens, index, path, visibility, False
)
if visibility != "private":
if document and not member.get("desc"):
member_doc_lines = {line.line for line in document}
if member_doc_lines != used_class_doc_lines:
fail(path, document[0].line, "成员文档块无法绑定")
members.append(member)
fail(path, name_token.line, "class 缺少 end;")
def parse_interface_function(source, tokens, function_index, path):
name, parameters, declared_return, end_index = parse_function_signature(
source, tokens, function_index, path
)
document = document_before(source, tokens[function_index].line)
if document:
details = parse_document(document, parameters, declared_return, path)
else:
details = {"desc": ""}
if parameters:
details["params"] = converted_parameters(parameters)
details["returns"] = declared_return
names = ", ".join(parameter.name for parameter in parameters)
return {
"kind": "function",
"name": name,
"signature": f"{name}({names})",
**details,
}, end_index + 1
def parse_unit_constant(source, tokens, const_index, path):
member, next_index = parse_constant_member(
source, tokens, const_index, path, "public", False
)
member.pop("visibility", None)
return member, next_index
def parse_unit_variable(source, tokens, var_index, path):
if var_index + 1 >= len(tokens):
fail(path, tokens[var_index].line, "var 声明缺少变量")
member, next_index = parse_field_member(
source, tokens, var_index + 1, path, "public", False
)
member["kind"] = "variable"
member.pop("visibility", None)
return member, next_index
def parse_bare_unit_variable(source, tokens, start_index, path):
member, next_index = parse_field_member(
source, tokens, start_index, path, "public", False
)
member["kind"] = "variable"
member.pop("visibility", None)
return member, next_index
def parse_bare_unit_constant(source, tokens, start_index, path):
end_index = declaration_end(tokens, start_index + 1, path)
name_token = tokens[start_index]
if not is_valid_identifier_token(name_token):
fail(path, name_token.line, "const 声明缺少有效名称")
equals_index = next(
(
index
for index in range(start_index + 1, end_index)
if tokens[index].value == "="
),
None,
)
if equals_index is None:
fail(path, name_token.line, "const 声明缺少值")
colon_index = next(
(
index
for index in range(start_index + 1, equals_index)
if tokens[index].value == ":"
),
None,
)
name_end_index = colon_index if colon_index is not None else equals_index
if any(
tokens[index].value == ","
for index in range(start_index + 1, name_end_index)
):
fail(path, name_token.line, "对外常量必须一项一条声明")
member = {
"kind": "constant",
"name": name_token.value,
**parse_simple_document(
document_before(source, name_token.line), path, "常量"
),
"value": source[tokens[equals_index].end : tokens[end_index].start].strip(),
}
if colon_index is not None:
member["type"] = source[
tokens[colon_index].end : tokens[equals_index].start
].strip()
return member, end_index + 1
def parse_unit(source, tokens, unit_index, path):
if unit_index + 2 >= len(tokens):
fail(path, tokens[unit_index].line, "unit 声明不完整")
name_token = tokens[unit_index + 1]
if (
not is_valid_identifier_token(name_token)
or tokens[unit_index + 2].value != ";"
):
fail(path, tokens[unit_index].line, "unit 声明格式错误")
if name_token.value.casefold() != Path(path).stem.casefold():
fail(path, name_token.line, "unit 名称必须与文件名一致")
interface_index = next(
(
index
for index in range(unit_index + 3, len(tokens))
if keyword_at(tokens, index, "interface")
),
None,
)
if interface_index is None:
fail(path, name_token.line, "unit API 文档要求显式 interface")
implementation_index = next(
(
index
for index in range(interface_index + 1, len(tokens))
if keyword_at(tokens, index, "implementation")
),
None,
)
if implementation_index is None:
fail(path, tokens[interface_index].line, "完整 unit 缺少 implementation")
terminal_end_indexes = [
index
for index in range(implementation_index + 1, len(tokens))
if keyword_at(tokens, index, "end")
and index + 1 < len(tokens)
and tokens[index + 1].value == "."
]
if not terminal_end_indexes:
fail(path, tokens[implementation_index].line, "完整 unit 缺少 end.")
terminal_end_index = terminal_end_indexes[-1]
if terminal_end_index + 2 != len(tokens):
fail(
path,
tokens[terminal_end_index + 2].line,
"unit 的 end. 必须结束整个文件",
)
unit_document = document_before(
source, tokens[interface_index].line, tokens[unit_index + 2].line + 1
)
details = parse_simple_document(unit_document, path, "unit")
members = []
allowed_document_lines = {line.line for line in unit_document}
index = interface_index + 1
declaration_section = None
while index < implementation_index:
token = tokens[index]
lowered = token.value.casefold() if token.kind == "identifier" else ""
if lowered == "uses":
declaration_section = None
index = declaration_end(tokens, index + 1, path) + 1
continue
if lowered == "function":
declaration_section = None
allowed_document_lines.update(
line.line for line in document_before(source, token.line)
)
member, index = parse_interface_function(source, tokens, index, path)
members.append(member)
continue
if lowered == "procedure":
fail(path, token.line, "unit interface 暂不支持 procedure")
if lowered == "type":
declaration_section = None
if index + 3 >= implementation_index or not keyword_at(
tokens, index + 3, "class"
):
fail(path, token.line, "unit interface 只支持 class type")
class_start_line = token.line
class_data, index = parse_class(
source,
tokens,
index,
path,
require_filename_match=False,
)
class_end_line = tokens[index - 1].line
allowed_document_lines.update(
line_number
for line_number, text in enumerate(
source.splitlines()[class_start_line - 1 : class_end_line],
start=class_start_line,
)
if DOCUMENT_LINE_RE.match(text)
)
members.append({"kind": "class", **class_data})
continue
if lowered == "const":
declaration_section = "const"
if index + 1 < implementation_index:
allowed_document_lines.update(
line.line
for line in document_before(source, tokens[index + 1].line)
)
member, index = parse_unit_constant(source, tokens, index, path)
members.append(member)
continue
if lowered == "var":
declaration_section = "var"
if index + 1 < implementation_index:
allowed_document_lines.update(
line.line
for line in document_before(source, tokens[index + 1].line)
)
member, index = parse_unit_variable(source, tokens, index, path)
members.append(member)
continue
if token.kind == "identifier" and declaration_section == "var":
allowed_document_lines.update(
line.line for line in document_before(source, token.line)
)
member, index = parse_bare_unit_variable(source, tokens, index, path)
members.append(member)
continue
if token.kind == "identifier" and declaration_section == "const":
allowed_document_lines.update(
line.line for line in document_before(source, token.line)
)
member, index = parse_bare_unit_constant(source, tokens, index, path)
members.append(member)
continue
fail(path, token.line, f"unit interface 不支持声明:{token.value}")
unbound_line = first_unbound_document_line(
source,
tokens[unit_index + 2].line + 1,
tokens[implementation_index].line - 1,
allowed_document_lines,
)
if unbound_line is not None:
fail(path, unbound_line, "unit interface 文档块无法绑定")
return {
"name": name_token.value,
**details,
"members": members,
}
def mask_document_lines(source):
masked = []
for line in source.splitlines(keepends=True):
content = line.rstrip("\r\n")
ending = line[len(content) :]
if DOCUMENT_LINE_RE.match(content):
masked.append(" " * len(content) + ending)
else:
masked.append(line)
return "".join(masked)
def next_semicolon(tokens, start_index, limit):
depth = 0
for index in range(start_index, limit):
value = tokens[index].value
if value in {"(", "["}:
depth += 1
elif value in {
")",
"]",
}:
depth = max(0, depth - 1)
elif value == ";" and depth == 0:
return index
return None
def next_declaration_boundary(tokens, start_index, limit):
if start_index >= min(limit, len(tokens)):
return None
start_line = tokens[start_index].line
for index in range(start_index + 1, limit):
token = tokens[index]
if (
token.line > start_line
and token.kind == "identifier"
and token.value.casefold() in RECOVERY_DECLARATION_KEYWORDS
):
return index
return None
def declaration_scan_end(tokens, start_index, limit):
semicolon_index = next_semicolon(tokens, start_index, limit)
boundary_index = next_declaration_boundary(tokens, start_index, limit)
candidates = [
index
for index in (semicolon_index, boundary_index)
if index is not None
]
return min(candidates) if candidates else limit
def token_start_or_eof(tokens, index, source):
return tokens[index].start if index < len(tokens) else len(source)
def recover_after_declaration(tokens, start_index, limit):
boundary_index = next_declaration_boundary(tokens, start_index, limit)
semicolon_index = next_semicolon(tokens, start_index, limit)
if boundary_index is not None and (
semicolon_index is None or boundary_index < semicolon_index
):
if keyword_at(tokens, boundary_index, "begin"):
try:
return min(
skip_begin_block(tokens, boundary_index, Path("<draft>")),
limit,
)
except ConversionError:
return limit
return boundary_index
if semicolon_index is None:
return limit
index = semicolon_index + 1
while (
index + 1 < limit
and tokens[index].value.casefold() in {"overload", "virtual", "override"}
and tokens[index + 1].value == ";"
):
index += 2
if keyword_at(tokens, index, "begin"):
try:
return min(skip_begin_block(tokens, index, Path("<draft>")), limit)
except ConversionError:
return limit
return index
def matching_parenthesis_in_range(tokens, open_index, limit):
depth = 0
for index in range(open_index, limit):
value = tokens[index].value
if value == "(":
depth += 1
elif value == ")":
depth -= 1
if depth == 0:
return index
return None
def draft_parameter(source, tokens, start, end, fallback_line):
if not source[start:end].strip():
return None
try:
return parse_parameter(
source,
tokens,
start,
end,
Path("<draft>"),
fallback_line,
)
except ConversionError:
pass
equals = top_level_separator(tokens, start, end, "=")
declaration_end_offset = equals.start if equals else end
colon = top_level_separator(tokens, start, declaration_end_offset, ":")
name_end = colon.start if colon else declaration_end_offset
name_tokens = tokens_in_range(tokens, start, name_end)
if name_tokens and name_tokens[0].value.casefold() in PARAM_MODIFIERS:
name_tokens = name_tokens[1:]
name_token = next(
(
token
for token in name_tokens
if is_valid_identifier_token(token) or token.value == "..."
),
None,
)
if name_token is None:
return None
param_type = ""
if colon:
param_type = source[colon.end : declaration_end_offset].strip()
return Parameter(
name_token.value,
param_type,
equals is not None,
name_token.line,
)
def recover_signature_parts(
source,
tokens,
name_index,
scan_start,
fallback_name,
limit,
):
name_token = tokens[name_index] if name_index is not None else None
name = name_token.value if name_token is not None else fallback_name
open_index = (
scan_start
if scan_start < limit and tokens[scan_start].value == "("
else None
)
close_index = (
matching_parenthesis_in_range(tokens, open_index, limit)
if open_index is not None
else None
)
parameters = []
if open_index is not None:
if close_index is not None:
parameter_end = tokens[close_index].start
else:
boundary = next_declaration_boundary(tokens, open_index, limit)
begin_index = next(
(
index
for index in range(open_index + 1, limit)
if keyword_at(tokens, index, "begin")
),
None,
)
candidates = [
index
for index in (boundary, begin_index)
if index is not None
]
parameter_end_index = min(candidates) if candidates else limit
parameter_end = token_start_or_eof(
tokens, parameter_end_index, source
)
if source[tokens[open_index].end : parameter_end].strip():
for start, end in split_parameter_ranges(
tokens,
tokens[open_index].end,
parameter_end,
):
parameter = draft_parameter(
source,
tokens,
start,
end,
tokens[open_index].line,
)
if parameter is not None:
parameters.append(parameter)
return_type = ""
if open_index is None or close_index is not None:
return_start_index = (
close_index + 1 if close_index is not None else scan_start
)
return_end_index = limit
for index in range(return_start_index, limit):
if tokens[index].value == ";" or keyword_at(tokens, index, "begin"):
return_end_index = index
break
colon_index = next(
(
index
for index in range(return_start_index, return_end_index)
if tokens[index].value == ":"
),
None,
)
if colon_index is not None:
return_type = source[
tokens[colon_index].end : token_start_or_eof(
tokens, return_end_index, source
)
].strip()
return name, parameters, return_type
def draft_function_signature(source, tokens, function_index, path, limit=None):
limit = min(limit if limit is not None else len(tokens), len(tokens))
scan_end = declaration_scan_end(tokens, function_index, limit)
if scan_end < limit and tokens[scan_end].value == ";":
try:
name, parameters, return_type, end_index = parse_function_signature(
source,
tokens,
function_index,
Path("<draft>"),
)
if end_index == scan_end:
return name, parameters, return_type
except ConversionError:
pass
candidate_index = function_index + 1
name_index = None
if candidate_index < limit:
candidate = tokens[candidate_index]
if is_valid_identifier_token(candidate):
name_index = candidate_index
scan_start = (name_index + 1) if name_index is not None else candidate_index
return recover_signature_parts(
source,
tokens,
name_index,
scan_start,
Path(path).stem,
limit,
)
def draft_unknown_function(source, tokens, path):
name_index = next(
(
index
for index in range(len(tokens) - 1)
if is_valid_identifier_token(tokens[index])
and tokens[index + 1].value == "("
),
None,
)
if name_index is None:
name = Path(path).stem
parameters = []
return_type = ""
else:
name, parameters, return_type = recover_signature_parts(
source,
tokens,
name_index,
name_index + 1,
Path(path).stem,
len(tokens),
)
names = ", ".join(parameter.name for parameter in parameters)
return {
"kind": "function",
"name": name,
"signature": f"{name}({names})",
"desc": "",
"params": converted_parameters(parameters),
"returns": return_type,
}
def draft_method_member(source, tokens, function_index, visibility, binding, limit):
name_index = function_index + 1
if (
name_index >= limit
or not is_valid_identifier_token(tokens[name_index])
):
return None, recover_after_declaration(tokens, function_index, limit)
try:
member, next_index = parse_method_member(
source,
tokens,
function_index,
Path("<draft>"),
visibility,
binding,
)
return member, min(next_index, limit)
except ConversionError:
pass
name, parameters, declared_return = draft_function_signature(
source,
tokens,
function_index,
Path("<draft>"),
limit,
)
next_index = recover_after_declaration(tokens, function_index, limit)
names = ", ".join(parameter.name for parameter in parameters)
return {
"kind": "method",
"name": name,
"visibility": visibility,
"binding": binding,
"signature": f"{name}({names})",
"desc": "",
"params": converted_parameters(parameters),
"returns": declared_return,
}, next_index
def draft_property_member(source, tokens, property_index, visibility, limit):
try:
member, next_index = parse_property_member(
source, tokens, property_index, Path("<draft>"), visibility
)
return member, min(next_index, limit)
except ConversionError:
pass
name_token = tokens[property_index + 1] if property_index + 1 < limit else None
if name_token is None or not is_valid_identifier_token(name_token):
return None, recover_after_declaration(tokens, property_index, limit)
declaration_end_index = declaration_scan_end(tokens, property_index, limit)
has_read = any(
keyword_at(tokens, index, "read")
for index in range(property_index + 2, declaration_end_index)
)
has_write = any(
keyword_at(tokens, index, "write")
for index in range(property_index + 2, declaration_end_index)
)
access = "readwrite" if has_read and has_write else (
"read" if has_read else "write" if has_write else ""
)
return {
"kind": "property",
"name": name_token.value,
"visibility": visibility,
"desc": "",
"type": "",
"params": [],
"access": access,
}, recover_after_declaration(tokens, property_index, limit)
def draft_field_members(source, tokens, start_index, visibility, is_static, limit):
try:
member, next_index = parse_field_member(
source,
tokens,
start_index,
Path("<draft>"),
visibility,
is_static,
)
return [member], min(next_index, limit)
except ConversionError:
pass
cursor = start_index + (1 if is_static else 0)
if (
cursor < limit
and tokens[cursor].kind == "identifier"
and not is_valid_identifier_token(tokens[cursor])
):
return [], cursor
declaration_end_index = declaration_scan_end(tokens, start_index, limit)
colon_index = next(
(
index
for index in range(cursor, declaration_end_index)
if tokens[index].value == ":"
),
None,
)
equals_index = next(
(
index
for index in range(cursor, declaration_end_index)
if tokens[index].value == "="
),
None,
)
names_end = colon_index or equals_index or declaration_end_index
names = [
token.value
for token in tokens[cursor:names_end]
if is_valid_identifier_token(token)
]
field_type = ""
if colon_index is not None:
type_end = equals_index or declaration_end_index
field_type = source[
tokens[colon_index].end : token_start_or_eof(tokens, type_end, source)
].strip()
members = []
for name in names:
member = {
"kind": "field",
"name": name,
"visibility": visibility,
"desc": "",
"type": field_type,
}
if is_static:
member["static"] = True
members.append(member)
return members, recover_after_declaration(tokens, start_index, limit)
def draft_constant_members(
source,
tokens,
start_index,
visibility,
is_static,
limit,
*,
bare=False,
):
try:
if bare:
member, next_index = parse_bare_unit_constant(
source, tokens, start_index, Path("<draft>")
)
member["visibility"] = visibility
else:
member, next_index = parse_constant_member(
source,
tokens,
start_index,
Path("<draft>"),
visibility,
is_static,
)
return [member], min(next_index, limit)
except ConversionError:
pass
const_index = start_index + (1 if is_static else 0)
name_start_index = const_index if bare else const_index + 1
if name_start_index >= min(limit, len(tokens)):
return [], limit
if not is_valid_identifier_token(tokens[name_start_index]):
return [], name_start_index
declaration_end_index = declaration_scan_end(
tokens, name_start_index, limit
)
equals_index = next(
(
index
for index in range(name_start_index, declaration_end_index)
if tokens[index].value == "="
),
None,
)
names_end = equals_index or declaration_end_index
colon_index = next(
(
index
for index in range(name_start_index, names_end)
if tokens[index].value == ":"
),
None,
)
names_end = colon_index or names_end
names = [
token.value
for token in tokens[name_start_index:names_end]
if is_valid_identifier_token(token)
]
value = ""
if equals_index is not None:
value = source[
tokens[equals_index].end : token_start_or_eof(
tokens, declaration_end_index, source
)
].strip()
constant_type = ""
if colon_index is not None and equals_index is not None:
constant_type = source[
tokens[colon_index].end : tokens[equals_index].start
].strip()
members = []
for name in names:
member = {
"kind": "constant",
"name": name,
"visibility": visibility,
"desc": "",
"type": constant_type,
"value": value,
}
if is_static:
member["static"] = True
members.append(member)
return members, recover_after_declaration(tokens, name_start_index, limit)
def draft_class(source, tokens, type_index, path, limit=None):
limit = min(limit if limit is not None else len(tokens), len(tokens))
name_token = tokens[type_index + 1] if type_index + 1 < limit else None
has_declared_name = (
name_token is not None and is_valid_identifier_token(name_token)
)
name = name_token.value if has_declared_name else Path(path).stem
class_search_start = type_index + (2 if has_declared_name else 1)
class_index = next(
(
index
for index in range(class_search_start, min(type_index + 10, limit))
if keyword_at(tokens, index, "class")
),
None,
)
if class_index is None:
return {"name": name, "desc": "", "members": []}, limit
index = class_index + 1
bases = []
if index < len(tokens) and tokens[index].value == "(":
try:
bases, index, _ = parse_bases(source, tokens, class_index, Path("<draft>"))
if index > limit:
bases = []
index = (
next_declaration_boundary(tokens, class_index, limit)
or class_index + 1
)
except ConversionError:
index = (
next_declaration_boundary(tokens, class_index, limit)
or class_index + 1
)
masked_source = mask_document_lines(source)
visibility = "public"
members = []
while index < limit:
token = tokens[index]
lowered = token.value.casefold() if token.kind == "identifier" else ""
if lowered == "end":
has_semicolon = (
index + 1 < limit and tokens[index + 1].value == ";"
)
next_index = index + 2 if has_semicolon else index + 1
result = {"name": name, "desc": "", "members": members}
if bases:
result["bases"] = bases
return result, next_index
if lowered in {"public", "protected", "private"}:
visibility = lowered
index += 1
continue
if lowered == "uses":
index = recover_after_declaration(tokens, index, limit)
continue
member = None
parsed_members = None
if lowered == "class" and keyword_at(tokens, index + 1, "function"):
member, index = draft_method_member(
masked_source, tokens, index + 1, visibility, "class", limit
)
elif lowered in {"function", "procedure"}:
member, index = draft_method_member(
masked_source, tokens, index, visibility, "instance", limit
)
elif lowered == "property":
member, index = draft_property_member(
masked_source, tokens, index, visibility, limit
)
elif lowered == "static":
if keyword_at(tokens, index + 1, "function") or keyword_at(
tokens, index + 1, "procedure"
):
member, index = draft_method_member(
masked_source, tokens, index + 1, visibility, "class", limit
)
elif keyword_at(tokens, index + 1, "const"):
parsed_members, index = draft_constant_members(
masked_source, tokens, index, visibility, True, limit
)
elif keyword_at(tokens, index + 1, "property"):
member, index = draft_property_member(
masked_source, tokens, index + 1, visibility, limit
)
else:
parsed_members, index = draft_field_members(
masked_source, tokens, index, visibility, True, limit
)
elif lowered == "const":
parsed_members, index = draft_constant_members(
masked_source, tokens, index, visibility, False, limit
)
elif lowered in {
"constructor",
"destructor",
}:
member, index = draft_method_member(
masked_source, tokens, index, visibility, "instance", limit
)
elif lowered in {
"class",
"finalization",
"implementation",
"initialization",
"interface",
"type",
"unit",
"var",
}:
index = recover_after_declaration(tokens, index, limit)
elif token.kind == "identifier":
parsed_members, index = draft_field_members(
masked_source, tokens, index, visibility, False, limit
)
else:
index += 1
if visibility != "private":
if member is not None:
members.append(member)
if parsed_members:
members.extend(parsed_members)
result = {"name": name, "desc": "", "members": members}
if bases:
result["bases"] = bases
return result, limit
def as_unit_function(member):
if member is None:
return None
converted = dict(member)
converted["kind"] = "function"
converted.pop("visibility", None)
converted.pop("binding", None)
converted.pop("modifiers", None)
return converted
def draft_unit(source, tokens, unit_index, path):
name_token = tokens[unit_index + 1] if unit_index + 1 < len(tokens) else None
has_declared_name = (
name_token is not None and is_valid_identifier_token(name_token)
)
name = name_token.value if has_declared_name else Path(path).stem
interface_index = next(
(
index
for index in range(unit_index + 1, len(tokens))
if keyword_at(tokens, index, "interface")
),
None,
)
implementation_index = next(
(
index
for index in range(
interface_index + 1
if interface_index is not None
else unit_index + 1,
len(tokens),
)
if keyword_at(tokens, index, "implementation")
),
None,
)
terminal_end_index = next(
(
index
for index in range(unit_index + 1, len(tokens))
if keyword_at(tokens, index, "end")
and index + 1 < len(tokens)
and tokens[index + 1].value == "."
),
None,
)
header_end_index = unit_index + (2 if has_declared_name else 1)
if header_end_index < len(tokens) and tokens[header_end_index].value == ";":
header_end_index += 1
index = interface_index + 1 if interface_index is not None else header_end_index
limit = implementation_index or terminal_end_index or len(tokens)
masked_source = mask_document_lines(source)
members = []
declaration_section = None
while index < limit:
token = tokens[index]
lowered = token.value.casefold() if token.kind == "identifier" else ""
if lowered == "uses":
declaration_section = None
index = recover_after_declaration(tokens, index, limit)
continue
if lowered in {"function", "procedure"}:
declaration_section = None
member, index = draft_method_member(
masked_source, tokens, index, "public", "instance", limit
)
member = as_unit_function(member)
if member is not None:
members.append(member)
continue
if lowered == "type":
declaration_section = None
if index + 3 < limit and keyword_at(tokens, index + 3, "class"):
class_data, index = draft_class(
masked_source,
tokens,
index,
path,
limit,
)
members.append({"kind": "class", **class_data})
else:
index = recover_after_declaration(tokens, index, limit)
continue
if lowered == "const":
declaration_section = "const"
constants, index = draft_constant_members(
masked_source, tokens, index, "public", False, limit
)
for constant in constants:
constant.pop("visibility", None)
members.extend(constants)
continue
if lowered == "var":
declaration_section = "var"
if index + 1 >= limit:
index += 1
continue
variables, index = draft_field_members(
masked_source, tokens, index + 1, "public", False, limit
)
for variable in variables:
variable["kind"] = "variable"
variable.pop("visibility", None)
members.extend(variables)
continue
if token.kind == "identifier" and declaration_section == "const":
constants, index = draft_constant_members(
masked_source,
tokens,
index,
"public",
False,
limit,
bare=True,
)
for constant in constants:
constant.pop("visibility", None)
members.extend(constants)
continue
if token.kind == "identifier" and declaration_section == "var":
variables, index = draft_field_members(
masked_source, tokens, index, "public", False, limit
)
for variable in variables:
variable["kind"] = "variable"
variable.pop("visibility", None)
members.extend(variables)
continue
index = recover_after_declaration(tokens, index, limit)
return {"name": name, "desc": "", "members": members}
def convert_source(source, path):
tokens = tokenize(source)
structure_token = first_structure_token(tokens)
if structure_token is None:
return draft_unknown_function(source, tokens, path)
structure_index = tokens.index(structure_token)
kind = structure_token.value.casefold()
if kind in {"function", "procedure"}:
try:
name, parameters, declared_return, begin_token = parse_declaration(
source, tokens, structure_token, path
)
except ConversionError:
name, parameters, declared_return = draft_function_signature(
source,
tokens,
structure_index,
path,
)
begin_token = function_begin_before_next_structure(
tokens,
structure_index + 1,
)
document = (
extract_document_lines(source, begin_token)
if begin_token is not None
else []
)
function = best_effort_function_details(
document, parameters, declared_return, path
)
if kind == "procedure":
function["returns"] = ""
names = ", ".join(parameter.name for parameter in parameters)
return {
"kind": "function",
"name": name,
"signature": f"{name}({names})",
**function,
}
if kind == "type":
try:
class_data, _ = parse_class(source, tokens, structure_index, path)
except ConversionError:
class_data, _ = draft_class(source, tokens, structure_index, path)
return {"kind": "class", **class_data}
if kind == "unit":
try:
unit_data = parse_unit(source, tokens, structure_index, path)
except ConversionError:
unit_data = draft_unit(source, tokens, structure_index, path)
return {"kind": "unit", **unit_data}
return draft_unknown_function(source, tokens, path)
def read_tsf(path):
if path.suffix.lower() != ".tsf":
fail(path, None, "输入文件扩展名必须是 .tsf")
if not path.is_file():
fail(path, None, "输入文件不存在")
source_bytes = path.read_bytes()
for encoding in ("utf-8-sig", "gb18030"):
try:
return source_bytes.decode(encoding)
except UnicodeDecodeError:
continue
fail(path, None, "文件不是有效的 utf-8 或 gb18030 编码")
def complete_draft_fields(item):
item.setdefault("desc", "")
kind = item["kind"]
if kind in {"function", "method"}:
item.setdefault("params", [])
item.setdefault("returns", "")
for parameter in item["params"]:
parameter.setdefault("type", "")
parameter.setdefault("desc", "")
return item
if kind == "property":
item.setdefault("type", "")
item.setdefault("params", [])
for parameter in item["params"]:
parameter.setdefault("type", "")
parameter.setdefault("desc", "")
return item
if kind in {"field", "variable", "constant"}:
item.setdefault("type", "")
return item
if kind in {"class", "unit"}:
item.setdefault("members", [])
for member in item["members"]:
complete_draft_fields(member)
return item
def serialize(data, output_format, output_path):
if output_format == "json":
return json.dumps(data, ensure_ascii=False, indent=2) + "\n"
try:
import yaml
except ImportError:
fail(
output_path, None, "生成 yaml 需要安装 pyyamlpython -m pip install pyyaml"
)
return yaml.safe_dump(
data,
allow_unicode=True,
sort_keys=False,
default_flow_style=False,
)
def atomic_write(path, text):
path.parent.mkdir(parents=True, exist_ok=True)
descriptor, temporary_name = tempfile.mkstemp(
prefix=f".{path.name}.", suffix=".tmp", dir=path.parent
)
temporary_path = Path(temporary_name)
try:
with os.fdopen(
descriptor, "w", encoding="utf-8", newline="\n"
) as handle:
handle.write(text)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary_path, path)
finally:
if temporary_path.exists():
temporary_path.unlink()
def build_parser():
parser = argparse.ArgumentParser(
description=(
"把 tsf function、class 和 unit "
"转换为 json 或 yaml declarations 录入文件"
),
formatter_class=ChineseHelpFormatter,
add_help=False,
allow_abbrev=False,
)
parser._positionals.title = "位置参数"
parser._optionals.title = "选项"
parser.add_argument(
"inputs",
nargs="+",
metavar="tsf文件",
help="一个或多个 tsf 文件;按给定顺序生成 declarations",
)
parser.add_argument(
"--format",
required=True,
choices=["json", "yaml"],
metavar="格式",
help="输出格式,可选 json 或 yaml",
)
parser.add_argument(
"--module",
required=True,
metavar="模块标题",
help="录入文件的 module 值,即 markdown 一级标题",
)
parser.add_argument(
"--path",
required=True,
metavar="页面路径",
help="录入文件的 path 值,不包含 .md 后缀",
)
parser.add_argument(
"--output",
required=True,
metavar="输出文件",
help="json 或 yaml 输出文件路径",
)
parser.add_argument(
"--help",
action="help",
help="显示此帮助信息并退出(不提供 -h 短选项)",
)
return parser
def main(argv=None):
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
if hasattr(sys.stderr, "reconfigure"):
sys.stderr.reconfigure(encoding="utf-8")
args = build_parser().parse_args(argv)
output_path = Path(args.output)
try:
input_paths = [Path(input_value) for input_value in args.inputs]
resolved_output = output_path.resolve()
if any(input_path.resolve() == resolved_output for input_path in input_paths):
fail(output_path, None, "输出文件不能覆盖输入 tsf")
declarations = [
complete_draft_fields(convert_source(read_tsf(input_path), input_path))
for input_path in input_paths
]
data = {
"module": args.module,
"path": args.path,
"declarations": declarations,
}
output_text = serialize(data, args.format, output_path)
atomic_write(output_path, output_text)
except ConversionError as exc:
print(f"错误:{exc}", file=sys.stderr)
return 1
except OSError as exc:
print(f"错误:{exc}", file=sys.stderr)
return 1
print(f"已写入 {output_path}", file=sys.stderr)
return 0
if __name__ == "__main__":
raise SystemExit(main())