Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+424
View File
@@ -0,0 +1,424 @@
"""Source-preserving Markdown block analysis and static annotations."""
from __future__ import annotations
from collections import defaultdict
from dataclasses import dataclass
import re
from typing import Iterable
import yaml
from markdown_it import MarkdownIt
from .models import Annotation, BodyBlock
class DocumentError(RuntimeError):
"""The Skill Markdown cannot be parsed safely."""
HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*$")
FENCE_RE = re.compile(r"^[ \t]*(```+|~~~+)")
LIST_RE = re.compile(r"^([ \t]*)(?:[-+*]|\d+[.)])[ \t]+")
TABLE_RE = re.compile(r"^[ \t]*\|.*\|[ \t]*(?:\r?\n)?$")
INLINE_PROTECTED_RE = re.compile(
r"`[^`\n]+`|https?://[^\s)>]+|(?<![A-Za-z0-9_])(?:\./|\.\./|/)"
r"[A-Za-z0-9_./{}$@%:+-]*[A-Za-z0-9_/{}$@%:+-]"
r"|\b[A-Za-z_][A-Za-z0-9_]*\.(?:json|ya?ml|toml|md|py|sh|js|ts|csv|xml)\b"
r"|\b\d+(?:\.\d+)*%?\b"
)
SECTION_ALIASES = {
"definitions": {"definitions", "definition", "术语", "术语定义", "定义"},
"critical_rules": {
"critical rules",
"critical rule",
"rules",
"constraints",
"关键规则",
"规则",
"约束",
},
"evidence_priority": {"evidence priority", "证据优先级"},
"scope": {"scope", "范围"},
"decision_criteria": {"decision criteria", "criteria", "判断标准", "决策标准"},
"uncertainty_rule": {"uncertainty rule", "uncertainty", "不确定性规则"},
"completion_criterion": {
"completion criterion",
"completion criteria",
"completion",
"完成条件",
},
"task": {"task", "workflow", "instructions", "任务", "工作流", "步骤", "执行"},
"inputs": {"input", "inputs", "输入"},
"output": {"output", "outputs", "输出"},
"validation": {"validation", "validate", "checks", "验证", "检查"},
}
CANONICAL_HEADINGS = {
"en": {
"definitions": "Definitions",
"critical_rules": "Critical Rules",
"evidence_priority": "Evidence Priority",
"scope": "Scope",
"decision_criteria": "Decision Criteria",
"uncertainty_rule": "Uncertainty Rule",
"completion_criterion": "Completion Criterion",
"task": "Task",
"inputs": "Inputs",
"output": "Output",
"validation": "Validation",
},
"zh": {
"definitions": "术语定义",
"critical_rules": "关键规则",
"evidence_priority": "证据优先级",
"scope": "范围",
"decision_criteria": "判断标准",
"uncertainty_rule": "不确定性规则",
"completion_criterion": "完成条件",
"task": "任务",
"inputs": "输入",
"output": "输出",
"validation": "验证",
},
}
ANNOTATION_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
(
"completion_criterion",
re.compile(
r"只有.+才(?:算|可以|可|能).*(?:完成|结束)|完成条件\s*[::]|"
r"only\s+.+\s+(?:counts?\s+as|is)\s+(?:complete|done)",
re.I,
),
),
(
"evidence_priority_rule",
re.compile(
r"以.+为准|.+优先于.+|(?:冲突|不一致)时.+(?:为准|优先)|"
r"\b.+takes?\s+precedence\s+over\b.+|\bprefer\s+.+\s+over\b",
re.I,
),
),
(
"uncertainty_rule",
re.compile(
r"无法确定|证据不足|不得猜测|不要猜测|不应推断|"
r"\bdo\s+not\s+guess\b|\binsufficient\s+evidence\b|\buncertain\b",
re.I,
),
),
(
"scope_rule",
re.compile(
r"仅指|不包括|范围为|范围包括|\bscope\s*[::]|\bdoes\s+not\s+include\b",
re.I,
),
),
(
"decision_criterion",
re.compile(
r"按.+(?:排序|判断)|根据.+判断|判断标准\s*[::]|\bcriteria\s*[::]",
re.I,
),
),
(
"definition",
re.compile(
r"(?:此处|这里|本任务中).+?(?:是指|指的是|定义为)|"
r"^[A-Za-z][A-Za-z0-9 _-]{0,40}\s+(?:means|refers to|is defined as)\b",
re.I,
),
),
(
"critical_rule",
re.compile(
r"\bMUST(?:\s+NOT)?\b|\b(?:IMPORTANT|CRITICAL)\s*[::]|"
r"必须|不得|禁止|仅可|只能|不能",
re.I,
),
),
]
VIEWPOINT_A_RE = re.compile(
r"^(?:[-+*]\s*)?(?:支持|赞成|优点|收益|采用|in favor|advantages?|benefits?)\s*[::]",
re.I,
)
VIEWPOINT_B_RE = re.compile(
r"^(?:[-+*]\s*)?(?:反对|缺点|风险|不采用|against|disadvantages?|risks?)\s*[::]",
re.I,
)
@dataclass
class SkillDocument:
original: str
frontmatter: str
body: str
blocks: list[BodyBlock]
newline: str
@property
def block_index(self) -> dict[str, BodyBlock]:
return {block.id: block for block in self.blocks}
@property
def language(self) -> str:
nonspace = [char for char in self.body if not char.isspace()]
if not nonspace:
return "en"
cjk = sum("\u4e00" <= char <= "\u9fff" for char in nonspace)
return "zh" if cjk / len(nonspace) >= 0.30 else "en"
def split_frontmatter(content: str) -> tuple[str, str]:
if not content.startswith("---"):
raise DocumentError("SKILL.md requires YAML frontmatter")
match = re.search(r"\A---[ \t]*\r?\n.*?\r?\n---[ \t]*(?:\r?\n|\Z)", content, re.S)
if not match:
raise DocumentError("unterminated YAML frontmatter")
frontmatter = match.group(0)
yaml_text = re.sub(r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", frontmatter)
try:
loaded = yaml.safe_load(yaml_text)
except yaml.YAMLError as exc:
raise DocumentError(f"invalid YAML frontmatter: {exc}") from exc
if not isinstance(loaded, dict):
raise DocumentError("YAML frontmatter must be a mapping")
return frontmatter, content[match.end() :]
def _protected_spans(text: str, *, whole_block: bool = False) -> list[tuple[int, int]]:
if whole_block:
return [(0, len(text))]
return [(match.start(), match.end()) for match in INLINE_PROTECTED_RE.finditer(text)]
def _looks_like_code(text: str) -> bool:
lines = [line for line in text.splitlines() if line.strip()]
if len(lines) < 2:
return False
code_line = re.compile(
r"^[ \t]{2,}(?:def |class |if |elif |else:|for |while |return |"
r"print\(|raise |try:|except |[A-Za-z_][A-Za-z0-9_]*\s*=|[}\]])"
)
signals = sum(bool(code_line.match(line)) for line in lines)
return signals >= 2 and signals >= len(lines) / 2
def _line_offsets(body: str) -> tuple[list[str], list[int]]:
lines = body.splitlines(keepends=True)
if body and not lines:
lines = [body]
offsets: list[int] = []
position = 0
for line in lines:
offsets.append(position)
position += len(line)
return lines, offsets
def parse_document(content: str) -> SkillDocument:
frontmatter, body = split_frontmatter(content)
newline = "\r\n" if "\r\n" in content else "\n"
MarkdownIt("commonmark", {"html": True}).parse(body)
lines, offsets = _line_offsets(body)
blocks: list[BodyBlock] = []
index = 0
parent_heading: str | None = None
block_number = 0
def add_block(start: int, end: int, kind: str, heading_level: int | None = None) -> None:
nonlocal block_number, parent_heading
raw = "".join(lines[start:end]).rstrip("\r\n")
if not raw:
return
if kind in {"paragraph", "list_item"} and _looks_like_code(raw):
kind = "code_like"
block_number += 1
block_id = f"B{block_number:03d}"
start_offset = offsets[start]
end_offset = start_offset + len("".join(lines[start:end]))
list_match = LIST_RE.match(raw)
block = BodyBlock(
id=block_id,
kind=kind,
text=raw,
start_line=start + 1,
end_line=end,
start_offset=start_offset,
end_offset=end_offset,
parent_heading=parent_heading,
heading_level=heading_level,
list_depth=(len(list_match.group(1).replace("\t", " ")) // 2 if list_match else 0),
protected_spans=_protected_spans(
raw, whole_block=kind in {"code", "code_like", "html", "table"}
),
)
blocks.append(block)
if kind == "heading":
heading = HEADING_RE.match(raw)
parent_heading = heading.group(2).strip() if heading else raw
while index < len(lines):
stripped = lines[index].strip()
if not stripped:
index += 1
continue
fence = FENCE_RE.match(lines[index])
if fence:
marker = fence.group(1)[0]
end = index + 1
while end < len(lines) and not re.match(rf"^[ \t]*{re.escape(marker)}{{3,}}", lines[end]):
end += 1
end = min(end + 1, len(lines))
add_block(index, end, "code")
index = end
continue
heading = HEADING_RE.match(lines[index].rstrip("\r\n"))
if heading:
add_block(index, index + 1, "heading", len(heading.group(1)))
index += 1
continue
if lines[index].lstrip().startswith("<"):
add_block(index, index + 1, "html")
index += 1
continue
if TABLE_RE.match(lines[index]):
end = index + 1
while end < len(lines) and TABLE_RE.match(lines[end]):
end += 1
add_block(index, end, "table")
index = end
continue
if LIST_RE.match(lines[index]):
end = index + 1
while (
end < len(lines)
and lines[end].strip()
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
and not LIST_RE.match(lines[end])
and not FENCE_RE.match(lines[end])
):
end += 1
add_block(index, end, "list_item")
index = end
continue
end = index + 1
while (
end < len(lines)
and lines[end].strip()
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
and not LIST_RE.match(lines[end])
and not TABLE_RE.match(lines[end])
and not FENCE_RE.match(lines[end])
):
end += 1
add_block(index, end, "paragraph")
index = end
return SkillDocument(content, frontmatter, body, blocks, newline)
def section_key(title: str | None) -> str | None:
if not title:
return None
normalized = " ".join(title.lower().strip().rstrip("::-–—").split())
for key, aliases in SECTION_ALIASES.items():
if normalized in aliases:
return key
return None
def _sentences(text: str) -> Iterable[str]:
prefix = ""
list_match = LIST_RE.match(text)
content = text
if list_match:
prefix = text[: list_match.end()]
content = text[list_match.end() :]
parts = re.split(r"(?<=[。!?.!?;;])(?:[ \t]+|\r?\n+)", content)
for index, part in enumerate(parts):
clean = part.strip()
if clean:
yield (prefix if index == 0 else "") + clean
def static_annotations(document: SkillDocument) -> list[Annotation]:
annotations: list[Annotation] = []
for block in document.blocks:
if block.kind in {"code", "code_like", "html", "heading", "table"}:
continue
parent_key = section_key(block.parent_heading)
candidates = list(_sentences(block.text))
for quote in candidates:
found: list[str] = []
if parent_key == "definitions":
found.append("definition")
elif parent_key == "completion_criterion":
found.append("completion_criterion")
elif parent_key == "evidence_priority":
found.append("evidence_priority_rule")
elif parent_key == "scope":
found.append("scope_rule")
elif parent_key == "decision_criteria":
found.append("decision_criterion")
elif parent_key == "uncertainty_rule":
found.append("uncertainty_rule")
elif parent_key == "critical_rules":
found.append("critical_rule")
for annotation_type, pattern in ANNOTATION_PATTERNS:
if pattern.search(quote):
found.append(annotation_type)
if VIEWPOINT_A_RE.search(quote):
found.append("viewpoint_side_a")
if VIEWPOINT_B_RE.search(quote):
found.append("viewpoint_side_b")
for annotation_type in dict.fromkeys(found):
annotations.append(
Annotation(annotation_type, block.id, quote, 1.0, "static")
)
return resolve_annotation_conflicts(annotations)
ANNOTATION_PRIORITY = {
"completion_criterion": 100,
"evidence_priority_rule": 90,
"uncertainty_rule": 80,
"scope_rule": 70,
"decision_criterion": 60,
"definition": 50,
"viewpoint_side_a": 40,
"viewpoint_side_b": 40,
"critical_rule": 10,
"coreference": 5,
}
def resolve_annotation_conflicts(annotations: list[Annotation]) -> list[Annotation]:
grouped: dict[tuple[str, str], list[Annotation]] = defaultdict(list)
for annotation in annotations:
grouped[(annotation.block_id, annotation.quote)].append(annotation)
resolved: list[Annotation] = []
for values in grouped.values():
values.sort(
key=lambda item: (
ANNOTATION_PRIORITY.get(item.type, 0),
item.confidence,
item.source == "static",
),
reverse=True,
)
resolved.append(values[0])
return sorted(resolved, key=lambda item: (item.block_id, item.quote))
def skill_name(document: SkillDocument) -> str:
yaml_text = re.sub(
r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", document.frontmatter
)
loaded = yaml.safe_load(yaml_text)
name = loaded.get("name") if isinstance(loaded, dict) else None
if not isinstance(name, str) or not name.strip():
raise DocumentError("frontmatter requires non-empty name")
return name.strip()