Files
2026-09-04 14:58:42 +08:00

425 lines
14 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Source-preserving Markdown block analysis and static annotations."""
from __future__ import annotations
from collections import defaultdict
from dataclasses import dataclass
import re
from typing import Iterable
import yaml
from markdown_it import MarkdownIt
from .models import Annotation, BodyBlock
class DocumentError(RuntimeError):
"""The Skill Markdown cannot be parsed safely."""
HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*$")
FENCE_RE = re.compile(r"^[ \t]*(```+|~~~+)")
LIST_RE = re.compile(r"^([ \t]*)(?:[-+*]|\d+[.)])[ \t]+")
TABLE_RE = re.compile(r"^[ \t]*\|.*\|[ \t]*(?:\r?\n)?$")
INLINE_PROTECTED_RE = re.compile(
r"`[^`\n]+`|https?://[^\s)>]+|(?<![A-Za-z0-9_])(?:\./|\.\./|/)"
r"[A-Za-z0-9_./{}$@%:+-]*[A-Za-z0-9_/{}$@%:+-]"
r"|\b[A-Za-z_][A-Za-z0-9_]*\.(?:json|ya?ml|toml|md|py|sh|js|ts|csv|xml)\b"
r"|\b\d+(?:\.\d+)*%?\b"
)
SECTION_ALIASES = {
"definitions": {"definitions", "definition", "术语", "术语定义", "定义"},
"critical_rules": {
"critical rules",
"critical rule",
"rules",
"constraints",
"关键规则",
"规则",
"约束",
},
"evidence_priority": {"evidence priority", "证据优先级"},
"scope": {"scope", "范围"},
"decision_criteria": {"decision criteria", "criteria", "判断标准", "决策标准"},
"uncertainty_rule": {"uncertainty rule", "uncertainty", "不确定性规则"},
"completion_criterion": {
"completion criterion",
"completion criteria",
"completion",
"完成条件",
},
"task": {"task", "workflow", "instructions", "任务", "工作流", "步骤", "执行"},
"inputs": {"input", "inputs", "输入"},
"output": {"output", "outputs", "输出"},
"validation": {"validation", "validate", "checks", "验证", "检查"},
}
CANONICAL_HEADINGS = {
"en": {
"definitions": "Definitions",
"critical_rules": "Critical Rules",
"evidence_priority": "Evidence Priority",
"scope": "Scope",
"decision_criteria": "Decision Criteria",
"uncertainty_rule": "Uncertainty Rule",
"completion_criterion": "Completion Criterion",
"task": "Task",
"inputs": "Inputs",
"output": "Output",
"validation": "Validation",
},
"zh": {
"definitions": "术语定义",
"critical_rules": "关键规则",
"evidence_priority": "证据优先级",
"scope": "范围",
"decision_criteria": "判断标准",
"uncertainty_rule": "不确定性规则",
"completion_criterion": "完成条件",
"task": "任务",
"inputs": "输入",
"output": "输出",
"validation": "验证",
},
}
ANNOTATION_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
(
"completion_criterion",
re.compile(
r"只有.+才(?:算|可以|可|能).*(?:完成|结束)|完成条件\s*[::]|"
r"only\s+.+\s+(?:counts?\s+as|is)\s+(?:complete|done)",
re.I,
),
),
(
"evidence_priority_rule",
re.compile(
r"以.+为准|.+优先于.+|(?:冲突|不一致)时.+(?:为准|优先)|"
r"\b.+takes?\s+precedence\s+over\b.+|\bprefer\s+.+\s+over\b",
re.I,
),
),
(
"uncertainty_rule",
re.compile(
r"无法确定|证据不足|不得猜测|不要猜测|不应推断|"
r"\bdo\s+not\s+guess\b|\binsufficient\s+evidence\b|\buncertain\b",
re.I,
),
),
(
"scope_rule",
re.compile(
r"仅指|不包括|范围为|范围包括|\bscope\s*[::]|\bdoes\s+not\s+include\b",
re.I,
),
),
(
"decision_criterion",
re.compile(
r"按.+(?:排序|判断)|根据.+判断|判断标准\s*[::]|\bcriteria\s*[::]",
re.I,
),
),
(
"definition",
re.compile(
r"(?:此处|这里|本任务中).+?(?:是指|指的是|定义为)|"
r"^[A-Za-z][A-Za-z0-9 _-]{0,40}\s+(?:means|refers to|is defined as)\b",
re.I,
),
),
(
"critical_rule",
re.compile(
r"\bMUST(?:\s+NOT)?\b|\b(?:IMPORTANT|CRITICAL)\s*[::]|"
r"必须|不得|禁止|仅可|只能|不能",
re.I,
),
),
]
VIEWPOINT_A_RE = re.compile(
r"^(?:[-+*]\s*)?(?:支持|赞成|优点|收益|采用|in favor|advantages?|benefits?)\s*[::]",
re.I,
)
VIEWPOINT_B_RE = re.compile(
r"^(?:[-+*]\s*)?(?:反对|缺点|风险|不采用|against|disadvantages?|risks?)\s*[::]",
re.I,
)
@dataclass
class SkillDocument:
original: str
frontmatter: str
body: str
blocks: list[BodyBlock]
newline: str
@property
def block_index(self) -> dict[str, BodyBlock]:
return {block.id: block for block in self.blocks}
@property
def language(self) -> str:
nonspace = [char for char in self.body if not char.isspace()]
if not nonspace:
return "en"
cjk = sum("\u4e00" <= char <= "\u9fff" for char in nonspace)
return "zh" if cjk / len(nonspace) >= 0.30 else "en"
def split_frontmatter(content: str) -> tuple[str, str]:
if not content.startswith("---"):
raise DocumentError("SKILL.md requires YAML frontmatter")
match = re.search(r"\A---[ \t]*\r?\n.*?\r?\n---[ \t]*(?:\r?\n|\Z)", content, re.S)
if not match:
raise DocumentError("unterminated YAML frontmatter")
frontmatter = match.group(0)
yaml_text = re.sub(r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", frontmatter)
try:
loaded = yaml.safe_load(yaml_text)
except yaml.YAMLError as exc:
raise DocumentError(f"invalid YAML frontmatter: {exc}") from exc
if not isinstance(loaded, dict):
raise DocumentError("YAML frontmatter must be a mapping")
return frontmatter, content[match.end() :]
def _protected_spans(text: str, *, whole_block: bool = False) -> list[tuple[int, int]]:
if whole_block:
return [(0, len(text))]
return [(match.start(), match.end()) for match in INLINE_PROTECTED_RE.finditer(text)]
def _looks_like_code(text: str) -> bool:
lines = [line for line in text.splitlines() if line.strip()]
if len(lines) < 2:
return False
code_line = re.compile(
r"^[ \t]{2,}(?:def |class |if |elif |else:|for |while |return |"
r"print\(|raise |try:|except |[A-Za-z_][A-Za-z0-9_]*\s*=|[}\]])"
)
signals = sum(bool(code_line.match(line)) for line in lines)
return signals >= 2 and signals >= len(lines) / 2
def _line_offsets(body: str) -> tuple[list[str], list[int]]:
lines = body.splitlines(keepends=True)
if body and not lines:
lines = [body]
offsets: list[int] = []
position = 0
for line in lines:
offsets.append(position)
position += len(line)
return lines, offsets
def parse_document(content: str) -> SkillDocument:
frontmatter, body = split_frontmatter(content)
newline = "\r\n" if "\r\n" in content else "\n"
MarkdownIt("commonmark", {"html": True}).parse(body)
lines, offsets = _line_offsets(body)
blocks: list[BodyBlock] = []
index = 0
parent_heading: str | None = None
block_number = 0
def add_block(start: int, end: int, kind: str, heading_level: int | None = None) -> None:
nonlocal block_number, parent_heading
raw = "".join(lines[start:end]).rstrip("\r\n")
if not raw:
return
if kind in {"paragraph", "list_item"} and _looks_like_code(raw):
kind = "code_like"
block_number += 1
block_id = f"B{block_number:03d}"
start_offset = offsets[start]
end_offset = start_offset + len("".join(lines[start:end]))
list_match = LIST_RE.match(raw)
block = BodyBlock(
id=block_id,
kind=kind,
text=raw,
start_line=start + 1,
end_line=end,
start_offset=start_offset,
end_offset=end_offset,
parent_heading=parent_heading,
heading_level=heading_level,
list_depth=(len(list_match.group(1).replace("\t", " ")) // 2 if list_match else 0),
protected_spans=_protected_spans(
raw, whole_block=kind in {"code", "code_like", "html", "table"}
),
)
blocks.append(block)
if kind == "heading":
heading = HEADING_RE.match(raw)
parent_heading = heading.group(2).strip() if heading else raw
while index < len(lines):
stripped = lines[index].strip()
if not stripped:
index += 1
continue
fence = FENCE_RE.match(lines[index])
if fence:
marker = fence.group(1)[0]
end = index + 1
while end < len(lines) and not re.match(rf"^[ \t]*{re.escape(marker)}{{3,}}", lines[end]):
end += 1
end = min(end + 1, len(lines))
add_block(index, end, "code")
index = end
continue
heading = HEADING_RE.match(lines[index].rstrip("\r\n"))
if heading:
add_block(index, index + 1, "heading", len(heading.group(1)))
index += 1
continue
if lines[index].lstrip().startswith("<"):
add_block(index, index + 1, "html")
index += 1
continue
if TABLE_RE.match(lines[index]):
end = index + 1
while end < len(lines) and TABLE_RE.match(lines[end]):
end += 1
add_block(index, end, "table")
index = end
continue
if LIST_RE.match(lines[index]):
end = index + 1
while (
end < len(lines)
and lines[end].strip()
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
and not LIST_RE.match(lines[end])
and not FENCE_RE.match(lines[end])
):
end += 1
add_block(index, end, "list_item")
index = end
continue
end = index + 1
while (
end < len(lines)
and lines[end].strip()
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
and not LIST_RE.match(lines[end])
and not TABLE_RE.match(lines[end])
and not FENCE_RE.match(lines[end])
):
end += 1
add_block(index, end, "paragraph")
index = end
return SkillDocument(content, frontmatter, body, blocks, newline)
def section_key(title: str | None) -> str | None:
if not title:
return None
normalized = " ".join(title.lower().strip().rstrip("::-–—").split())
for key, aliases in SECTION_ALIASES.items():
if normalized in aliases:
return key
return None
def _sentences(text: str) -> Iterable[str]:
prefix = ""
list_match = LIST_RE.match(text)
content = text
if list_match:
prefix = text[: list_match.end()]
content = text[list_match.end() :]
parts = re.split(r"(?<=[。!?.!?;;])(?:[ \t]+|\r?\n+)", content)
for index, part in enumerate(parts):
clean = part.strip()
if clean:
yield (prefix if index == 0 else "") + clean
def static_annotations(document: SkillDocument) -> list[Annotation]:
annotations: list[Annotation] = []
for block in document.blocks:
if block.kind in {"code", "code_like", "html", "heading", "table"}:
continue
parent_key = section_key(block.parent_heading)
candidates = list(_sentences(block.text))
for quote in candidates:
found: list[str] = []
if parent_key == "definitions":
found.append("definition")
elif parent_key == "completion_criterion":
found.append("completion_criterion")
elif parent_key == "evidence_priority":
found.append("evidence_priority_rule")
elif parent_key == "scope":
found.append("scope_rule")
elif parent_key == "decision_criteria":
found.append("decision_criterion")
elif parent_key == "uncertainty_rule":
found.append("uncertainty_rule")
elif parent_key == "critical_rules":
found.append("critical_rule")
for annotation_type, pattern in ANNOTATION_PATTERNS:
if pattern.search(quote):
found.append(annotation_type)
if VIEWPOINT_A_RE.search(quote):
found.append("viewpoint_side_a")
if VIEWPOINT_B_RE.search(quote):
found.append("viewpoint_side_b")
for annotation_type in dict.fromkeys(found):
annotations.append(
Annotation(annotation_type, block.id, quote, 1.0, "static")
)
return resolve_annotation_conflicts(annotations)
ANNOTATION_PRIORITY = {
"completion_criterion": 100,
"evidence_priority_rule": 90,
"uncertainty_rule": 80,
"scope_rule": 70,
"decision_criterion": 60,
"definition": 50,
"viewpoint_side_a": 40,
"viewpoint_side_b": 40,
"critical_rule": 10,
"coreference": 5,
}
def resolve_annotation_conflicts(annotations: list[Annotation]) -> list[Annotation]:
grouped: dict[tuple[str, str], list[Annotation]] = defaultdict(list)
for annotation in annotations:
grouped[(annotation.block_id, annotation.quote)].append(annotation)
resolved: list[Annotation] = []
for values in grouped.values():
values.sort(
key=lambda item: (
ANNOTATION_PRIORITY.get(item.type, 0),
item.confidence,
item.source == "static",
),
reverse=True,
)
resolved.append(values[0])
return sorted(resolved, key=lambda item: (item.block_id, item.quote))
def skill_name(document: SkillDocument) -> str:
yaml_text = re.sub(
r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", document.frontmatter
)
loaded = yaml.safe_load(yaml_text)
name = loaded.get("name") if isinstance(loaded, dict) else None
if not isinstance(name, str) or not name.strip():
raise DocumentError("frontmatter requires non-empty name")
return name.strip()