Initial commit
This commit is contained in:
@@ -0,0 +1,424 @@
|
||||
"""Source-preserving Markdown block analysis and static annotations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
import re
|
||||
from typing import Iterable
|
||||
|
||||
import yaml
|
||||
from markdown_it import MarkdownIt
|
||||
|
||||
from .models import Annotation, BodyBlock
|
||||
|
||||
|
||||
class DocumentError(RuntimeError):
|
||||
"""The Skill Markdown cannot be parsed safely."""
|
||||
|
||||
|
||||
HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*$")
|
||||
FENCE_RE = re.compile(r"^[ \t]*(```+|~~~+)")
|
||||
LIST_RE = re.compile(r"^([ \t]*)(?:[-+*]|\d+[.)])[ \t]+")
|
||||
TABLE_RE = re.compile(r"^[ \t]*\|.*\|[ \t]*(?:\r?\n)?$")
|
||||
INLINE_PROTECTED_RE = re.compile(
|
||||
r"`[^`\n]+`|https?://[^\s)>]+|(?<![A-Za-z0-9_])(?:\./|\.\./|/)"
|
||||
r"[A-Za-z0-9_./{}$@%:+-]*[A-Za-z0-9_/{}$@%:+-]"
|
||||
r"|\b[A-Za-z_][A-Za-z0-9_]*\.(?:json|ya?ml|toml|md|py|sh|js|ts|csv|xml)\b"
|
||||
r"|\b\d+(?:\.\d+)*%?\b"
|
||||
)
|
||||
|
||||
SECTION_ALIASES = {
|
||||
"definitions": {"definitions", "definition", "术语", "术语定义", "定义"},
|
||||
"critical_rules": {
|
||||
"critical rules",
|
||||
"critical rule",
|
||||
"rules",
|
||||
"constraints",
|
||||
"关键规则",
|
||||
"规则",
|
||||
"约束",
|
||||
},
|
||||
"evidence_priority": {"evidence priority", "证据优先级"},
|
||||
"scope": {"scope", "范围"},
|
||||
"decision_criteria": {"decision criteria", "criteria", "判断标准", "决策标准"},
|
||||
"uncertainty_rule": {"uncertainty rule", "uncertainty", "不确定性规则"},
|
||||
"completion_criterion": {
|
||||
"completion criterion",
|
||||
"completion criteria",
|
||||
"completion",
|
||||
"完成条件",
|
||||
},
|
||||
"task": {"task", "workflow", "instructions", "任务", "工作流", "步骤", "执行"},
|
||||
"inputs": {"input", "inputs", "输入"},
|
||||
"output": {"output", "outputs", "输出"},
|
||||
"validation": {"validation", "validate", "checks", "验证", "检查"},
|
||||
}
|
||||
|
||||
CANONICAL_HEADINGS = {
|
||||
"en": {
|
||||
"definitions": "Definitions",
|
||||
"critical_rules": "Critical Rules",
|
||||
"evidence_priority": "Evidence Priority",
|
||||
"scope": "Scope",
|
||||
"decision_criteria": "Decision Criteria",
|
||||
"uncertainty_rule": "Uncertainty Rule",
|
||||
"completion_criterion": "Completion Criterion",
|
||||
"task": "Task",
|
||||
"inputs": "Inputs",
|
||||
"output": "Output",
|
||||
"validation": "Validation",
|
||||
},
|
||||
"zh": {
|
||||
"definitions": "术语定义",
|
||||
"critical_rules": "关键规则",
|
||||
"evidence_priority": "证据优先级",
|
||||
"scope": "范围",
|
||||
"decision_criteria": "判断标准",
|
||||
"uncertainty_rule": "不确定性规则",
|
||||
"completion_criterion": "完成条件",
|
||||
"task": "任务",
|
||||
"inputs": "输入",
|
||||
"output": "输出",
|
||||
"validation": "验证",
|
||||
},
|
||||
}
|
||||
|
||||
ANNOTATION_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
|
||||
(
|
||||
"completion_criterion",
|
||||
re.compile(
|
||||
r"只有.+才(?:算|可以|可|能).*(?:完成|结束)|完成条件\s*[::]|"
|
||||
r"only\s+.+\s+(?:counts?\s+as|is)\s+(?:complete|done)",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"evidence_priority_rule",
|
||||
re.compile(
|
||||
r"以.+为准|.+优先于.+|(?:冲突|不一致)时.+(?:为准|优先)|"
|
||||
r"\b.+takes?\s+precedence\s+over\b.+|\bprefer\s+.+\s+over\b",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"uncertainty_rule",
|
||||
re.compile(
|
||||
r"无法确定|证据不足|不得猜测|不要猜测|不应推断|"
|
||||
r"\bdo\s+not\s+guess\b|\binsufficient\s+evidence\b|\buncertain\b",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"scope_rule",
|
||||
re.compile(
|
||||
r"仅指|不包括|范围为|范围包括|\bscope\s*[::]|\bdoes\s+not\s+include\b",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"decision_criterion",
|
||||
re.compile(
|
||||
r"按.+(?:排序|判断)|根据.+判断|判断标准\s*[::]|\bcriteria\s*[::]",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"definition",
|
||||
re.compile(
|
||||
r"(?:此处|这里|本任务中).+?(?:是指|指的是|定义为)|"
|
||||
r"^[A-Za-z][A-Za-z0-9 _-]{0,40}\s+(?:means|refers to|is defined as)\b",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
(
|
||||
"critical_rule",
|
||||
re.compile(
|
||||
r"\bMUST(?:\s+NOT)?\b|\b(?:IMPORTANT|CRITICAL)\s*[::]|"
|
||||
r"必须|不得|禁止|仅可|只能|不能",
|
||||
re.I,
|
||||
),
|
||||
),
|
||||
]
|
||||
|
||||
VIEWPOINT_A_RE = re.compile(
|
||||
r"^(?:[-+*]\s*)?(?:支持|赞成|优点|收益|采用|in favor|advantages?|benefits?)\s*[::]",
|
||||
re.I,
|
||||
)
|
||||
VIEWPOINT_B_RE = re.compile(
|
||||
r"^(?:[-+*]\s*)?(?:反对|缺点|风险|不采用|against|disadvantages?|risks?)\s*[::]",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkillDocument:
|
||||
original: str
|
||||
frontmatter: str
|
||||
body: str
|
||||
blocks: list[BodyBlock]
|
||||
newline: str
|
||||
|
||||
@property
|
||||
def block_index(self) -> dict[str, BodyBlock]:
|
||||
return {block.id: block for block in self.blocks}
|
||||
|
||||
@property
|
||||
def language(self) -> str:
|
||||
nonspace = [char for char in self.body if not char.isspace()]
|
||||
if not nonspace:
|
||||
return "en"
|
||||
cjk = sum("\u4e00" <= char <= "\u9fff" for char in nonspace)
|
||||
return "zh" if cjk / len(nonspace) >= 0.30 else "en"
|
||||
|
||||
|
||||
def split_frontmatter(content: str) -> tuple[str, str]:
|
||||
if not content.startswith("---"):
|
||||
raise DocumentError("SKILL.md requires YAML frontmatter")
|
||||
match = re.search(r"\A---[ \t]*\r?\n.*?\r?\n---[ \t]*(?:\r?\n|\Z)", content, re.S)
|
||||
if not match:
|
||||
raise DocumentError("unterminated YAML frontmatter")
|
||||
frontmatter = match.group(0)
|
||||
yaml_text = re.sub(r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", frontmatter)
|
||||
try:
|
||||
loaded = yaml.safe_load(yaml_text)
|
||||
except yaml.YAMLError as exc:
|
||||
raise DocumentError(f"invalid YAML frontmatter: {exc}") from exc
|
||||
if not isinstance(loaded, dict):
|
||||
raise DocumentError("YAML frontmatter must be a mapping")
|
||||
return frontmatter, content[match.end() :]
|
||||
|
||||
|
||||
def _protected_spans(text: str, *, whole_block: bool = False) -> list[tuple[int, int]]:
|
||||
if whole_block:
|
||||
return [(0, len(text))]
|
||||
return [(match.start(), match.end()) for match in INLINE_PROTECTED_RE.finditer(text)]
|
||||
|
||||
|
||||
def _looks_like_code(text: str) -> bool:
|
||||
lines = [line for line in text.splitlines() if line.strip()]
|
||||
if len(lines) < 2:
|
||||
return False
|
||||
code_line = re.compile(
|
||||
r"^[ \t]{2,}(?:def |class |if |elif |else:|for |while |return |"
|
||||
r"print\(|raise |try:|except |[A-Za-z_][A-Za-z0-9_]*\s*=|[}\]])"
|
||||
)
|
||||
signals = sum(bool(code_line.match(line)) for line in lines)
|
||||
return signals >= 2 and signals >= len(lines) / 2
|
||||
|
||||
|
||||
def _line_offsets(body: str) -> tuple[list[str], list[int]]:
|
||||
lines = body.splitlines(keepends=True)
|
||||
if body and not lines:
|
||||
lines = [body]
|
||||
offsets: list[int] = []
|
||||
position = 0
|
||||
for line in lines:
|
||||
offsets.append(position)
|
||||
position += len(line)
|
||||
return lines, offsets
|
||||
|
||||
|
||||
def parse_document(content: str) -> SkillDocument:
|
||||
frontmatter, body = split_frontmatter(content)
|
||||
newline = "\r\n" if "\r\n" in content else "\n"
|
||||
MarkdownIt("commonmark", {"html": True}).parse(body)
|
||||
lines, offsets = _line_offsets(body)
|
||||
blocks: list[BodyBlock] = []
|
||||
index = 0
|
||||
parent_heading: str | None = None
|
||||
block_number = 0
|
||||
|
||||
def add_block(start: int, end: int, kind: str, heading_level: int | None = None) -> None:
|
||||
nonlocal block_number, parent_heading
|
||||
raw = "".join(lines[start:end]).rstrip("\r\n")
|
||||
if not raw:
|
||||
return
|
||||
if kind in {"paragraph", "list_item"} and _looks_like_code(raw):
|
||||
kind = "code_like"
|
||||
block_number += 1
|
||||
block_id = f"B{block_number:03d}"
|
||||
start_offset = offsets[start]
|
||||
end_offset = start_offset + len("".join(lines[start:end]))
|
||||
list_match = LIST_RE.match(raw)
|
||||
block = BodyBlock(
|
||||
id=block_id,
|
||||
kind=kind,
|
||||
text=raw,
|
||||
start_line=start + 1,
|
||||
end_line=end,
|
||||
start_offset=start_offset,
|
||||
end_offset=end_offset,
|
||||
parent_heading=parent_heading,
|
||||
heading_level=heading_level,
|
||||
list_depth=(len(list_match.group(1).replace("\t", " ")) // 2 if list_match else 0),
|
||||
protected_spans=_protected_spans(
|
||||
raw, whole_block=kind in {"code", "code_like", "html", "table"}
|
||||
),
|
||||
)
|
||||
blocks.append(block)
|
||||
if kind == "heading":
|
||||
heading = HEADING_RE.match(raw)
|
||||
parent_heading = heading.group(2).strip() if heading else raw
|
||||
|
||||
while index < len(lines):
|
||||
stripped = lines[index].strip()
|
||||
if not stripped:
|
||||
index += 1
|
||||
continue
|
||||
fence = FENCE_RE.match(lines[index])
|
||||
if fence:
|
||||
marker = fence.group(1)[0]
|
||||
end = index + 1
|
||||
while end < len(lines) and not re.match(rf"^[ \t]*{re.escape(marker)}{{3,}}", lines[end]):
|
||||
end += 1
|
||||
end = min(end + 1, len(lines))
|
||||
add_block(index, end, "code")
|
||||
index = end
|
||||
continue
|
||||
heading = HEADING_RE.match(lines[index].rstrip("\r\n"))
|
||||
if heading:
|
||||
add_block(index, index + 1, "heading", len(heading.group(1)))
|
||||
index += 1
|
||||
continue
|
||||
if lines[index].lstrip().startswith("<"):
|
||||
add_block(index, index + 1, "html")
|
||||
index += 1
|
||||
continue
|
||||
if TABLE_RE.match(lines[index]):
|
||||
end = index + 1
|
||||
while end < len(lines) and TABLE_RE.match(lines[end]):
|
||||
end += 1
|
||||
add_block(index, end, "table")
|
||||
index = end
|
||||
continue
|
||||
if LIST_RE.match(lines[index]):
|
||||
end = index + 1
|
||||
while (
|
||||
end < len(lines)
|
||||
and lines[end].strip()
|
||||
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
|
||||
and not LIST_RE.match(lines[end])
|
||||
and not FENCE_RE.match(lines[end])
|
||||
):
|
||||
end += 1
|
||||
add_block(index, end, "list_item")
|
||||
index = end
|
||||
continue
|
||||
end = index + 1
|
||||
while (
|
||||
end < len(lines)
|
||||
and lines[end].strip()
|
||||
and not HEADING_RE.match(lines[end].rstrip("\r\n"))
|
||||
and not LIST_RE.match(lines[end])
|
||||
and not TABLE_RE.match(lines[end])
|
||||
and not FENCE_RE.match(lines[end])
|
||||
):
|
||||
end += 1
|
||||
add_block(index, end, "paragraph")
|
||||
index = end
|
||||
return SkillDocument(content, frontmatter, body, blocks, newline)
|
||||
|
||||
|
||||
def section_key(title: str | None) -> str | None:
|
||||
if not title:
|
||||
return None
|
||||
normalized = " ".join(title.lower().strip().rstrip("::-–—").split())
|
||||
for key, aliases in SECTION_ALIASES.items():
|
||||
if normalized in aliases:
|
||||
return key
|
||||
return None
|
||||
|
||||
|
||||
def _sentences(text: str) -> Iterable[str]:
|
||||
prefix = ""
|
||||
list_match = LIST_RE.match(text)
|
||||
content = text
|
||||
if list_match:
|
||||
prefix = text[: list_match.end()]
|
||||
content = text[list_match.end() :]
|
||||
parts = re.split(r"(?<=[。!?.!?;;])(?:[ \t]+|\r?\n+)", content)
|
||||
for index, part in enumerate(parts):
|
||||
clean = part.strip()
|
||||
if clean:
|
||||
yield (prefix if index == 0 else "") + clean
|
||||
|
||||
|
||||
def static_annotations(document: SkillDocument) -> list[Annotation]:
|
||||
annotations: list[Annotation] = []
|
||||
for block in document.blocks:
|
||||
if block.kind in {"code", "code_like", "html", "heading", "table"}:
|
||||
continue
|
||||
parent_key = section_key(block.parent_heading)
|
||||
candidates = list(_sentences(block.text))
|
||||
for quote in candidates:
|
||||
found: list[str] = []
|
||||
if parent_key == "definitions":
|
||||
found.append("definition")
|
||||
elif parent_key == "completion_criterion":
|
||||
found.append("completion_criterion")
|
||||
elif parent_key == "evidence_priority":
|
||||
found.append("evidence_priority_rule")
|
||||
elif parent_key == "scope":
|
||||
found.append("scope_rule")
|
||||
elif parent_key == "decision_criteria":
|
||||
found.append("decision_criterion")
|
||||
elif parent_key == "uncertainty_rule":
|
||||
found.append("uncertainty_rule")
|
||||
elif parent_key == "critical_rules":
|
||||
found.append("critical_rule")
|
||||
for annotation_type, pattern in ANNOTATION_PATTERNS:
|
||||
if pattern.search(quote):
|
||||
found.append(annotation_type)
|
||||
if VIEWPOINT_A_RE.search(quote):
|
||||
found.append("viewpoint_side_a")
|
||||
if VIEWPOINT_B_RE.search(quote):
|
||||
found.append("viewpoint_side_b")
|
||||
for annotation_type in dict.fromkeys(found):
|
||||
annotations.append(
|
||||
Annotation(annotation_type, block.id, quote, 1.0, "static")
|
||||
)
|
||||
return resolve_annotation_conflicts(annotations)
|
||||
|
||||
|
||||
ANNOTATION_PRIORITY = {
|
||||
"completion_criterion": 100,
|
||||
"evidence_priority_rule": 90,
|
||||
"uncertainty_rule": 80,
|
||||
"scope_rule": 70,
|
||||
"decision_criterion": 60,
|
||||
"definition": 50,
|
||||
"viewpoint_side_a": 40,
|
||||
"viewpoint_side_b": 40,
|
||||
"critical_rule": 10,
|
||||
"coreference": 5,
|
||||
}
|
||||
|
||||
|
||||
def resolve_annotation_conflicts(annotations: list[Annotation]) -> list[Annotation]:
|
||||
grouped: dict[tuple[str, str], list[Annotation]] = defaultdict(list)
|
||||
for annotation in annotations:
|
||||
grouped[(annotation.block_id, annotation.quote)].append(annotation)
|
||||
resolved: list[Annotation] = []
|
||||
for values in grouped.values():
|
||||
values.sort(
|
||||
key=lambda item: (
|
||||
ANNOTATION_PRIORITY.get(item.type, 0),
|
||||
item.confidence,
|
||||
item.source == "static",
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
resolved.append(values[0])
|
||||
return sorted(resolved, key=lambda item: (item.block_id, item.quote))
|
||||
|
||||
|
||||
def skill_name(document: SkillDocument) -> str:
|
||||
yaml_text = re.sub(
|
||||
r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", document.frontmatter
|
||||
)
|
||||
loaded = yaml.safe_load(yaml_text)
|
||||
name = loaded.get("name") if isinstance(loaded, dict) else None
|
||||
if not isinstance(name, str) or not name.strip():
|
||||
raise DocumentError("frontmatter requires non-empty name")
|
||||
return name.strip()
|
||||
Reference in New Issue
Block a user