"""Source-preserving Markdown block analysis and static annotations.""" from __future__ import annotations from collections import defaultdict from dataclasses import dataclass import re from typing import Iterable import yaml from markdown_it import MarkdownIt from .models import Annotation, BodyBlock class DocumentError(RuntimeError): """The Skill Markdown cannot be parsed safely.""" HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*$") FENCE_RE = re.compile(r"^[ \t]*(```+|~~~+)") LIST_RE = re.compile(r"^([ \t]*)(?:[-+*]|\d+[.)])[ \t]+") TABLE_RE = re.compile(r"^[ \t]*\|.*\|[ \t]*(?:\r?\n)?$") INLINE_PROTECTED_RE = re.compile( r"`[^`\n]+`|https?://[^\s)>]+|(? dict[str, BodyBlock]: return {block.id: block for block in self.blocks} @property def language(self) -> str: nonspace = [char for char in self.body if not char.isspace()] if not nonspace: return "en" cjk = sum("\u4e00" <= char <= "\u9fff" for char in nonspace) return "zh" if cjk / len(nonspace) >= 0.30 else "en" def split_frontmatter(content: str) -> tuple[str, str]: if not content.startswith("---"): raise DocumentError("SKILL.md requires YAML frontmatter") match = re.search(r"\A---[ \t]*\r?\n.*?\r?\n---[ \t]*(?:\r?\n|\Z)", content, re.S) if not match: raise DocumentError("unterminated YAML frontmatter") frontmatter = match.group(0) yaml_text = re.sub(r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", frontmatter) try: loaded = yaml.safe_load(yaml_text) except yaml.YAMLError as exc: raise DocumentError(f"invalid YAML frontmatter: {exc}") from exc if not isinstance(loaded, dict): raise DocumentError("YAML frontmatter must be a mapping") return frontmatter, content[match.end() :] def _protected_spans(text: str, *, whole_block: bool = False) -> list[tuple[int, int]]: if whole_block: return [(0, len(text))] return [(match.start(), match.end()) for match in INLINE_PROTECTED_RE.finditer(text)] def _looks_like_code(text: str) -> bool: lines = [line for line in text.splitlines() if line.strip()] if len(lines) < 2: return False code_line = re.compile( r"^[ \t]{2,}(?:def |class |if |elif |else:|for |while |return |" r"print\(|raise |try:|except |[A-Za-z_][A-Za-z0-9_]*\s*=|[}\]])" ) signals = sum(bool(code_line.match(line)) for line in lines) return signals >= 2 and signals >= len(lines) / 2 def _line_offsets(body: str) -> tuple[list[str], list[int]]: lines = body.splitlines(keepends=True) if body and not lines: lines = [body] offsets: list[int] = [] position = 0 for line in lines: offsets.append(position) position += len(line) return lines, offsets def parse_document(content: str) -> SkillDocument: frontmatter, body = split_frontmatter(content) newline = "\r\n" if "\r\n" in content else "\n" MarkdownIt("commonmark", {"html": True}).parse(body) lines, offsets = _line_offsets(body) blocks: list[BodyBlock] = [] index = 0 parent_heading: str | None = None block_number = 0 def add_block(start: int, end: int, kind: str, heading_level: int | None = None) -> None: nonlocal block_number, parent_heading raw = "".join(lines[start:end]).rstrip("\r\n") if not raw: return if kind in {"paragraph", "list_item"} and _looks_like_code(raw): kind = "code_like" block_number += 1 block_id = f"B{block_number:03d}" start_offset = offsets[start] end_offset = start_offset + len("".join(lines[start:end])) list_match = LIST_RE.match(raw) block = BodyBlock( id=block_id, kind=kind, text=raw, start_line=start + 1, end_line=end, start_offset=start_offset, end_offset=end_offset, parent_heading=parent_heading, heading_level=heading_level, list_depth=(len(list_match.group(1).replace("\t", " ")) // 2 if list_match else 0), protected_spans=_protected_spans( raw, whole_block=kind in {"code", "code_like", "html", "table"} ), ) blocks.append(block) if kind == "heading": heading = HEADING_RE.match(raw) parent_heading = heading.group(2).strip() if heading else raw while index < len(lines): stripped = lines[index].strip() if not stripped: index += 1 continue fence = FENCE_RE.match(lines[index]) if fence: marker = fence.group(1)[0] end = index + 1 while end < len(lines) and not re.match(rf"^[ \t]*{re.escape(marker)}{{3,}}", lines[end]): end += 1 end = min(end + 1, len(lines)) add_block(index, end, "code") index = end continue heading = HEADING_RE.match(lines[index].rstrip("\r\n")) if heading: add_block(index, index + 1, "heading", len(heading.group(1))) index += 1 continue if lines[index].lstrip().startswith("<"): add_block(index, index + 1, "html") index += 1 continue if TABLE_RE.match(lines[index]): end = index + 1 while end < len(lines) and TABLE_RE.match(lines[end]): end += 1 add_block(index, end, "table") index = end continue if LIST_RE.match(lines[index]): end = index + 1 while ( end < len(lines) and lines[end].strip() and not HEADING_RE.match(lines[end].rstrip("\r\n")) and not LIST_RE.match(lines[end]) and not FENCE_RE.match(lines[end]) ): end += 1 add_block(index, end, "list_item") index = end continue end = index + 1 while ( end < len(lines) and lines[end].strip() and not HEADING_RE.match(lines[end].rstrip("\r\n")) and not LIST_RE.match(lines[end]) and not TABLE_RE.match(lines[end]) and not FENCE_RE.match(lines[end]) ): end += 1 add_block(index, end, "paragraph") index = end return SkillDocument(content, frontmatter, body, blocks, newline) def section_key(title: str | None) -> str | None: if not title: return None normalized = " ".join(title.lower().strip().rstrip("::-–—").split()) for key, aliases in SECTION_ALIASES.items(): if normalized in aliases: return key return None def _sentences(text: str) -> Iterable[str]: prefix = "" list_match = LIST_RE.match(text) content = text if list_match: prefix = text[: list_match.end()] content = text[list_match.end() :] parts = re.split(r"(?<=[。!?.!?;;])(?:[ \t]+|\r?\n+)", content) for index, part in enumerate(parts): clean = part.strip() if clean: yield (prefix if index == 0 else "") + clean def static_annotations(document: SkillDocument) -> list[Annotation]: annotations: list[Annotation] = [] for block in document.blocks: if block.kind in {"code", "code_like", "html", "heading", "table"}: continue parent_key = section_key(block.parent_heading) candidates = list(_sentences(block.text)) for quote in candidates: found: list[str] = [] if parent_key == "definitions": found.append("definition") elif parent_key == "completion_criterion": found.append("completion_criterion") elif parent_key == "evidence_priority": found.append("evidence_priority_rule") elif parent_key == "scope": found.append("scope_rule") elif parent_key == "decision_criteria": found.append("decision_criterion") elif parent_key == "uncertainty_rule": found.append("uncertainty_rule") elif parent_key == "critical_rules": found.append("critical_rule") for annotation_type, pattern in ANNOTATION_PATTERNS: if pattern.search(quote): found.append(annotation_type) if VIEWPOINT_A_RE.search(quote): found.append("viewpoint_side_a") if VIEWPOINT_B_RE.search(quote): found.append("viewpoint_side_b") for annotation_type in dict.fromkeys(found): annotations.append( Annotation(annotation_type, block.id, quote, 1.0, "static") ) return resolve_annotation_conflicts(annotations) ANNOTATION_PRIORITY = { "completion_criterion": 100, "evidence_priority_rule": 90, "uncertainty_rule": 80, "scope_rule": 70, "decision_criterion": 60, "definition": 50, "viewpoint_side_a": 40, "viewpoint_side_b": 40, "critical_rule": 10, "coreference": 5, } def resolve_annotation_conflicts(annotations: list[Annotation]) -> list[Annotation]: grouped: dict[tuple[str, str], list[Annotation]] = defaultdict(list) for annotation in annotations: grouped[(annotation.block_id, annotation.quote)].append(annotation) resolved: list[Annotation] = [] for values in grouped.values(): values.sort( key=lambda item: ( ANNOTATION_PRIORITY.get(item.type, 0), item.confidence, item.source == "static", ), reverse=True, ) resolved.append(values[0]) return sorted(resolved, key=lambda item: (item.block_id, item.quote)) def skill_name(document: SkillDocument) -> str: yaml_text = re.sub( r"\A---[ \t]*\r?\n|\r?\n---[ \t]*(?:\r?\n)?\Z", "", document.frontmatter ) loaded = yaml.safe_load(yaml_text) name = loaded.get("name") if isinstance(loaded, dict) else None if not isinstance(name, str) or not name.strip(): raise DocumentError("frontmatter requires non-empty name") return name.strip()