"""Markdown 单元解析与编辑边界校验。""" from __future__ import annotations from collections import Counter import re from .models import SkillUnit _HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)") _FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})") def _line_offsets(text: str) -> list[tuple[int, int, str]]: rows: list[tuple[int, int, str]] = [] offset = 0 for line in text.splitlines(keepends=True): rows.append((offset, offset + len(line), line)) offset += len(line) return rows def _headings(text: str) -> list[tuple[int, int, int, str]]: found = [] fence_char = "" fence_size = 0 for start, end, line in _line_offsets(text): fence = _FENCE.match(line) if fence: marker = fence.group(1) if not fence_char: fence_char, fence_size = marker[0], len(marker) elif marker[0] == fence_char and len(marker) >= fence_size: fence_char, fence_size = "", 0 continue if fence_char: continue match = _HEADING.match(line) if match: found.append((start, end, len(match.group(1)), match.group(2).strip())) return found def _frontmatter_end(text: str) -> int: if not text.startswith("---"): return 0 lines = text.splitlines(keepends=True) offset = len(lines[0]) if lines else 0 for line in lines[1:]: offset += len(line) if line.strip() == "---": return offset return 0 def parse_sections(text: str) -> list[SkillUnit]: headings = _headings(text) body_start = _frontmatter_end(text) body_headings = [item for item in headings if item[0] >= body_start] title = body_headings[0] if body_headings else None after_title = title[1] if title else body_start candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]] if not candidates: body = text[body_start:] return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)] section_depth = min(item[2] for item in candidates) peers = [item for item in candidates if item[2] == section_depth] spans: list[tuple[int, int, str, int | None]] = [] preamble = text[after_title:peers[0][0]] if preamble.strip(): spans.append((after_title, peers[0][0], "Preamble", section_depth)) for index, heading in enumerate(peers): end = peers[index + 1][0] if index + 1 < len(peers) else len(text) spans.append((heading[0], end, heading[3], heading[2])) return [ SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth) for index, (start, end, heading, depth) in enumerate(spans) ] def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str: return new_text.rstrip() + unit.text[len(unit.text.rstrip()):] def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str: replacement = preserve_unit_boundary(unit, new_text) return document[:unit.start] + replacement + document[unit.end:] def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]: text = section.text base = section.start rows = _line_offsets(text) blocks: list[tuple[int, int]] = [] start: int | None = None fence_char = "" fence_size = 0 for row_start, row_end, line in rows: fence = _FENCE.match(line) if fence: marker = fence.group(1) if start is None: start = row_start if not fence_char: fence_char, fence_size = marker[0], len(marker) elif marker[0] == fence_char and len(marker) >= fence_size: fence_char, fence_size = "", 0 continue if not fence_char and not line.strip(): if start is not None: blocks.append((start, row_start)) start = None continue if start is None: start = row_start if start is not None: blocks.append((start, len(text))) merged: list[tuple[int, int]] = [] index = 0 while index < len(blocks): start, end = blocks[index] block = text[start:end] if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"): merged.append((start, blocks[index + 1][1])) index += 2 else: merged.append((start, end)) index += 1 blocks = merged units = [] for index, (start, end) in enumerate(blocks): block = text[start:end] heading_match = next((item for item in _headings(block)), None) units.append(SkillUnit( f"{section.unit_id}.P{index + 1:03d}", "paragraph", section.unit_id, heading_match[3] if heading_match else "", block, index, base + start, base + end, heading_match[2] if heading_match else None, )) return units def fenced_blocks(text: str) -> Counter[str]: blocks: list[str] = [] current: list[str] | None = None fence_char = "" fence_size = 0 for line in text.splitlines(keepends=True): fence = _FENCE.match(line) if current is None: if fence: marker = fence.group(1) fence_char, fence_size = marker[0], len(marker) current = [line] continue current.append(line) if fence: marker = fence.group(1) if marker[0] == fence_char and len(marker) >= fence_size: blocks.append("".join(current)) current = None fence_char, fence_size = "", 0 return Counter(blocks) def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None: if not new_text.strip() or new_text == unit.text: raise ValueError("local edit must produce non-empty changed text") if unit.level == "section": old_headings = _headings(unit.text) new_headings = _headings(new_text) depth = unit.heading_depth old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth] new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth] if old_peers != new_peers: raise ValueError("section edit must preserve its peer heading") if fenced_blocks(unit.text) != fenced_blocks(new_text): raise ValueError("section edit must preserve fenced code contents") elif section_depth is not None: old_peers = [ (item[2], item[3]) for item in _headings(unit.text) if item[2] <= section_depth ] new_peers = [ (item[2], item[3]) for item in _headings(new_text) if item[2] <= section_depth ] if old_peers != new_peers: raise ValueError("paragraph edit must not add or change a section heading")