196 lines
6.9 KiB
Python
196 lines
6.9 KiB
Python
"""Markdown 单元解析与编辑边界校验。"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections import Counter
|
|
import re
|
|
|
|
from .models import SkillUnit
|
|
|
|
|
|
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
|
|
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
|
|
|
|
|
|
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
|
|
rows: list[tuple[int, int, str]] = []
|
|
offset = 0
|
|
for line in text.splitlines(keepends=True):
|
|
rows.append((offset, offset + len(line), line))
|
|
offset += len(line)
|
|
return rows
|
|
|
|
|
|
def _headings(text: str) -> list[tuple[int, int, int, str]]:
|
|
found = []
|
|
fence_char = ""
|
|
fence_size = 0
|
|
for start, end, line in _line_offsets(text):
|
|
fence = _FENCE.match(line)
|
|
if fence:
|
|
marker = fence.group(1)
|
|
if not fence_char:
|
|
fence_char, fence_size = marker[0], len(marker)
|
|
elif marker[0] == fence_char and len(marker) >= fence_size:
|
|
fence_char, fence_size = "", 0
|
|
continue
|
|
if fence_char:
|
|
continue
|
|
match = _HEADING.match(line)
|
|
if match:
|
|
found.append((start, end, len(match.group(1)), match.group(2).strip()))
|
|
return found
|
|
|
|
|
|
def _frontmatter_end(text: str) -> int:
|
|
if not text.startswith("---"):
|
|
return 0
|
|
lines = text.splitlines(keepends=True)
|
|
offset = len(lines[0]) if lines else 0
|
|
for line in lines[1:]:
|
|
offset += len(line)
|
|
if line.strip() == "---":
|
|
return offset
|
|
return 0
|
|
|
|
|
|
def parse_sections(text: str) -> list[SkillUnit]:
|
|
headings = _headings(text)
|
|
body_start = _frontmatter_end(text)
|
|
body_headings = [item for item in headings if item[0] >= body_start]
|
|
title = body_headings[0] if body_headings else None
|
|
after_title = title[1] if title else body_start
|
|
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
|
|
if not candidates:
|
|
body = text[body_start:]
|
|
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
|
|
section_depth = min(item[2] for item in candidates)
|
|
peers = [item for item in candidates if item[2] == section_depth]
|
|
spans: list[tuple[int, int, str, int | None]] = []
|
|
preamble = text[after_title:peers[0][0]]
|
|
if preamble.strip():
|
|
spans.append((after_title, peers[0][0], "Preamble", section_depth))
|
|
for index, heading in enumerate(peers):
|
|
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
|
|
spans.append((heading[0], end, heading[3], heading[2]))
|
|
return [
|
|
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
|
|
for index, (start, end, heading, depth) in enumerate(spans)
|
|
]
|
|
|
|
|
|
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
|
|
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
|
|
|
|
|
|
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
|
|
replacement = preserve_unit_boundary(unit, new_text)
|
|
return document[:unit.start] + replacement + document[unit.end:]
|
|
|
|
|
|
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
|
|
text = section.text
|
|
base = section.start
|
|
rows = _line_offsets(text)
|
|
blocks: list[tuple[int, int]] = []
|
|
start: int | None = None
|
|
fence_char = ""
|
|
fence_size = 0
|
|
for row_start, row_end, line in rows:
|
|
fence = _FENCE.match(line)
|
|
if fence:
|
|
marker = fence.group(1)
|
|
if start is None:
|
|
start = row_start
|
|
if not fence_char:
|
|
fence_char, fence_size = marker[0], len(marker)
|
|
elif marker[0] == fence_char and len(marker) >= fence_size:
|
|
fence_char, fence_size = "", 0
|
|
continue
|
|
if not fence_char and not line.strip():
|
|
if start is not None:
|
|
blocks.append((start, row_start))
|
|
start = None
|
|
continue
|
|
if start is None:
|
|
start = row_start
|
|
if start is not None:
|
|
blocks.append((start, len(text)))
|
|
merged: list[tuple[int, int]] = []
|
|
index = 0
|
|
while index < len(blocks):
|
|
start, end = blocks[index]
|
|
block = text[start:end]
|
|
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
|
|
merged.append((start, blocks[index + 1][1]))
|
|
index += 2
|
|
else:
|
|
merged.append((start, end))
|
|
index += 1
|
|
blocks = merged
|
|
units = []
|
|
for index, (start, end) in enumerate(blocks):
|
|
block = text[start:end]
|
|
heading_match = next((item for item in _headings(block)), None)
|
|
units.append(SkillUnit(
|
|
f"{section.unit_id}.P{index + 1:03d}",
|
|
"paragraph",
|
|
section.unit_id,
|
|
heading_match[3] if heading_match else "",
|
|
block,
|
|
index,
|
|
base + start,
|
|
base + end,
|
|
heading_match[2] if heading_match else None,
|
|
))
|
|
return units
|
|
|
|
|
|
def fenced_blocks(text: str) -> Counter[str]:
|
|
blocks: list[str] = []
|
|
current: list[str] | None = None
|
|
fence_char = ""
|
|
fence_size = 0
|
|
for line in text.splitlines(keepends=True):
|
|
fence = _FENCE.match(line)
|
|
if current is None:
|
|
if fence:
|
|
marker = fence.group(1)
|
|
fence_char, fence_size = marker[0], len(marker)
|
|
current = [line]
|
|
continue
|
|
current.append(line)
|
|
if fence:
|
|
marker = fence.group(1)
|
|
if marker[0] == fence_char and len(marker) >= fence_size:
|
|
blocks.append("".join(current))
|
|
current = None
|
|
fence_char, fence_size = "", 0
|
|
return Counter(blocks)
|
|
|
|
|
|
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
|
|
if not new_text.strip() or new_text == unit.text:
|
|
raise ValueError("local edit must produce non-empty changed text")
|
|
if unit.level == "section":
|
|
old_headings = _headings(unit.text)
|
|
new_headings = _headings(new_text)
|
|
depth = unit.heading_depth
|
|
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
|
|
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
|
|
if old_peers != new_peers:
|
|
raise ValueError("section edit must preserve its peer heading")
|
|
if fenced_blocks(unit.text) != fenced_blocks(new_text):
|
|
raise ValueError("section edit must preserve fenced code contents")
|
|
elif section_depth is not None:
|
|
old_peers = [
|
|
(item[2], item[3]) for item in _headings(unit.text)
|
|
if item[2] <= section_depth
|
|
]
|
|
new_peers = [
|
|
(item[2], item[3]) for item in _headings(new_text)
|
|
if item[2] <= section_depth
|
|
]
|
|
if old_peers != new_peers:
|
|
raise ValueError("paragraph edit must not add or change a section heading")
|