Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
@@ -0,0 +1,195 @@
"""Markdown 单元解析与编辑边界校验。"""
from __future__ import annotations
from collections import Counter
import re
from .models import SkillUnit
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
rows: list[tuple[int, int, str]] = []
offset = 0
for line in text.splitlines(keepends=True):
rows.append((offset, offset + len(line), line))
offset += len(line)
return rows
def _headings(text: str) -> list[tuple[int, int, int, str]]:
found = []
fence_char = ""
fence_size = 0
for start, end, line in _line_offsets(text):
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if fence_char:
continue
match = _HEADING.match(line)
if match:
found.append((start, end, len(match.group(1)), match.group(2).strip()))
return found
def _frontmatter_end(text: str) -> int:
if not text.startswith("---"):
return 0
lines = text.splitlines(keepends=True)
offset = len(lines[0]) if lines else 0
for line in lines[1:]:
offset += len(line)
if line.strip() == "---":
return offset
return 0
def parse_sections(text: str) -> list[SkillUnit]:
headings = _headings(text)
body_start = _frontmatter_end(text)
body_headings = [item for item in headings if item[0] >= body_start]
title = body_headings[0] if body_headings else None
after_title = title[1] if title else body_start
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
if not candidates:
body = text[body_start:]
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
section_depth = min(item[2] for item in candidates)
peers = [item for item in candidates if item[2] == section_depth]
spans: list[tuple[int, int, str, int | None]] = []
preamble = text[after_title:peers[0][0]]
if preamble.strip():
spans.append((after_title, peers[0][0], "Preamble", section_depth))
for index, heading in enumerate(peers):
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
spans.append((heading[0], end, heading[3], heading[2]))
return [
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
for index, (start, end, heading, depth) in enumerate(spans)
]
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
replacement = preserve_unit_boundary(unit, new_text)
return document[:unit.start] + replacement + document[unit.end:]
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
text = section.text
base = section.start
rows = _line_offsets(text)
blocks: list[tuple[int, int]] = []
start: int | None = None
fence_char = ""
fence_size = 0
for row_start, row_end, line in rows:
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if start is None:
start = row_start
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if not fence_char and not line.strip():
if start is not None:
blocks.append((start, row_start))
start = None
continue
if start is None:
start = row_start
if start is not None:
blocks.append((start, len(text)))
merged: list[tuple[int, int]] = []
index = 0
while index < len(blocks):
start, end = blocks[index]
block = text[start:end]
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
merged.append((start, blocks[index + 1][1]))
index += 2
else:
merged.append((start, end))
index += 1
blocks = merged
units = []
for index, (start, end) in enumerate(blocks):
block = text[start:end]
heading_match = next((item for item in _headings(block)), None)
units.append(SkillUnit(
f"{section.unit_id}.P{index + 1:03d}",
"paragraph",
section.unit_id,
heading_match[3] if heading_match else "",
block,
index,
base + start,
base + end,
heading_match[2] if heading_match else None,
))
return units
def fenced_blocks(text: str) -> Counter[str]:
blocks: list[str] = []
current: list[str] | None = None
fence_char = ""
fence_size = 0
for line in text.splitlines(keepends=True):
fence = _FENCE.match(line)
if current is None:
if fence:
marker = fence.group(1)
fence_char, fence_size = marker[0], len(marker)
current = [line]
continue
current.append(line)
if fence:
marker = fence.group(1)
if marker[0] == fence_char and len(marker) >= fence_size:
blocks.append("".join(current))
current = None
fence_char, fence_size = "", 0
return Counter(blocks)
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
if not new_text.strip() or new_text == unit.text:
raise ValueError("local edit must produce non-empty changed text")
if unit.level == "section":
old_headings = _headings(unit.text)
new_headings = _headings(new_text)
depth = unit.heading_depth
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
if old_peers != new_peers:
raise ValueError("section edit must preserve its peer heading")
if fenced_blocks(unit.text) != fenced_blocks(new_text):
raise ValueError("section edit must preserve fenced code contents")
elif section_depth is not None:
old_peers = [
(item[2], item[3]) for item in _headings(unit.text)
if item[2] <= section_depth
]
new_peers = [
(item[2], item[3]) for item in _headings(new_text)
if item[2] <= section_depth
]
if old_peers != new_peers:
raise ValueError("paragraph edit must not add or change a section heading")