Initial commit
This commit is contained in:
@@ -0,0 +1,195 @@
|
||||
"""Markdown 单元解析与编辑边界校验。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
import re
|
||||
|
||||
from .models import SkillUnit
|
||||
|
||||
|
||||
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
|
||||
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
|
||||
|
||||
|
||||
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
|
||||
rows: list[tuple[int, int, str]] = []
|
||||
offset = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
rows.append((offset, offset + len(line), line))
|
||||
offset += len(line)
|
||||
return rows
|
||||
|
||||
|
||||
def _headings(text: str) -> list[tuple[int, int, int, str]]:
|
||||
found = []
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for start, end, line in _line_offsets(text):
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if fence_char:
|
||||
continue
|
||||
match = _HEADING.match(line)
|
||||
if match:
|
||||
found.append((start, end, len(match.group(1)), match.group(2).strip()))
|
||||
return found
|
||||
|
||||
|
||||
def _frontmatter_end(text: str) -> int:
|
||||
if not text.startswith("---"):
|
||||
return 0
|
||||
lines = text.splitlines(keepends=True)
|
||||
offset = len(lines[0]) if lines else 0
|
||||
for line in lines[1:]:
|
||||
offset += len(line)
|
||||
if line.strip() == "---":
|
||||
return offset
|
||||
return 0
|
||||
|
||||
|
||||
def parse_sections(text: str) -> list[SkillUnit]:
|
||||
headings = _headings(text)
|
||||
body_start = _frontmatter_end(text)
|
||||
body_headings = [item for item in headings if item[0] >= body_start]
|
||||
title = body_headings[0] if body_headings else None
|
||||
after_title = title[1] if title else body_start
|
||||
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
|
||||
if not candidates:
|
||||
body = text[body_start:]
|
||||
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
|
||||
section_depth = min(item[2] for item in candidates)
|
||||
peers = [item for item in candidates if item[2] == section_depth]
|
||||
spans: list[tuple[int, int, str, int | None]] = []
|
||||
preamble = text[after_title:peers[0][0]]
|
||||
if preamble.strip():
|
||||
spans.append((after_title, peers[0][0], "Preamble", section_depth))
|
||||
for index, heading in enumerate(peers):
|
||||
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
|
||||
spans.append((heading[0], end, heading[3], heading[2]))
|
||||
return [
|
||||
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
|
||||
for index, (start, end, heading, depth) in enumerate(spans)
|
||||
]
|
||||
|
||||
|
||||
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
|
||||
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
|
||||
|
||||
|
||||
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
|
||||
replacement = preserve_unit_boundary(unit, new_text)
|
||||
return document[:unit.start] + replacement + document[unit.end:]
|
||||
|
||||
|
||||
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
|
||||
text = section.text
|
||||
base = section.start
|
||||
rows = _line_offsets(text)
|
||||
blocks: list[tuple[int, int]] = []
|
||||
start: int | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for row_start, row_end, line in rows:
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if start is None:
|
||||
start = row_start
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if not fence_char and not line.strip():
|
||||
if start is not None:
|
||||
blocks.append((start, row_start))
|
||||
start = None
|
||||
continue
|
||||
if start is None:
|
||||
start = row_start
|
||||
if start is not None:
|
||||
blocks.append((start, len(text)))
|
||||
merged: list[tuple[int, int]] = []
|
||||
index = 0
|
||||
while index < len(blocks):
|
||||
start, end = blocks[index]
|
||||
block = text[start:end]
|
||||
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
|
||||
merged.append((start, blocks[index + 1][1]))
|
||||
index += 2
|
||||
else:
|
||||
merged.append((start, end))
|
||||
index += 1
|
||||
blocks = merged
|
||||
units = []
|
||||
for index, (start, end) in enumerate(blocks):
|
||||
block = text[start:end]
|
||||
heading_match = next((item for item in _headings(block)), None)
|
||||
units.append(SkillUnit(
|
||||
f"{section.unit_id}.P{index + 1:03d}",
|
||||
"paragraph",
|
||||
section.unit_id,
|
||||
heading_match[3] if heading_match else "",
|
||||
block,
|
||||
index,
|
||||
base + start,
|
||||
base + end,
|
||||
heading_match[2] if heading_match else None,
|
||||
))
|
||||
return units
|
||||
|
||||
|
||||
def fenced_blocks(text: str) -> Counter[str]:
|
||||
blocks: list[str] = []
|
||||
current: list[str] | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
fence = _FENCE.match(line)
|
||||
if current is None:
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
current = [line]
|
||||
continue
|
||||
current.append(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if marker[0] == fence_char and len(marker) >= fence_size:
|
||||
blocks.append("".join(current))
|
||||
current = None
|
||||
fence_char, fence_size = "", 0
|
||||
return Counter(blocks)
|
||||
|
||||
|
||||
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
|
||||
if not new_text.strip() or new_text == unit.text:
|
||||
raise ValueError("local edit must produce non-empty changed text")
|
||||
if unit.level == "section":
|
||||
old_headings = _headings(unit.text)
|
||||
new_headings = _headings(new_text)
|
||||
depth = unit.heading_depth
|
||||
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
|
||||
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("section edit must preserve its peer heading")
|
||||
if fenced_blocks(unit.text) != fenced_blocks(new_text):
|
||||
raise ValueError("section edit must preserve fenced code contents")
|
||||
elif section_depth is not None:
|
||||
old_peers = [
|
||||
(item[2], item[3]) for item in _headings(unit.text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
new_peers = [
|
||||
(item[2], item[3]) for item in _headings(new_text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("paragraph edit must not add or change a section heading")
|
||||
@@ -0,0 +1,134 @@
|
||||
"""领域模型。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
DIMENSIONS = (
|
||||
"Clarity",
|
||||
"Structure",
|
||||
"Executability",
|
||||
"Completeness",
|
||||
"Constraint Salience",
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkillUnit:
|
||||
unit_id: str
|
||||
level: str
|
||||
parent_id: str | None
|
||||
heading: str
|
||||
text: str
|
||||
order: int
|
||||
start: int
|
||||
end: int
|
||||
heading_depth: int | None = None
|
||||
|
||||
@dataclass
|
||||
class CellScore:
|
||||
score: float
|
||||
evidence: list[str]
|
||||
reason: str
|
||||
|
||||
@dataclass
|
||||
class Coordinate:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
normalized_gap: float
|
||||
|
||||
@dataclass
|
||||
class LocalEdit:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
new_text: str
|
||||
edit_summary: str
|
||||
reason: str
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "LocalEdit":
|
||||
required = {"unit_id", "dimension", "new_text", "edit_summary", "reason"}
|
||||
missing = sorted(required - value.keys())
|
||||
if missing:
|
||||
raise ValueError(f"local edit missing fields: {', '.join(missing)}")
|
||||
if not all(isinstance(value[key], str) for key in required):
|
||||
raise ValueError("local edit fields must be strings")
|
||||
return cls(**{key: value[key] for key in cls.__dataclass_fields__})
|
||||
|
||||
|
||||
@dataclass
|
||||
class ScoreMatrix:
|
||||
level: str
|
||||
units: list[SkillUnit]
|
||||
columns: dict[str, dict[str, CellScore]]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"level": self.level,
|
||||
"units": [asdict(unit) for unit in self.units],
|
||||
"columns": {
|
||||
dimension: {unit_id: asdict(cell) for unit_id, cell in column.items()}
|
||||
for dimension, column in self.columns.items()
|
||||
},
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "ScoreMatrix":
|
||||
return cls(
|
||||
level=str(value["level"]),
|
||||
units=[SkillUnit(**item) for item in value["units"]],
|
||||
columns={
|
||||
dimension: {
|
||||
unit_id: CellScore(float(cell["score"]), list(cell["evidence"]), str(cell["reason"]))
|
||||
for unit_id, cell in column.items()
|
||||
}
|
||||
for dimension, column in value["columns"].items()
|
||||
},
|
||||
)
|
||||
|
||||
def unit(self, unit_id: str) -> SkillUnit:
|
||||
return next(unit for unit in self.units if unit.unit_id == unit_id)
|
||||
|
||||
def normalized_gaps(self) -> dict[str, float]:
|
||||
gaps: dict[str, float] = {}
|
||||
for dimension in DIMENSIONS:
|
||||
values = [self.columns[dimension][unit.unit_id].score for unit in self.units]
|
||||
gaps[dimension] = (max(values) - min(values)) / 4.0 if values else 0.0
|
||||
return gaps
|
||||
|
||||
def select_coordinate(
|
||||
self,
|
||||
threshold: float,
|
||||
dimension: str | None = None,
|
||||
excluded: set[tuple[str, str]] | None = None,
|
||||
) -> Coordinate | None:
|
||||
gaps = self.normalized_gaps()
|
||||
excluded = excluded or set()
|
||||
|
||||
def weak_units(item: str) -> list[SkillUnit]:
|
||||
column = self.columns[item]
|
||||
maximum = max((cell.score for cell in column.values()), default=0.0)
|
||||
return [
|
||||
unit for unit in self.units
|
||||
if (unit.unit_id, item) not in excluded
|
||||
and (maximum - column[unit.unit_id].score) / 4.0 > threshold
|
||||
]
|
||||
|
||||
available = [
|
||||
item for item in ([dimension] if dimension else DIMENSIONS)
|
||||
if item is not None and gaps[item] > threshold and weak_units(item)
|
||||
]
|
||||
if not available:
|
||||
return None
|
||||
dimension = max(available, key=lambda item: gaps[item])
|
||||
target = min(
|
||||
weak_units(dimension),
|
||||
key=lambda unit: (self.columns[dimension][unit.unit_id].score, unit.order),
|
||||
)
|
||||
return Coordinate(
|
||||
target.unit_id,
|
||||
dimension,
|
||||
gaps[dimension],
|
||||
)
|
||||
Reference in New Issue
Block a user