456 lines
15 KiB
Python
456 lines
15 KiB
Python
"""Deterministic, source-preserving rewrite planning and application."""
|
||
|
||
from __future__ import annotations
|
||
|
||
from collections import defaultdict
|
||
from dataclasses import dataclass, replace
|
||
import re
|
||
|
||
from .document import CANONICAL_HEADINGS, HEADING_RE, SkillDocument, section_key
|
||
from .models import Annotation, BodyBlock, Operation, Signal
|
||
|
||
|
||
ANNOTATION_SIGNAL = {
|
||
"critical_rule": "contextual_rule_adherence",
|
||
"definition": "contextual_rule_adherence",
|
||
"completion_criterion": "contextual_rule_adherence",
|
||
"evidence_priority_rule": "evidence_priority",
|
||
"scope_rule": "ambiguity_handling",
|
||
"decision_criterion": "ambiguity_handling",
|
||
"uncertainty_rule": "uncertainty_calibration",
|
||
"viewpoint_side_a": "balanced_presentation",
|
||
"viewpoint_side_b": "balanced_presentation",
|
||
"coreference": "semantic_robustness",
|
||
}
|
||
|
||
ANNOTATION_TARGET = {
|
||
"critical_rule": "critical_rules",
|
||
"definition": "definitions",
|
||
"completion_criterion": "completion_criterion",
|
||
"evidence_priority_rule": "evidence_priority",
|
||
"scope_rule": "scope",
|
||
"decision_criterion": "decision_criteria",
|
||
"uncertainty_rule": "uncertainty_rule",
|
||
"viewpoint_side_a": "viewpoint_side_a",
|
||
"viewpoint_side_b": "viewpoint_side_b",
|
||
}
|
||
|
||
PRE_SECTION_ORDER = (
|
||
"definitions",
|
||
"critical_rules",
|
||
"evidence_priority",
|
||
"scope",
|
||
"decision_criteria",
|
||
"uncertainty_rule",
|
||
"viewpoint_side_a",
|
||
"viewpoint_side_b",
|
||
)
|
||
|
||
LOCAL_EMPHASIS_TYPES = {
|
||
"critical_rule",
|
||
"completion_criterion",
|
||
"evidence_priority_rule",
|
||
}
|
||
|
||
PROMINENT_RULE_HEADING_RE = re.compile(
|
||
r"\b(?:critical|rules?|constraints?|requirements?|best\s+practices?|"
|
||
r"steps?|workflow|procedures?|strategy|priority|verification|validation|"
|
||
r"completion)\b|关键|规则|约束|要求|最佳实践|步骤|流程|策略|优先级|验证|完成",
|
||
re.I,
|
||
)
|
||
PROMINENT_RULE_LABEL_RE = re.compile(
|
||
r"^(?:(?:[-+*]|\d+[.)])\s+)?"
|
||
r"\*\*(?:important|critical|best\s+practice|requirement|rule|"
|
||
r"注意|重要|关键|规则|要求)\b",
|
||
re.I,
|
||
)
|
||
|
||
|
||
class RewriteError(RuntimeError):
|
||
"""A deterministic rewrite could not preserve its source span."""
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class _Patch:
|
||
start: int
|
||
end: int
|
||
replacement: str
|
||
|
||
|
||
def _enabled(signal: Signal, annotation: Annotation) -> bool:
|
||
return signal.level in {"low", "medium"}
|
||
|
||
|
||
def _heading_text(block: BodyBlock) -> str | None:
|
||
match = HEADING_RE.match(block.text)
|
||
return match.group(2).strip() if match else None
|
||
|
||
|
||
def _format_payload(quote: str) -> str:
|
||
stripped = quote.strip()
|
||
ordered = re.match(r"^\d+[.)]\s+(.+)$", stripped, re.S)
|
||
if ordered:
|
||
return f"- {ordered.group(1).strip()}"
|
||
if re.match(r"^[-+*]\s+", stripped):
|
||
return stripped
|
||
return f"- {stripped}"
|
||
|
||
|
||
def _section_heading(key: str, language: str) -> str:
|
||
if key == "viewpoint_side_a":
|
||
return "支持方" if language == "zh" else "Supporting View"
|
||
if key == "viewpoint_side_b":
|
||
return "反对方" if language == "zh" else "Opposing View"
|
||
return CANONICAL_HEADINGS[language][key]
|
||
|
||
|
||
def _normalize_known_heading(
|
||
block: BodyBlock, language: str
|
||
) -> tuple[str | None, Operation | None]:
|
||
title = _heading_text(block)
|
||
if title is None:
|
||
return None, None
|
||
key = section_key(title)
|
||
if key is None:
|
||
return None, None
|
||
canonical = CANONICAL_HEADINGS[language][key]
|
||
match = HEADING_RE.match(block.text)
|
||
assert match is not None
|
||
replacement = f"{match.group(1)} {canonical}"
|
||
if replacement == block.text:
|
||
return None, None
|
||
return (
|
||
replacement,
|
||
Operation(
|
||
type="RENAME_HEADING",
|
||
signal="semantic_robustness",
|
||
block_id=block.id,
|
||
quote=block.text,
|
||
replacement=replacement,
|
||
target_section=key,
|
||
),
|
||
)
|
||
|
||
|
||
def _split_explicit_requirements(
|
||
block: BodyBlock,
|
||
) -> tuple[str | None, Operation | None]:
|
||
if block.kind not in {"paragraph", "list_item"} or not re.search(
|
||
r"[;;]", block.text
|
||
):
|
||
return None, None
|
||
if not re.search(
|
||
r"\bMUST(?:\s+NOT)?\b|必须|不得|禁止|仅可|只能|不能", block.text, re.I
|
||
):
|
||
return None, None
|
||
parts = [part.strip() for part in re.split(r"[;;]", block.text) if part.strip()]
|
||
if len(parts) < 2:
|
||
return None, None
|
||
replacement = "\n".join(_format_payload(part) for part in parts)
|
||
return (
|
||
replacement,
|
||
Operation(
|
||
type="SPLIT_AT_EXISTING_DELIMITER",
|
||
signal="semantic_robustness",
|
||
block_id=block.id,
|
||
quote=block.text,
|
||
replacement=replacement,
|
||
),
|
||
)
|
||
|
||
|
||
def _is_standalone_rule(block: BodyBlock, quote: str) -> bool:
|
||
clean = quote.strip()
|
||
if block.kind not in {"paragraph", "list_item"}:
|
||
return False
|
||
if clean != block.text.strip():
|
||
return False
|
||
if clean.endswith((":", ":", "-", "—")):
|
||
return False
|
||
content = re.sub(r"^(?:[-+*]|\d+[.)])\s+", "", clean)
|
||
return len(content) >= 6
|
||
|
||
|
||
def _already_emphasized(block_text: str, quote_offset: int, quote: str) -> bool:
|
||
stripped = quote.strip()
|
||
stripped = re.sub(r"^(?:[-+*]|\d+[.)])\s+", "", stripped)
|
||
if stripped.startswith(("**", "__")) and stripped.endswith(("**", "__")):
|
||
return True
|
||
quote_end = quote_offset + len(quote)
|
||
for delimiter in ("**", "__"):
|
||
if (
|
||
block_text[max(0, quote_offset - len(delimiter)) : quote_offset]
|
||
== delimiter
|
||
and block_text[quote_end : quote_end + len(delimiter)] == delimiter
|
||
):
|
||
return True
|
||
return False
|
||
|
||
|
||
def _already_structurally_prominent(block: BodyBlock, quote: str) -> bool:
|
||
if PROMINENT_RULE_HEADING_RE.search(block.parent_heading or ""):
|
||
return True
|
||
return bool(PROMINENT_RULE_LABEL_RE.search(quote.strip()))
|
||
|
||
|
||
def _safe_to_emphasize(quote: str) -> bool:
|
||
# Wrapping a span that already contains emphasis creates ambiguous nested
|
||
# Markdown such as **use a **priority cascade**:**.
|
||
return not re.search(r"\*\*|__", quote)
|
||
|
||
|
||
def _emphasize(quote: str) -> str:
|
||
match = re.match(r"^((?:[-+*]|\d+[.)])\s+)(.+)$", quote.strip(), re.S)
|
||
if match:
|
||
return f"{match.group(1)}**{match.group(2)}**"
|
||
return f"**{quote.strip()}**"
|
||
|
||
|
||
def _apply_patches(body: str, patches: list[_Patch]) -> str:
|
||
ordered = sorted(patches, key=lambda item: (item.start, item.end), reverse=True)
|
||
last_start = len(body) + 1
|
||
result = body
|
||
for patch in ordered:
|
||
if patch.start < 0 or patch.end < patch.start or patch.end > len(body):
|
||
raise RewriteError("rewrite patch is outside the Markdown body")
|
||
if patch.end > last_start:
|
||
raise RewriteError("rewrite patches overlap")
|
||
result = result[: patch.start] + patch.replacement + result[patch.end :]
|
||
last_start = patch.start
|
||
return result
|
||
|
||
|
||
def _create_section_signal(key: str) -> str:
|
||
annotation_type = next(
|
||
annotation_type
|
||
for annotation_type, target in ANNOTATION_TARGET.items()
|
||
if target == key
|
||
)
|
||
return ANNOTATION_SIGNAL[annotation_type]
|
||
|
||
|
||
def rewrite_document(
|
||
document: SkillDocument,
|
||
signals: dict[str, Signal],
|
||
annotations: list[Annotation],
|
||
*,
|
||
reserved_block_ids: set[str] | None = None,
|
||
) -> tuple[str, list[Operation]]:
|
||
blocks = [replace(block) for block in document.blocks]
|
||
by_id = {block.id: block for block in blocks}
|
||
operations: list[Operation] = []
|
||
patches: list[_Patch] = []
|
||
reserved = reserved_block_ids or set()
|
||
patched_blocks: set[str] = set(reserved)
|
||
payloads: dict[str, list[str]] = defaultdict(list)
|
||
|
||
robustness = signals["semantic_robustness"]
|
||
if robustness.level in {"low", "medium"}:
|
||
for block in blocks:
|
||
if block.kind != "heading":
|
||
continue
|
||
replacement, operation = _normalize_known_heading(
|
||
block, document.language
|
||
)
|
||
if replacement is not None and operation is not None:
|
||
patches.append(
|
||
_Patch(
|
||
block.start_offset,
|
||
block.start_offset + len(block.text),
|
||
replacement,
|
||
)
|
||
)
|
||
patched_blocks.add(block.id)
|
||
operations.append(operation)
|
||
|
||
for annotation in annotations:
|
||
signal_name = ANNOTATION_SIGNAL.get(annotation.type)
|
||
if signal_name is None or not _enabled(signals[signal_name], annotation):
|
||
continue
|
||
block = by_id.get(annotation.block_id)
|
||
if block is None or annotation.quote not in block.text:
|
||
continue
|
||
if block.id in reserved:
|
||
continue
|
||
quote_offset = block.text.find(annotation.quote)
|
||
absolute_start = block.start_offset + quote_offset
|
||
absolute_end = absolute_start + len(annotation.quote)
|
||
|
||
if annotation.type == "coreference":
|
||
if (
|
||
robustness.level != "low"
|
||
or annotation.antecedent_quote is None
|
||
or block.id in patched_blocks
|
||
):
|
||
continue
|
||
patches.append(
|
||
_Patch(absolute_start, absolute_end, annotation.antecedent_quote)
|
||
)
|
||
patched_blocks.add(block.id)
|
||
operations.append(
|
||
Operation(
|
||
type="REPLACE_COREFERENCE_WITH_SOURCE_QUOTE",
|
||
signal=signal_name,
|
||
annotation_type=annotation.type,
|
||
block_id=block.id,
|
||
quote=annotation.quote,
|
||
replacement=annotation.antecedent_quote,
|
||
)
|
||
)
|
||
continue
|
||
|
||
target = ANNOTATION_TARGET[annotation.type]
|
||
if section_key(block.parent_heading) == target:
|
||
continue
|
||
if _already_structurally_prominent(block, annotation.quote):
|
||
continue
|
||
|
||
# If the exact rule already occurs more than once, a prior compilation
|
||
# has already added a summary copy. This makes compilation idempotent.
|
||
if document.body.count(annotation.quote) > 1:
|
||
continue
|
||
|
||
if _is_standalone_rule(block, annotation.quote):
|
||
formatted = _format_payload(annotation.quote)
|
||
if formatted not in payloads[target]:
|
||
payloads[target].append(formatted)
|
||
operations.append(
|
||
Operation(
|
||
type="DUPLICATE_EXACT",
|
||
signal=signal_name,
|
||
annotation_type=annotation.type,
|
||
block_id=block.id,
|
||
quote=annotation.quote,
|
||
replacement=formatted,
|
||
target_section=target,
|
||
)
|
||
)
|
||
continue
|
||
|
||
# Context-dependent fragments stay where they are. Critical directives
|
||
# get local Markdown emphasis; other non-standalone semantic fragments
|
||
# are left untouched.
|
||
if (
|
||
annotation.type in LOCAL_EMPHASIS_TYPES
|
||
and not _already_emphasized(
|
||
block.text, quote_offset, annotation.quote
|
||
)
|
||
and _safe_to_emphasize(annotation.quote)
|
||
and block.id not in patched_blocks
|
||
):
|
||
replacement = _emphasize(annotation.quote)
|
||
patches.append(_Patch(absolute_start, absolute_end, replacement))
|
||
patched_blocks.add(block.id)
|
||
operations.append(
|
||
Operation(
|
||
type="EMPHASIZE_IN_PLACE",
|
||
signal=signal_name,
|
||
annotation_type=annotation.type,
|
||
block_id=block.id,
|
||
quote=annotation.quote,
|
||
replacement=replacement,
|
||
target_section=target,
|
||
)
|
||
)
|
||
|
||
if robustness.level == "low":
|
||
for block in blocks:
|
||
if block.id in patched_blocks:
|
||
continue
|
||
replacement, operation = _split_explicit_requirements(block)
|
||
if replacement is not None and operation is not None:
|
||
patches.append(
|
||
_Patch(
|
||
block.start_offset,
|
||
block.start_offset + len(block.text),
|
||
replacement,
|
||
)
|
||
)
|
||
patched_blocks.add(block.id)
|
||
operations.append(operation)
|
||
|
||
existing_sections: dict[str, BodyBlock] = {}
|
||
for block in blocks:
|
||
if block.kind == "heading":
|
||
key = section_key(_heading_text(block))
|
||
if key is not None:
|
||
existing_sections[key] = block
|
||
|
||
insertions: dict[int, list[str]] = defaultdict(list)
|
||
new_pre_sections: list[str] = []
|
||
completion_section: str | None = None
|
||
for key in PRE_SECTION_ORDER + ("completion_criterion",):
|
||
values = payloads.get(key, [])
|
||
if not values:
|
||
continue
|
||
if key in existing_sections:
|
||
heading = existing_sections[key]
|
||
insertions[heading.end_offset].append(
|
||
document.newline + document.newline.join(values) + document.newline
|
||
)
|
||
continue
|
||
section = (
|
||
f"## {_section_heading(key, document.language)}"
|
||
f"{document.newline}{document.newline}"
|
||
+ document.newline.join(values)
|
||
)
|
||
operations.append(
|
||
Operation(
|
||
type="CREATE_SECTION",
|
||
signal=_create_section_signal(key),
|
||
target_section=key,
|
||
)
|
||
)
|
||
if key == "completion_criterion":
|
||
completion_section = section
|
||
else:
|
||
new_pre_sections.append(section)
|
||
|
||
if new_pre_sections:
|
||
first_h2 = next(
|
||
(
|
||
block
|
||
for block in blocks
|
||
if block.kind == "heading" and (block.heading_level or 0) >= 2
|
||
),
|
||
None,
|
||
)
|
||
if first_h2 is not None:
|
||
position = first_h2.start_offset
|
||
text = (
|
||
(document.newline * 2).join(new_pre_sections)
|
||
+ document.newline
|
||
+ document.newline
|
||
)
|
||
else:
|
||
first_h1 = next(
|
||
(
|
||
block
|
||
for block in blocks
|
||
if block.kind == "heading" and block.heading_level == 1
|
||
),
|
||
None,
|
||
)
|
||
position = first_h1.end_offset if first_h1 is not None else 0
|
||
text = (
|
||
document.newline
|
||
+ (document.newline * 2).join(new_pre_sections)
|
||
+ document.newline
|
||
+ document.newline
|
||
)
|
||
insertions[position].append(text)
|
||
|
||
if completion_section is not None:
|
||
prefix = "" if document.body.endswith(document.newline * 2) else document.newline
|
||
insertions[len(document.body)].append(
|
||
prefix + completion_section + document.newline
|
||
)
|
||
|
||
for position, values in insertions.items():
|
||
patches.append(_Patch(position, position, "".join(values)))
|
||
|
||
if not operations:
|
||
return document.original, []
|
||
body = _apply_patches(document.body, patches)
|
||
return document.frontmatter + body, operations
|