Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+480
View File
@@ -0,0 +1,480 @@
"""Directory compiler orchestration for model-profile Skill adaptation."""
from __future__ import annotations
import hashlib
import json
from pathlib import Path
import re
import shutil
import tempfile
from typing import Any, Callable
from .annotator import (
AnnotationError,
OpenCodeAnnotator,
SemanticPlanner,
plan_once,
)
from .document import (
DocumentError,
parse_document,
resolve_annotation_conflicts,
skill_name,
static_annotations,
)
from .guard import run_semantic_guard
from .format_policy import apply_format_style, reduce_format_policy
from .models import CompileResult, SemanticPlanResult, Signal
from .profile import (
ProfileError,
load_profile,
selected_passes,
target_model_id,
)
from .rewriter import RewriteError, rewrite_document
from .semantic_plan import apply_semantic_plan, semantic_plan_needed
class ModelCompilerError(RuntimeError):
"""A model preference compilation failed."""
ProgressCallback = Callable[[int, str], None]
def _notify(
progress: ProgressCallback | None,
percent: int,
message: str,
) -> None:
if progress is not None:
progress(percent, message)
def _sha256(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def _slug(value: str) -> str:
result = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
return result or "model"
def _validate_source_tree(source: Path) -> None:
source_resolved = source.resolve()
for path in source.rglob("*"):
if not path.is_symlink():
continue
try:
target = path.resolve(strict=True)
target.relative_to(source_resolved)
except (OSError, ValueError) as exc:
raise ModelCompilerError(
f"symlink escapes or is broken in Skill source: {path}"
) from exc
def _is_within(path: Path, parent: Path) -> bool:
try:
path.resolve().relative_to(parent.resolve())
return True
except ValueError:
return False
def _retained_diagnostics(profile: dict[str, Any]) -> dict[str, Any]:
dimensions = profile.get("behavioral_profile", {}).get("numeric_dimensions", [])
retained_ids = {"causal_chain", "abstract_reasoning"}
retained = [
dimension
for dimension in dimensions
if isinstance(dimension, dict) and dimension.get("id") in retained_ids
]
style = profile.get("behavioral_profile", {}).get("style_profile")
return {"numeric_dimensions": retained, "style_profile": style}
def _base_report(
source: Path,
source_bytes: bytes,
profile: dict[str, Any],
profile_hash: str,
signals: dict[str, Signal],
passes: list[str],
) -> dict[str, Any]:
return {
"schema_version": "1.0",
"status": "unchanged",
"source": {
"path": str(source),
"sha256": _sha256(source_bytes),
},
"target_model": {
"id": target_model_id(profile),
"profile_sha256": profile_hash,
},
"signals": {
name: signal.to_dict() for name, signal in sorted(signals.items())
},
"selected_passes": passes,
"retained_diagnostics": _retained_diagnostics(profile),
"semantic_plan": SemanticPlanResult().to_dict(),
"operations": [],
"semantic_guard": {},
"warnings": [],
}
def _write_output(
source_dir: Path,
destination: Path,
skill_content: str,
report: dict[str, Any],
*,
force: bool,
) -> None:
if destination.exists() and not force:
raise ModelCompilerError(
f"output already exists (use --force to replace it): {destination}"
)
destination.parent.mkdir(parents=True, exist_ok=True)
staging = Path(
tempfile.mkdtemp(prefix=f".{destination.name}.tmp-", dir=destination.parent)
)
try:
shutil.rmtree(staging)
shutil.copytree(source_dir, staging, symlinks=True)
(staging / "SKILL.md").write_text(
skill_content, encoding="utf-8", newline=""
)
(staging / "rewrite-report.json").write_text(
json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
if destination.exists():
shutil.rmtree(destination)
staging.replace(destination)
finally:
if staging.exists():
shutil.rmtree(staging)
def _copy_pack_scaffolding(
pack_dir: Path,
destination: Path,
skill_dirs: list[Path],
*,
force: bool,
) -> None:
"""Copy files owned by a Skill pack rather than by one of its Skills.
Each Skill is copied by ``compile_skill`` so its SKILL.md can be replaced.
This preserves pack-level manifests, shared assets, and intermediate
directories without copying an old SKILL.md over a rewritten one.
"""
if destination.exists() and not force:
return
destination.mkdir(parents=True, exist_ok=True)
for path in sorted(pack_dir.rglob("*"), key=lambda item: item.as_posix()):
if any(path == skill_dir or skill_dir in path.parents for skill_dir in skill_dirs):
continue
target = destination / path.relative_to(pack_dir)
if path.is_dir():
target.mkdir(parents=True, exist_ok=True)
else:
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(path, target, follow_symlinks=False)
def compile_skill(
input_dir: Path,
profile_path: Path,
out_root: Path,
*,
mode: str = "deterministic",
annotator_model: str | None = None,
allow_deterministic_fallback: bool = False,
dry_run: bool = False,
force: bool = False,
annotator: SemanticPlanner | None = None,
output_group: str | None = None,
output_relative_path: Path | None = None,
progress: ProgressCallback | None = None,
) -> CompileResult:
_notify(progress, 3, f"{input_dir.name}: reading Skill and profile")
if mode not in {"deterministic", "hybrid"}:
raise ModelCompilerError(f"unsupported mode: {mode}")
source_dir = input_dir.resolve()
skill_path = source_dir / "SKILL.md"
if not skill_path.is_file():
raise ModelCompilerError(f"source Skill directory requires SKILL.md: {input_dir}")
_validate_source_tree(source_dir)
try:
source_bytes = skill_path.read_bytes()
source_text = source_bytes.decode("utf-8")
document = parse_document(source_text)
name = skill_name(document)
profile, signals, profile_hash = load_profile(profile_path.resolve())
except (OSError, UnicodeDecodeError, DocumentError, ProfileError) as exc:
raise ModelCompilerError(str(exc)) from exc
passes = selected_passes(signals)
_notify(progress, 15, f"{name}: profile reduced; {len(passes)} pass(es) selected")
report = _base_report(
skill_path, source_bytes, profile, profile_hash, signals, passes
)
format_policy = reduce_format_policy(profile)
selected_format_styles = list(format_policy.styles) if format_policy.enabled else []
selected_format_style = selected_format_styles[-1] if selected_format_styles else None
report["format_policy"] = format_policy.to_dict()
report["selected_format_styles"] = [
style.to_dict() for style in selected_format_styles
]
report["selected_format_style"] = (
selected_format_style.to_dict() if selected_format_style else None
)
static = static_annotations(document)
_notify(progress, 25, f"{name}: Markdown analyzed; protected blocks identified")
needs_llm = mode == "hybrid" and semantic_plan_needed(document)
report["dry_run"] = dry_run
report["expected_llm_call"] = needs_llm
if dry_run:
_notify(progress, 100, f"{name}: dry run complete")
return CompileResult(None, report, name)
plan_result = SemanticPlanResult()
if needs_llm:
_notify(progress, 30, f"{name}: requesting source-grounded semantic plan")
try:
active_planner = annotator
if active_planner is None:
if not annotator_model:
raise AnnotationError(
"hybrid semantic planning requires a provider-qualified model"
)
active_planner = OpenCodeAnnotator(
annotator_model,
progress=progress,
)
plan_result = plan_once(
active_planner,
document,
signals,
passes,
)
except AnnotationError as exc:
if not allow_deterministic_fallback:
raise ModelCompilerError(str(exc)) from exc
plan_result = SemanticPlanResult(
used=True,
model=(annotator.model_id if annotator is not None else annotator_model),
error=str(exc),
)
report["warnings"].append(
f"semantic planning failed; deterministic fallback used: {exc}"
)
if plan_result.repair_error is not None:
report["warnings"].append(
"semantic repair failed; valid units from the initial plan were retained: "
f"{plan_result.repair_error}"
)
_notify(
progress,
52,
f"{name}: semantic plan ready "
f"({plan_result.accepted} accepted, {plan_result.rejected} rejected)",
)
report["semantic_plan"] = plan_result.to_dict()
reserved_block_ids = {
unit.source_refs[0].block_id
for unit in plan_result.units
if unit.kind == "replace_block"
}
annotations = resolve_annotation_conflicts(
[item for item in static if item.block_id not in reserved_block_ids]
)
try:
_notify(progress, 62, f"{name}: applying deterministic behavioral passes")
rewritten, operations = rewrite_document(
document,
signals,
annotations,
reserved_block_ids=reserved_block_ids,
)
if plan_result.units:
_notify(progress, 72, f"{name}: applying validated semantic rewrites")
rewritten, semantic_operations, skipped = apply_semantic_plan(
rewritten, document, plan_result.units
)
operations.extend(semantic_operations)
plan_result.applied = len(semantic_operations)
plan_result.skipped = len(skipped)
plan_result.skip_reasons = skipped
report["semantic_plan"] = plan_result.to_dict()
if selected_format_styles:
for index, format_style in enumerate(selected_format_styles, start=1):
_notify(
progress,
80 + min(10, index),
f"{name}: applying model format preference {index}/{len(selected_format_styles)}",
)
rewritten, format_operations = apply_format_style(
rewritten, format_style
)
operations.extend(format_operations)
except RewriteError as exc:
raise ModelCompilerError(str(exc)) from exc
_notify(progress, 90, f"{name}: running semantic guard")
guard = run_semantic_guard(source_text, rewritten, operations)
report["operations"] = [operation.to_dict() for operation in operations]
report["semantic_guard"] = guard.to_dict()
if not guard.passed:
output_content = source_text
report["status"] = "rolled_back"
report["warnings"].append(
"semantic guard failed; output SKILL.md was rolled back to source"
)
elif plan_result.error is not None:
output_content = rewritten
report["status"] = "deterministic_fallback"
elif rewritten == source_text:
output_content = source_text
report["status"] = "unchanged"
else:
output_content = rewritten
report["status"] = "adapted"
model_root = out_root.resolve() / _slug(target_model_id(profile))
if output_group is not None and output_relative_path is not None:
raise ModelCompilerError(
"output_group and output_relative_path cannot be used together"
)
if output_relative_path is not None:
if output_relative_path.is_absolute() or any(
part in {"", ".", ".."} for part in output_relative_path.parts
):
raise ModelCompilerError(
f"invalid relative output path: {output_relative_path}"
)
destination = model_root / output_relative_path
elif output_group is not None:
if (
not output_group
or output_group in {".", ".."}
or Path(output_group).name != output_group
):
raise ModelCompilerError(
f"invalid output collection directory name: {output_group!r}"
)
model_root = model_root / output_group
destination = model_root / name
else:
destination = model_root / name
if _is_within(destination, source_dir):
raise ModelCompilerError("output directory must not be inside the source Skill")
_notify(progress, 96, f"{name}: writing compiled Skill and report")
_write_output(
source_dir,
destination,
output_content,
report,
force=force,
)
if skill_path.read_bytes() != source_bytes:
raise ModelCompilerError("source SKILL.md changed during compilation")
_notify(progress, 100, f"{name}: compilation complete ({report['status']})")
return CompileResult(destination, report, name)
def compile_input(
input_dir: Path,
profile_path: Path,
out_root: Path,
**kwargs: Any,
) -> tuple[CompileResult, ...]:
progress = kwargs.pop("progress", None)
source = input_dir.resolve()
if not source.is_dir():
raise ModelCompilerError(f"input directory not found: {input_dir}")
if (source / "SKILL.md").is_file():
return (
compile_skill(
source,
profile_path,
out_root,
progress=progress,
**kwargs,
),
)
_validate_source_tree(source)
skill_dirs = sorted(
(path.parent for path in source.rglob("SKILL.md") if path.is_file()),
key=lambda child: child.relative_to(source).as_posix(),
)
if not skill_dirs:
raise ModelCompilerError(
f"input requires a Skill directory or a Skill pack containing SKILL.md files: "
f"{input_dir}"
)
try:
profile, _, _ = load_profile(profile_path.resolve())
except ProfileError as exc:
raise ModelCompilerError(str(exc)) from exc
pack_destination = (
out_root.resolve() / _slug(target_model_id(profile)) / source.name
)
if _is_within(pack_destination, source):
raise ModelCompilerError("output directory must not be inside the source Skill pack")
if not kwargs.get("dry_run", False):
_copy_pack_scaffolding(
source,
pack_destination,
skill_dirs,
force=bool(kwargs.get("force", False)),
)
# A pack is a batch boundary, not a transaction. Compile Skills
# Skills sequentially in a stable order and isolate an expected failure to
# the current Skill. This preserves the strict single-Skill behavior while
# ensuring one provider/validation/output error cannot skip later Skills.
results: list[CompileResult] = []
total = len(skill_dirs)
for index, skill_dir in enumerate(skill_dirs):
child_progress: ProgressCallback | None = None
if progress is not None:
def child_progress(
percent: int,
message: str,
*,
_index: int = index,
) -> None:
overall = int(((_index + percent / 100) / total) * 100)
progress(overall, f"[{_index + 1}/{total}] {message}")
try:
result = compile_skill(
skill_dir,
profile_path,
out_root,
# A pack mirrors each Skill's path below the pack root. Using the
# directory path rather than frontmatter name also avoids collisions
# when separate subdirectories contain Skills with the same name.
output_relative_path=Path(source.name) / skill_dir.relative_to(source),
progress=child_progress,
**kwargs,
)
except ModelCompilerError as exc:
result = CompileResult(
output_dir=None,
skill_name=skill_dir.name,
report={
"schema_version": "1.0",
"status": "failed",
"source": {
"path": str((skill_dir / "SKILL.md").resolve()),
},
"error": str(exc),
"warnings": [f"Skill compilation failed: {exc}"],
},
)
results.append(result)
return tuple(results)