Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+58
View File
@@ -0,0 +1,58 @@
"""完整编译流水线的唯一命令行入口。"""
from __future__ import annotations
import argparse
import sys
from scripts.provider_router import parse_model_reference
from .pipeline import run_pipeline
def _model_reference(value: str) -> str:
try:
return parse_model_reference(value).value
except ValueError as exc:
raise argparse.ArgumentTypeError(str(exc)) from exc
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="python -m scripts.compile_pipeline",
description="Run static compilation, Fast optimization, and Deep iteration for one SkillsBench task.",
)
parser.add_argument("--harness", required=True, choices=("opencode", "claude-code", "claude"))
parser.add_argument(
"--model", required=True, type=_model_reference,
help="Target model used by BenchFlow, in provider/model format.",
)
parser.add_argument(
"--external-model", required=True, type=_model_reference,
help="Model used for static planning, Fast analysis, and Deep analysis.",
)
parser.add_argument(
"--task", required=True,
help="SkillsBench task name or a path below data/skills-bench/tasks.",
)
parser.add_argument(
"--run-dir",
help="Run directory below results/compile-pipeline; reuse it to resume an interrupted run.",
)
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
final_skill = run_pipeline(
harness=args.harness, model=args.model, external_model=args.external_model,
task=args.task, run_dir=args.run_dir,
)
except (OSError, RuntimeError, ValueError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
print(final_skill)
return 0
raise SystemExit(main())
+166
View File
@@ -0,0 +1,166 @@
"""BenchFlow 评测执行、完整性判断与恢复。"""
from __future__ import annotations
import json
from pathlib import Path
import subprocess
import sys
from typing import Any
import uuid
from scripts.dynamic_compile.fast.storage import sha256_file
from .manifest import read_json_object
from .paths import PROJECT_ROOT
EVALUATION_SCRIPT = PROJECT_ROOT / "scripts" / "evaluate" / "run-raw-task.sh"
MAX_PARALLEL = 3
def evaluation_artifacts_complete(test_dir: Path) -> bool:
"""Return whether one BenchFlow attempt has all required usable artifacts."""
summary = read_json_object(test_dir / "summary.json")
required_skill = read_json_object(test_dir / "required-skill.json")
if summary is None or required_skill is None:
return False
try:
total = int(summary.get("total", 0) or 0)
passed = int(summary.get("passed", summary.get("pass", 0)) or 0)
failed = int(summary.get("failed", summary.get("fail", 0)) or 0)
errored = int(summary.get("errored", summary.get("error", 0)) or 0)
verifier_errored = int(summary.get("verifier_errored", 0) or 0)
except (TypeError, ValueError):
return False
if total != 1 or passed + failed != 1 or errored or verifier_errored:
return False
if required_skill.get("invoked") is not True or required_skill.get("parse_errors"):
return False
trajectories = sorted(test_dir.rglob("acp_trajectory.jsonl"))
canonical = [path for path in trajectories if "trajectory" in path.parts]
trajectory = canonical[0] if canonical else (trajectories[0] if trajectories else None)
if trajectory is None:
return False
result = read_json_object(trajectory.parent.parent / "result.json")
if result is None:
return False
try:
events = [
json.loads(line)
for line in trajectory.read_text(encoding="utf-8").splitlines()
if line.strip()
]
except (OSError, json.JSONDecodeError):
return False
return bool(events) and all(isinstance(event, dict) for event in events)
def completed_evaluation_count(output: Path) -> int:
if not output.is_dir():
return 0
return sum(
evaluation_artifacts_complete(test_dir)
for test_dir in output.glob("test-*")
if test_dir.is_dir()
)
def quarantine_incomplete_evaluations(output: Path) -> int:
if not output.is_dir():
return 0
incomplete = [
test_dir
for test_dir in sorted(output.glob("test-*"))
if test_dir.is_dir() and not evaluation_artifacts_complete(test_dir)
]
if not incomplete:
return 0
quarantine = output / ".incomplete"
quarantine.mkdir(exist_ok=True)
for test_dir in incomplete:
destination = quarantine / test_dir.name
if destination.exists():
destination = quarantine / f"{test_dir.name}-{uuid.uuid4().hex[:6]}"
test_dir.replace(destination)
print(
f"[compile-pipeline] preserved incomplete evaluation at {destination}",
file=sys.stderr,
flush=True,
)
return len(incomplete)
def evaluate(
*,
harness: str,
model: str,
task_dir: Path,
skill_source: Path,
output: Path,
repeat: int,
) -> None:
completed = completed_evaluation_count(output)
if completed >= repeat:
print(
f"[compile-pipeline] evaluation complete: {output} ({completed}/{repeat}); skipping",
file=sys.stderr,
flush=True,
)
return
quarantine_incomplete_evaluations(output)
missing = repeat - completed
if completed:
print(
f"[compile-pipeline] evaluation incomplete: {output} "
f"({completed}/{repeat}); running {missing} missing rollout(s)",
file=sys.stderr,
flush=True,
)
command = [
"bash", str(EVALUATION_SCRIPT), "--harness", harness, "--model", model,
"--task", str(task_dir), "--skill-source", str(skill_source), "--output", str(output),
"--require-skill", "--repeat", str(missing), "--max-parallel", str(MAX_PARALLEL),
]
try:
subprocess.run(command, cwd=PROJECT_ROOT, check=True)
except subprocess.CalledProcessError as exc:
raise RuntimeError(
f"evaluation failed for {skill_source} with exit code {exc.returncode}"
) from exc
completed = completed_evaluation_count(output)
if completed < repeat:
raise RuntimeError(
f"evaluation produced only {completed}/{repeat} complete rollout artifacts "
f"under {output}"
)
def complete_deep_skill(output: Path) -> Path | None:
state = read_json_object(output / "run.json")
report = read_json_object(output / "report.json")
skill = output / "S_final"
if (
state is not None and state.get("status") == "complete"
and report is not None and report.get("status") == "complete"
and (skill / "SKILL.md").is_file()
):
return skill
return None
def deep_final_rollouts(output: Path, final_skill: Path, repeat: int) -> Path | None:
report = read_json_object(output / "report.json")
if report is None:
return None
value = report.get("final_rollouts")
expected_hash = report.get("final_rollout_skill_sha256")
if not isinstance(value, str) or not isinstance(expected_hash, str):
return None
rollouts = Path(value).resolve()
if (
not rollouts.is_dir() or expected_hash != sha256_file(final_skill / "SKILL.md")
or completed_evaluation_count(rollouts) < repeat
):
return None
return rollouts
+78
View File
@@ -0,0 +1,78 @@
"""可恢复运行的原子清单持久化。"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any
from scripts.dynamic_compile.fast.storage import atomic_write_json
def read_json_object(path: Path) -> dict[str, Any] | None:
try:
value = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return None
return value if isinstance(value, dict) else None
class RunManifest:
def __init__(
self,
path: Path,
*,
harness: str,
model: str,
external_model: str,
task_dir: Path,
):
self.path = path
inputs = {
"harness": harness,
"model": model,
"external_model": external_model,
"task": str(task_dir),
}
if path.is_file():
existing = read_json_object(path)
if existing is None:
raise ValueError(f"invalid run manifest: {path}")
if existing.get("schema_version") != "1.0":
raise ValueError(f"unsupported run manifest schema: {path}")
if existing.get("inputs") != inputs:
raise ValueError(
f"run directory belongs to different pipeline inputs: {path.parent}"
)
stages = existing.get("stages")
if not isinstance(stages, dict):
raise ValueError(f"invalid stages in run manifest: {path}")
self.value = existing
self.value["status"] = "running"
self.value.pop("error", None)
self.save()
return
self.value: dict[str, Any] = {
"schema_version": "1.0",
"status": "running",
"inputs": inputs,
"stages": {},
}
self.save()
def save(self) -> None:
atomic_write_json(self.path, self.value)
def stage(self, name: str, status: str, **details: Any) -> None:
self.value["stages"][name] = {"status": status, **details}
self.save()
def complete(self, final_skill: Path) -> None:
self.value["status"] = "complete"
self.value["final_skill"] = str(final_skill)
self.save()
def fail(self, error: Exception) -> None:
self.value["status"] = "failed"
self.value["error"] = f"{type(error).__name__}: {error}"
self.save()
+60
View File
@@ -0,0 +1,60 @@
"""项目路径解析与输入范围校验。"""
from __future__ import annotations
from datetime import datetime
from pathlib import Path
import uuid
PROJECT_ROOT = Path(__file__).resolve().parents[2]
TASKS_ROOT = PROJECT_ROOT / "data" / "skills-bench" / "tasks"
RESULTS_ROOT = PROJECT_ROOT / "results" / "compile-pipeline"
def task_directory(value: str) -> Path:
requested = Path(value).expanduser()
if requested.is_absolute():
candidate = requested.resolve()
elif len(requested.parts) == 1:
candidate = (TASKS_ROOT / requested).resolve()
else:
candidate = (PROJECT_ROOT / requested).resolve()
try:
candidate.relative_to(TASKS_ROOT.resolve())
except ValueError as exc:
raise ValueError(f"--task must resolve below {TASKS_ROOT}") from exc
if not (candidate / "task.md").is_file():
raise ValueError(f"not a SkillsBench task directory: {candidate}")
skills = candidate / "environment" / "skills"
if not skills.is_dir():
raise ValueError(f"SkillsBench task has no environment/skills directory: {candidate}")
return candidate
def single_skill_source(task_dir: Path) -> Path:
skills_root = task_dir / "environment" / "skills"
skill_files = sorted(skills_root.rglob("SKILL.md"))
if len(skill_files) != 1:
raise ValueError(
"the complete pipeline currently requires exactly one task Skill because "
f"Fast and Deep accept one Skill package; found {len(skill_files)} under {skills_root}"
)
return skills_root
def requested_run_root(value: str | Path | None) -> Path:
if value is None:
run_id = datetime.now().strftime("%Y%m%d-%H%M%S") + "-" + uuid.uuid4().hex[:6]
return (RESULTS_ROOT / run_id).resolve()
requested = Path(value).expanduser()
candidate = (
requested.resolve()
if requested.is_absolute()
else (PROJECT_ROOT / requested).resolve()
)
try:
candidate.relative_to(RESULTS_ROOT.resolve())
except ValueError as exc:
raise ValueError(f"--run-dir must resolve below {RESULTS_ROOT}") from exc
return candidate
+195
View File
@@ -0,0 +1,195 @@
"""完整静态、Fast 与 Deep 技能编译流水线。"""
from __future__ import annotations
from pathlib import Path
import shutil
import sys
from typing import Any
from scripts.dynamic_compile.deep.pipeline import DeepLoop
from scripts.dynamic_compile.fast.pipeline import run_pipeline as run_fast_pipeline
from scripts.static_compile.compiler.compiler import compile_input
from scripts.static_compile.profile_generation.pipeline import ensure_profile
from .evaluation import (
MAX_PARALLEL,
complete_deep_skill,
deep_final_rollouts,
evaluate,
)
from .manifest import RunManifest, read_json_object
from .paths import requested_run_root, single_skill_source, task_directory
ORIGINAL_ROLLOUTS = 6
STATIC_ROLLOUTS = 6
FAST_ROLLOUTS = 3
FINAL_ROLLOUTS = 3
def _complete_static_skill(output: Path) -> tuple[Path, str] | None:
complete: list[tuple[Path, str]] = []
if not output.is_dir():
return None
for report_path in output.rglob("rewrite-report.json"):
skill = report_path.parent
report = read_json_object(report_path)
if report is None or not (skill / "SKILL.md").is_file():
continue
status = str(report.get("status", "failed"))
if status not in {"failed", "rolled_back"}:
complete.append((skill, status))
return complete[0] if len(complete) == 1 else None
def run_pipeline(
*,
harness: str,
model: str,
external_model: str,
task: str,
run_dir: str | Path | None = None,
) -> Path:
task_dir = task_directory(task)
source_skills = single_skill_source(task_dir)
task_name = task_dir.name
run_root = requested_run_root(run_dir)
if (
run_root.is_dir()
and any(run_root.iterdir())
and not (run_root / "manifest.json").is_file()
):
raise ValueError(
f"non-empty run directory has no manifest and cannot be resumed: {run_root}"
)
run_root.mkdir(parents=True, exist_ok=True)
manifest = RunManifest(
run_root / "manifest.json",
harness=harness,
model=model,
external_model=external_model,
task_dir=task_dir,
)
artifacts = run_root / "artifacts"
trace_task_root = run_root / "traces" / task_name
original_traces = trace_task_root / "ori_skill"
static_root = artifacts / "static"
static_traces = trace_task_root / "model_skill"
fast_score = artifacts / "fast" / "score"
fast_output = artifacts / "fast" / "optimization"
fast_traces = trace_task_root / "fast_skill"
deep_output = artifacts / "deep"
final_traces = trace_task_root / "final_skill"
print(f"[compile-pipeline] run directory: {run_root}", file=sys.stderr, flush=True)
try:
manifest.stage("original_evaluation", "running", output=str(original_traces))
evaluate(
harness=harness, model=model, task_dir=task_dir, skill_source=source_skills,
output=original_traces, repeat=ORIGINAL_ROLLOUTS,
)
manifest.stage(
"original_evaluation", "complete", output=str(original_traces),
rollouts=ORIGINAL_ROLLOUTS, max_parallel=MAX_PARALLEL,
)
manifest.stage("profile", "running")
profile_path, generated = ensure_profile(model)
manifest.stage("profile", "complete", path=str(profile_path), generated=generated)
cached_static = _complete_static_skill(static_root)
if cached_static is not None:
static_skill, static_status = cached_static
print(
f"[compile-pipeline] static compilation complete: {static_skill}; skipping",
file=sys.stderr, flush=True,
)
else:
manifest.stage("static_compile", "running")
static_results = compile_input(
source_skills, profile_path, static_root, mode="hybrid",
annotator_model=external_model, force=static_root.exists(),
)
if len(static_results) != 1 or static_results[0].output_dir is None:
raise RuntimeError("static compilation did not produce exactly one Skill package")
static_status = str(static_results[0].report.get("status", "failed"))
if static_status in {"failed", "rolled_back"}:
raise RuntimeError(f"static compilation ended with status {static_status}")
static_skill = static_results[0].output_dir
manifest.stage(
"static_compile", "complete", skill=str(static_skill), compile_status=static_status,
)
manifest.stage("static_evaluation", "running", output=str(static_traces))
evaluate(
harness=harness, model=model, task_dir=task_dir, skill_source=static_skill,
output=static_traces, repeat=STATIC_ROLLOUTS,
)
manifest.stage(
"static_evaluation", "complete", output=str(static_traces),
rollouts=STATIC_ROLLOUTS, max_parallel=MAX_PARALLEL,
)
manifest.stage("fast_compile", "running")
fast_skill = run_fast_pipeline(
static_traces, static_skill, score_output=fast_score, output=fast_output,
model=external_model, max_parallel=MAX_PARALLEL,
)
manifest.stage("fast_compile", "complete", skill=str(fast_skill))
manifest.stage("fast_evaluation", "running", output=str(fast_traces))
evaluate(
harness=harness, model=model, task_dir=task_dir, skill_source=fast_skill,
output=fast_traces, repeat=FAST_ROLLOUTS,
)
manifest.stage(
"fast_evaluation", "complete", output=str(fast_traces),
rollouts=FAST_ROLLOUTS, max_parallel=MAX_PARALLEL,
)
final_skill = complete_deep_skill(deep_output)
if final_skill is not None:
print(
f"[compile-pipeline] deep compilation complete: {final_skill}; skipping",
file=sys.stderr, flush=True,
)
else:
manifest.stage("deep_compile", "running", output=str(deep_output))
deep_loop = (
DeepLoop(deep_output)
if (deep_output / "run.json").is_file()
else DeepLoop.create(fast_skill, fast_traces, deep_output, model=external_model)
)
final_skill = deep_loop.drive()
if complete_deep_skill(deep_output) is None:
raise RuntimeError(
f"deep compilation did not produce complete artifacts under {deep_output}"
)
manifest.stage("deep_compile", "complete", skill=str(final_skill))
reusable_rollouts = deep_final_rollouts(
deep_output, final_skill, FINAL_ROLLOUTS,
)
if reusable_rollouts is not None:
print(
f"[compile-pipeline] copying Deep final rollouts to: {final_traces}",
file=sys.stderr, flush=True,
)
shutil.copytree(reusable_rollouts, final_traces, dirs_exist_ok=True)
final_evaluation_output = final_traces
manifest.stage("final_evaluation", "running", output=str(final_evaluation_output))
evaluate(
harness=harness, model=model, task_dir=task_dir, skill_source=final_skill,
output=final_evaluation_output, repeat=FINAL_ROLLOUTS,
)
manifest.stage(
"final_evaluation", "complete", output=str(final_evaluation_output),
rollouts=FINAL_ROLLOUTS, max_parallel=MAX_PARALLEL,
reused_deep_rollouts=reusable_rollouts is not None,
)
manifest.complete(final_skill)
return final_skill
except Exception as exc:
manifest.fail(exc)
raise