Initial commit
This commit is contained in:
@@ -0,0 +1,58 @@
|
||||
"""完整编译流水线的唯一命令行入口。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
from scripts.provider_router import parse_model_reference
|
||||
|
||||
from .pipeline import run_pipeline
|
||||
|
||||
|
||||
def _model_reference(value: str) -> str:
|
||||
try:
|
||||
return parse_model_reference(value).value
|
||||
except ValueError as exc:
|
||||
raise argparse.ArgumentTypeError(str(exc)) from exc
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="python -m scripts.compile_pipeline",
|
||||
description="Run static compilation, Fast optimization, and Deep iteration for one SkillsBench task.",
|
||||
)
|
||||
parser.add_argument("--harness", required=True, choices=("opencode", "claude-code", "claude"))
|
||||
parser.add_argument(
|
||||
"--model", required=True, type=_model_reference,
|
||||
help="Target model used by BenchFlow, in provider/model format.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--external-model", required=True, type=_model_reference,
|
||||
help="Model used for static planning, Fast analysis, and Deep analysis.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--task", required=True,
|
||||
help="SkillsBench task name or a path below data/skills-bench/tasks.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--run-dir",
|
||||
help="Run directory below results/compile-pipeline; reuse it to resume an interrupted run.",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _parser().parse_args(argv)
|
||||
try:
|
||||
final_skill = run_pipeline(
|
||||
harness=args.harness, model=args.model, external_model=args.external_model,
|
||||
task=args.task, run_dir=args.run_dir,
|
||||
)
|
||||
except (OSError, RuntimeError, ValueError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
print(final_skill)
|
||||
return 0
|
||||
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,166 @@
|
||||
"""BenchFlow 评测执行、完整性判断与恢复。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any
|
||||
import uuid
|
||||
|
||||
from scripts.dynamic_compile.fast.storage import sha256_file
|
||||
|
||||
from .manifest import read_json_object
|
||||
from .paths import PROJECT_ROOT
|
||||
|
||||
|
||||
EVALUATION_SCRIPT = PROJECT_ROOT / "scripts" / "evaluate" / "run-raw-task.sh"
|
||||
MAX_PARALLEL = 3
|
||||
|
||||
|
||||
def evaluation_artifacts_complete(test_dir: Path) -> bool:
|
||||
"""Return whether one BenchFlow attempt has all required usable artifacts."""
|
||||
summary = read_json_object(test_dir / "summary.json")
|
||||
required_skill = read_json_object(test_dir / "required-skill.json")
|
||||
if summary is None or required_skill is None:
|
||||
return False
|
||||
try:
|
||||
total = int(summary.get("total", 0) or 0)
|
||||
passed = int(summary.get("passed", summary.get("pass", 0)) or 0)
|
||||
failed = int(summary.get("failed", summary.get("fail", 0)) or 0)
|
||||
errored = int(summary.get("errored", summary.get("error", 0)) or 0)
|
||||
verifier_errored = int(summary.get("verifier_errored", 0) or 0)
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
if total != 1 or passed + failed != 1 or errored or verifier_errored:
|
||||
return False
|
||||
if required_skill.get("invoked") is not True or required_skill.get("parse_errors"):
|
||||
return False
|
||||
trajectories = sorted(test_dir.rglob("acp_trajectory.jsonl"))
|
||||
canonical = [path for path in trajectories if "trajectory" in path.parts]
|
||||
trajectory = canonical[0] if canonical else (trajectories[0] if trajectories else None)
|
||||
if trajectory is None:
|
||||
return False
|
||||
result = read_json_object(trajectory.parent.parent / "result.json")
|
||||
if result is None:
|
||||
return False
|
||||
try:
|
||||
events = [
|
||||
json.loads(line)
|
||||
for line in trajectory.read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return False
|
||||
return bool(events) and all(isinstance(event, dict) for event in events)
|
||||
|
||||
|
||||
def completed_evaluation_count(output: Path) -> int:
|
||||
if not output.is_dir():
|
||||
return 0
|
||||
return sum(
|
||||
evaluation_artifacts_complete(test_dir)
|
||||
for test_dir in output.glob("test-*")
|
||||
if test_dir.is_dir()
|
||||
)
|
||||
|
||||
|
||||
def quarantine_incomplete_evaluations(output: Path) -> int:
|
||||
if not output.is_dir():
|
||||
return 0
|
||||
incomplete = [
|
||||
test_dir
|
||||
for test_dir in sorted(output.glob("test-*"))
|
||||
if test_dir.is_dir() and not evaluation_artifacts_complete(test_dir)
|
||||
]
|
||||
if not incomplete:
|
||||
return 0
|
||||
quarantine = output / ".incomplete"
|
||||
quarantine.mkdir(exist_ok=True)
|
||||
for test_dir in incomplete:
|
||||
destination = quarantine / test_dir.name
|
||||
if destination.exists():
|
||||
destination = quarantine / f"{test_dir.name}-{uuid.uuid4().hex[:6]}"
|
||||
test_dir.replace(destination)
|
||||
print(
|
||||
f"[compile-pipeline] preserved incomplete evaluation at {destination}",
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
return len(incomplete)
|
||||
|
||||
|
||||
def evaluate(
|
||||
*,
|
||||
harness: str,
|
||||
model: str,
|
||||
task_dir: Path,
|
||||
skill_source: Path,
|
||||
output: Path,
|
||||
repeat: int,
|
||||
) -> None:
|
||||
completed = completed_evaluation_count(output)
|
||||
if completed >= repeat:
|
||||
print(
|
||||
f"[compile-pipeline] evaluation complete: {output} ({completed}/{repeat}); skipping",
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
return
|
||||
quarantine_incomplete_evaluations(output)
|
||||
missing = repeat - completed
|
||||
if completed:
|
||||
print(
|
||||
f"[compile-pipeline] evaluation incomplete: {output} "
|
||||
f"({completed}/{repeat}); running {missing} missing rollout(s)",
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
command = [
|
||||
"bash", str(EVALUATION_SCRIPT), "--harness", harness, "--model", model,
|
||||
"--task", str(task_dir), "--skill-source", str(skill_source), "--output", str(output),
|
||||
"--require-skill", "--repeat", str(missing), "--max-parallel", str(MAX_PARALLEL),
|
||||
]
|
||||
try:
|
||||
subprocess.run(command, cwd=PROJECT_ROOT, check=True)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise RuntimeError(
|
||||
f"evaluation failed for {skill_source} with exit code {exc.returncode}"
|
||||
) from exc
|
||||
completed = completed_evaluation_count(output)
|
||||
if completed < repeat:
|
||||
raise RuntimeError(
|
||||
f"evaluation produced only {completed}/{repeat} complete rollout artifacts "
|
||||
f"under {output}"
|
||||
)
|
||||
|
||||
|
||||
def complete_deep_skill(output: Path) -> Path | None:
|
||||
state = read_json_object(output / "run.json")
|
||||
report = read_json_object(output / "report.json")
|
||||
skill = output / "S_final"
|
||||
if (
|
||||
state is not None and state.get("status") == "complete"
|
||||
and report is not None and report.get("status") == "complete"
|
||||
and (skill / "SKILL.md").is_file()
|
||||
):
|
||||
return skill
|
||||
return None
|
||||
|
||||
|
||||
def deep_final_rollouts(output: Path, final_skill: Path, repeat: int) -> Path | None:
|
||||
report = read_json_object(output / "report.json")
|
||||
if report is None:
|
||||
return None
|
||||
value = report.get("final_rollouts")
|
||||
expected_hash = report.get("final_rollout_skill_sha256")
|
||||
if not isinstance(value, str) or not isinstance(expected_hash, str):
|
||||
return None
|
||||
rollouts = Path(value).resolve()
|
||||
if (
|
||||
not rollouts.is_dir() or expected_hash != sha256_file(final_skill / "SKILL.md")
|
||||
or completed_evaluation_count(rollouts) < repeat
|
||||
):
|
||||
return None
|
||||
return rollouts
|
||||
@@ -0,0 +1,78 @@
|
||||
"""可恢复运行的原子清单持久化。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.storage import atomic_write_json
|
||||
|
||||
|
||||
def read_json_object(path: Path) -> dict[str, Any] | None:
|
||||
try:
|
||||
value = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return None
|
||||
return value if isinstance(value, dict) else None
|
||||
|
||||
|
||||
class RunManifest:
|
||||
def __init__(
|
||||
self,
|
||||
path: Path,
|
||||
*,
|
||||
harness: str,
|
||||
model: str,
|
||||
external_model: str,
|
||||
task_dir: Path,
|
||||
):
|
||||
self.path = path
|
||||
inputs = {
|
||||
"harness": harness,
|
||||
"model": model,
|
||||
"external_model": external_model,
|
||||
"task": str(task_dir),
|
||||
}
|
||||
if path.is_file():
|
||||
existing = read_json_object(path)
|
||||
if existing is None:
|
||||
raise ValueError(f"invalid run manifest: {path}")
|
||||
if existing.get("schema_version") != "1.0":
|
||||
raise ValueError(f"unsupported run manifest schema: {path}")
|
||||
if existing.get("inputs") != inputs:
|
||||
raise ValueError(
|
||||
f"run directory belongs to different pipeline inputs: {path.parent}"
|
||||
)
|
||||
stages = existing.get("stages")
|
||||
if not isinstance(stages, dict):
|
||||
raise ValueError(f"invalid stages in run manifest: {path}")
|
||||
self.value = existing
|
||||
self.value["status"] = "running"
|
||||
self.value.pop("error", None)
|
||||
self.save()
|
||||
return
|
||||
self.value: dict[str, Any] = {
|
||||
"schema_version": "1.0",
|
||||
"status": "running",
|
||||
"inputs": inputs,
|
||||
"stages": {},
|
||||
}
|
||||
self.save()
|
||||
|
||||
def save(self) -> None:
|
||||
atomic_write_json(self.path, self.value)
|
||||
|
||||
def stage(self, name: str, status: str, **details: Any) -> None:
|
||||
self.value["stages"][name] = {"status": status, **details}
|
||||
self.save()
|
||||
|
||||
def complete(self, final_skill: Path) -> None:
|
||||
self.value["status"] = "complete"
|
||||
self.value["final_skill"] = str(final_skill)
|
||||
self.save()
|
||||
|
||||
def fail(self, error: Exception) -> None:
|
||||
self.value["status"] = "failed"
|
||||
self.value["error"] = f"{type(error).__name__}: {error}"
|
||||
self.save()
|
||||
@@ -0,0 +1,60 @@
|
||||
"""项目路径解析与输入范围校验。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import uuid
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
||||
TASKS_ROOT = PROJECT_ROOT / "data" / "skills-bench" / "tasks"
|
||||
RESULTS_ROOT = PROJECT_ROOT / "results" / "compile-pipeline"
|
||||
|
||||
|
||||
def task_directory(value: str) -> Path:
|
||||
requested = Path(value).expanduser()
|
||||
if requested.is_absolute():
|
||||
candidate = requested.resolve()
|
||||
elif len(requested.parts) == 1:
|
||||
candidate = (TASKS_ROOT / requested).resolve()
|
||||
else:
|
||||
candidate = (PROJECT_ROOT / requested).resolve()
|
||||
try:
|
||||
candidate.relative_to(TASKS_ROOT.resolve())
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"--task must resolve below {TASKS_ROOT}") from exc
|
||||
if not (candidate / "task.md").is_file():
|
||||
raise ValueError(f"not a SkillsBench task directory: {candidate}")
|
||||
skills = candidate / "environment" / "skills"
|
||||
if not skills.is_dir():
|
||||
raise ValueError(f"SkillsBench task has no environment/skills directory: {candidate}")
|
||||
return candidate
|
||||
|
||||
|
||||
def single_skill_source(task_dir: Path) -> Path:
|
||||
skills_root = task_dir / "environment" / "skills"
|
||||
skill_files = sorted(skills_root.rglob("SKILL.md"))
|
||||
if len(skill_files) != 1:
|
||||
raise ValueError(
|
||||
"the complete pipeline currently requires exactly one task Skill because "
|
||||
f"Fast and Deep accept one Skill package; found {len(skill_files)} under {skills_root}"
|
||||
)
|
||||
return skills_root
|
||||
|
||||
|
||||
def requested_run_root(value: str | Path | None) -> Path:
|
||||
if value is None:
|
||||
run_id = datetime.now().strftime("%Y%m%d-%H%M%S") + "-" + uuid.uuid4().hex[:6]
|
||||
return (RESULTS_ROOT / run_id).resolve()
|
||||
requested = Path(value).expanduser()
|
||||
candidate = (
|
||||
requested.resolve()
|
||||
if requested.is_absolute()
|
||||
else (PROJECT_ROOT / requested).resolve()
|
||||
)
|
||||
try:
|
||||
candidate.relative_to(RESULTS_ROOT.resolve())
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"--run-dir must resolve below {RESULTS_ROOT}") from exc
|
||||
return candidate
|
||||
@@ -0,0 +1,195 @@
|
||||
"""完整静态、Fast 与 Deep 技能编译流水线。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.deep.pipeline import DeepLoop
|
||||
from scripts.dynamic_compile.fast.pipeline import run_pipeline as run_fast_pipeline
|
||||
from scripts.static_compile.compiler.compiler import compile_input
|
||||
from scripts.static_compile.profile_generation.pipeline import ensure_profile
|
||||
|
||||
from .evaluation import (
|
||||
MAX_PARALLEL,
|
||||
complete_deep_skill,
|
||||
deep_final_rollouts,
|
||||
evaluate,
|
||||
)
|
||||
from .manifest import RunManifest, read_json_object
|
||||
from .paths import requested_run_root, single_skill_source, task_directory
|
||||
|
||||
|
||||
ORIGINAL_ROLLOUTS = 6
|
||||
STATIC_ROLLOUTS = 6
|
||||
FAST_ROLLOUTS = 3
|
||||
FINAL_ROLLOUTS = 3
|
||||
|
||||
|
||||
def _complete_static_skill(output: Path) -> tuple[Path, str] | None:
|
||||
complete: list[tuple[Path, str]] = []
|
||||
if not output.is_dir():
|
||||
return None
|
||||
for report_path in output.rglob("rewrite-report.json"):
|
||||
skill = report_path.parent
|
||||
report = read_json_object(report_path)
|
||||
if report is None or not (skill / "SKILL.md").is_file():
|
||||
continue
|
||||
status = str(report.get("status", "failed"))
|
||||
if status not in {"failed", "rolled_back"}:
|
||||
complete.append((skill, status))
|
||||
return complete[0] if len(complete) == 1 else None
|
||||
|
||||
|
||||
def run_pipeline(
|
||||
*,
|
||||
harness: str,
|
||||
model: str,
|
||||
external_model: str,
|
||||
task: str,
|
||||
run_dir: str | Path | None = None,
|
||||
) -> Path:
|
||||
task_dir = task_directory(task)
|
||||
source_skills = single_skill_source(task_dir)
|
||||
task_name = task_dir.name
|
||||
run_root = requested_run_root(run_dir)
|
||||
if (
|
||||
run_root.is_dir()
|
||||
and any(run_root.iterdir())
|
||||
and not (run_root / "manifest.json").is_file()
|
||||
):
|
||||
raise ValueError(
|
||||
f"non-empty run directory has no manifest and cannot be resumed: {run_root}"
|
||||
)
|
||||
run_root.mkdir(parents=True, exist_ok=True)
|
||||
manifest = RunManifest(
|
||||
run_root / "manifest.json",
|
||||
harness=harness,
|
||||
model=model,
|
||||
external_model=external_model,
|
||||
task_dir=task_dir,
|
||||
)
|
||||
artifacts = run_root / "artifacts"
|
||||
trace_task_root = run_root / "traces" / task_name
|
||||
original_traces = trace_task_root / "ori_skill"
|
||||
static_root = artifacts / "static"
|
||||
static_traces = trace_task_root / "model_skill"
|
||||
fast_score = artifacts / "fast" / "score"
|
||||
fast_output = artifacts / "fast" / "optimization"
|
||||
fast_traces = trace_task_root / "fast_skill"
|
||||
deep_output = artifacts / "deep"
|
||||
final_traces = trace_task_root / "final_skill"
|
||||
|
||||
print(f"[compile-pipeline] run directory: {run_root}", file=sys.stderr, flush=True)
|
||||
try:
|
||||
manifest.stage("original_evaluation", "running", output=str(original_traces))
|
||||
evaluate(
|
||||
harness=harness, model=model, task_dir=task_dir, skill_source=source_skills,
|
||||
output=original_traces, repeat=ORIGINAL_ROLLOUTS,
|
||||
)
|
||||
manifest.stage(
|
||||
"original_evaluation", "complete", output=str(original_traces),
|
||||
rollouts=ORIGINAL_ROLLOUTS, max_parallel=MAX_PARALLEL,
|
||||
)
|
||||
|
||||
manifest.stage("profile", "running")
|
||||
profile_path, generated = ensure_profile(model)
|
||||
manifest.stage("profile", "complete", path=str(profile_path), generated=generated)
|
||||
|
||||
cached_static = _complete_static_skill(static_root)
|
||||
if cached_static is not None:
|
||||
static_skill, static_status = cached_static
|
||||
print(
|
||||
f"[compile-pipeline] static compilation complete: {static_skill}; skipping",
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
else:
|
||||
manifest.stage("static_compile", "running")
|
||||
static_results = compile_input(
|
||||
source_skills, profile_path, static_root, mode="hybrid",
|
||||
annotator_model=external_model, force=static_root.exists(),
|
||||
)
|
||||
if len(static_results) != 1 or static_results[0].output_dir is None:
|
||||
raise RuntimeError("static compilation did not produce exactly one Skill package")
|
||||
static_status = str(static_results[0].report.get("status", "failed"))
|
||||
if static_status in {"failed", "rolled_back"}:
|
||||
raise RuntimeError(f"static compilation ended with status {static_status}")
|
||||
static_skill = static_results[0].output_dir
|
||||
manifest.stage(
|
||||
"static_compile", "complete", skill=str(static_skill), compile_status=static_status,
|
||||
)
|
||||
|
||||
manifest.stage("static_evaluation", "running", output=str(static_traces))
|
||||
evaluate(
|
||||
harness=harness, model=model, task_dir=task_dir, skill_source=static_skill,
|
||||
output=static_traces, repeat=STATIC_ROLLOUTS,
|
||||
)
|
||||
manifest.stage(
|
||||
"static_evaluation", "complete", output=str(static_traces),
|
||||
rollouts=STATIC_ROLLOUTS, max_parallel=MAX_PARALLEL,
|
||||
)
|
||||
|
||||
manifest.stage("fast_compile", "running")
|
||||
fast_skill = run_fast_pipeline(
|
||||
static_traces, static_skill, score_output=fast_score, output=fast_output,
|
||||
model=external_model, max_parallel=MAX_PARALLEL,
|
||||
)
|
||||
manifest.stage("fast_compile", "complete", skill=str(fast_skill))
|
||||
|
||||
manifest.stage("fast_evaluation", "running", output=str(fast_traces))
|
||||
evaluate(
|
||||
harness=harness, model=model, task_dir=task_dir, skill_source=fast_skill,
|
||||
output=fast_traces, repeat=FAST_ROLLOUTS,
|
||||
)
|
||||
manifest.stage(
|
||||
"fast_evaluation", "complete", output=str(fast_traces),
|
||||
rollouts=FAST_ROLLOUTS, max_parallel=MAX_PARALLEL,
|
||||
)
|
||||
|
||||
final_skill = complete_deep_skill(deep_output)
|
||||
if final_skill is not None:
|
||||
print(
|
||||
f"[compile-pipeline] deep compilation complete: {final_skill}; skipping",
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
else:
|
||||
manifest.stage("deep_compile", "running", output=str(deep_output))
|
||||
deep_loop = (
|
||||
DeepLoop(deep_output)
|
||||
if (deep_output / "run.json").is_file()
|
||||
else DeepLoop.create(fast_skill, fast_traces, deep_output, model=external_model)
|
||||
)
|
||||
final_skill = deep_loop.drive()
|
||||
if complete_deep_skill(deep_output) is None:
|
||||
raise RuntimeError(
|
||||
f"deep compilation did not produce complete artifacts under {deep_output}"
|
||||
)
|
||||
manifest.stage("deep_compile", "complete", skill=str(final_skill))
|
||||
|
||||
reusable_rollouts = deep_final_rollouts(
|
||||
deep_output, final_skill, FINAL_ROLLOUTS,
|
||||
)
|
||||
if reusable_rollouts is not None:
|
||||
print(
|
||||
f"[compile-pipeline] copying Deep final rollouts to: {final_traces}",
|
||||
file=sys.stderr, flush=True,
|
||||
)
|
||||
shutil.copytree(reusable_rollouts, final_traces, dirs_exist_ok=True)
|
||||
final_evaluation_output = final_traces
|
||||
manifest.stage("final_evaluation", "running", output=str(final_evaluation_output))
|
||||
evaluate(
|
||||
harness=harness, model=model, task_dir=task_dir, skill_source=final_skill,
|
||||
output=final_evaluation_output, repeat=FINAL_ROLLOUTS,
|
||||
)
|
||||
manifest.stage(
|
||||
"final_evaluation", "complete", output=str(final_evaluation_output),
|
||||
rollouts=FINAL_ROLLOUTS, max_parallel=MAX_PARALLEL,
|
||||
reused_deep_rollouts=reusable_rollouts is not None,
|
||||
)
|
||||
manifest.complete(final_skill)
|
||||
return final_skill
|
||||
except Exception as exc:
|
||||
manifest.fail(exc)
|
||||
raise
|
||||
Reference in New Issue
Block a user