Initial commit
This commit is contained in:
@@ -0,0 +1,53 @@
|
||||
"""Deep 编译流水线的唯一命令行入口。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.provider_router import parse_model_reference
|
||||
|
||||
from .pipeline import DeepLoop
|
||||
|
||||
|
||||
def _provider_model(value: str) -> str:
|
||||
try:
|
||||
return parse_model_reference(value).value
|
||||
except ValueError as exc:
|
||||
raise argparse.ArgumentTypeError(str(exc)) from exc
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(prog="python -m scripts.dynamic_compile.deep")
|
||||
commands = parser.add_subparsers(dest="command", required=True)
|
||||
run = commands.add_parser("run", help="run Deep Loop from a skill and its BenchFlow traces")
|
||||
run.add_argument("--skill", type=Path, required=True)
|
||||
run.add_argument("--traces", type=Path, required=True)
|
||||
run.add_argument("--output", type=Path)
|
||||
run.add_argument(
|
||||
"--model",
|
||||
required=True,
|
||||
type=_provider_model,
|
||||
help="all external model calls use this provider/model",
|
||||
)
|
||||
resume = commands.add_parser("resume", help="resume a Deep Loop run")
|
||||
resume.add_argument("--run", type=Path, required=True)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _parser().parse_args(argv)
|
||||
try:
|
||||
loop = (
|
||||
DeepLoop.create(args.skill, args.traces, args.output, model=args.model)
|
||||
if args.command == "run"
|
||||
else DeepLoop(args.run)
|
||||
)
|
||||
print(loop.drive())
|
||||
return 0
|
||||
except (OSError, ValueError, RuntimeError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,216 @@
|
||||
"""BenchFlow 输入与运行时轨迹适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.traces.benchflow import trajectory_for_test
|
||||
from scripts.dynamic_compile.fast.scoring.state import acp_events_to_state, read_acp_events
|
||||
|
||||
|
||||
@dataclass
|
||||
class BenchFlowInput:
|
||||
task_name: str
|
||||
task_dir: Path
|
||||
agent: str
|
||||
model: str
|
||||
prompt: str
|
||||
traces: list[RolloutTrace]
|
||||
|
||||
|
||||
def _project_root() -> Path:
|
||||
return Path(__file__).resolve().parents[4]
|
||||
|
||||
|
||||
def _run_config(test_dir: Path) -> dict[str, Any]:
|
||||
paths = sorted(test_dir.rglob("config.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow run config under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError(f"invalid BenchFlow run config: {paths[0]}")
|
||||
return value
|
||||
|
||||
|
||||
def _prompt(test_dir: Path) -> str:
|
||||
paths = sorted(test_dir.rglob("prompts.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow prompts.json under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, list) or not value or not isinstance(value[0], str):
|
||||
raise ValueError(f"invalid BenchFlow prompts: {paths[0]}")
|
||||
return value[0].strip()
|
||||
|
||||
|
||||
def load_runtime_trace(test_dir: Path, task_name: str, compile_type: str) -> RolloutTrace:
|
||||
"""Load agent runtime evidence without opening verifier/result artifacts."""
|
||||
trajectory = trajectory_for_test(test_dir)
|
||||
if trajectory is None:
|
||||
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
|
||||
events = read_acp_events(trajectory)
|
||||
state = acp_events_to_state(events)
|
||||
skill_invoked = any(
|
||||
event.get("type") == "tool_call"
|
||||
and event.get("status") == "completed"
|
||||
and any(
|
||||
str(event.get(field, "")).strip().lower() == "skill"
|
||||
for field in ("title", "kind")
|
||||
)
|
||||
for event in events
|
||||
)
|
||||
timed_out = any(event.get("type") == "agent_timeout" for event in events)
|
||||
return RolloutTrace(
|
||||
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
|
||||
task_name=task_name,
|
||||
compile_type=compile_type,
|
||||
test_name=test_dir.name,
|
||||
state=state,
|
||||
skill_invoked=skill_invoked,
|
||||
timed_out=timed_out,
|
||||
metadata={
|
||||
"termination": "timeout" if timed_out else "completed",
|
||||
"tool_calls": sum(event.get("type") == "tool_call" for event in events),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def load_benchflow_input(source: Path) -> BenchFlowInput:
|
||||
source = source.resolve()
|
||||
tests = sorted(path for path in source.glob("test-*") if path.is_dir())[-5:]
|
||||
if not tests:
|
||||
raise ValueError(f"no test-* BenchFlow traces under {source}")
|
||||
task_name = source.parent.name
|
||||
traces = [load_runtime_trace(test, task_name, "custom") for test in tests]
|
||||
configs = [_run_config(test) for test in tests]
|
||||
agents = {str(item.get("agent", "")).strip() for item in configs}
|
||||
models = {str(item.get("model", "")).strip() for item in configs}
|
||||
if "" in agents or len(agents) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one agent: {sorted(agents)}")
|
||||
if "" in models or len(models) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one model: {sorted(models)}")
|
||||
prompts = {_prompt(test) for test in tests}
|
||||
if len(prompts) != 1:
|
||||
raise ValueError(f"BenchFlow traces contain {len(prompts)} different task prompts")
|
||||
task_dir = _project_root() / "data" / "skills-bench" / "tasks" / task_name
|
||||
if not (task_dir / "task.md").is_file():
|
||||
raise ValueError(f"cannot resolve SkillsBench task directory: {task_dir}")
|
||||
return BenchFlowInput(
|
||||
task_name, task_dir, next(iter(agents)), next(iter(models)),
|
||||
next(iter(prompts)), traces,
|
||||
)
|
||||
|
||||
|
||||
def _slug(value: str) -> str:
|
||||
return re.sub(r"^-+|-+$", "", re.sub(r"[^a-z0-9]+", "-", value.lower()))
|
||||
|
||||
|
||||
class SkillsBenchDevelopmentRolloutRunner:
|
||||
"""Development-only runner; verifier output is never projected into traces."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
context: BenchFlowInput,
|
||||
work_root: Path,
|
||||
max_parallel: int = 3,
|
||||
archive_root: Path | None = None,
|
||||
):
|
||||
self.context = context
|
||||
self.work_root = work_root
|
||||
self.max_parallel = max_parallel
|
||||
self.archive_root = archive_root
|
||||
|
||||
def _variant_dir(self, jobs_root: Path) -> Path:
|
||||
return (
|
||||
jobs_root
|
||||
/ _slug(f"{self.context.agent}-{self.context.model}")
|
||||
/ self.context.task_name
|
||||
/ "custom_skill"
|
||||
)
|
||||
|
||||
def _completed_traces(self, variant: Path, batch_id: str) -> list[RolloutTrace]:
|
||||
completed: list[RolloutTrace] = []
|
||||
for test in sorted(path for path in variant.glob("test-*") if path.is_dir()):
|
||||
try:
|
||||
requirement = json.loads(
|
||||
(test / "required-skill.json").read_text(encoding="utf-8")
|
||||
)
|
||||
if requirement.get("invoked") is not True:
|
||||
continue
|
||||
trace = load_runtime_trace(test, self.context.task_name, batch_id)
|
||||
except (AttributeError, json.JSONDecodeError, OSError, ValueError):
|
||||
continue
|
||||
completed.append(trace)
|
||||
return completed
|
||||
|
||||
def artifacts_dir(self, batch_id: str) -> Path | None:
|
||||
if self.archive_root is None:
|
||||
return None
|
||||
return self.archive_root / batch_id / "custom_skill"
|
||||
|
||||
def run_batch(
|
||||
self,
|
||||
skill_package: Path,
|
||||
_prompt: str,
|
||||
batch_id: str,
|
||||
_task_name: str,
|
||||
count: int,
|
||||
progress: Any | None = None,
|
||||
) -> list[RolloutTrace]:
|
||||
jobs_root = self.work_root / batch_id
|
||||
log_path = jobs_root / "runner.log"
|
||||
variant = self._variant_dir(jobs_root)
|
||||
archive = self.artifacts_dir(batch_id)
|
||||
if archive is not None and archive.is_dir():
|
||||
archived_traces = self._completed_traces(archive, batch_id)
|
||||
if len(archived_traces) >= count:
|
||||
traces = archived_traces
|
||||
missing = 0
|
||||
else:
|
||||
variant.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(archive, variant, dirs_exist_ok=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
else:
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
if missing:
|
||||
jobs_root.mkdir(parents=True, exist_ok=True)
|
||||
command = [
|
||||
"bash", str(_project_root() / "scripts" / "evaluate" / "run-raw-task.sh"),
|
||||
"--harness", self.context.agent,
|
||||
"--model", self.context.model,
|
||||
"--task", str(self.context.task_dir),
|
||||
"--skill-source", str(skill_package.resolve()),
|
||||
"--require-skill",
|
||||
"--repeat", str(missing),
|
||||
"--max-parallel", str(self.max_parallel),
|
||||
"--output", str(variant),
|
||||
]
|
||||
with log_path.open("a", encoding="utf-8") as handle:
|
||||
subprocess.run(command, stdout=handle, stderr=subprocess.STDOUT, text=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
if archive is not None and variant.is_dir():
|
||||
archive.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(variant, archive, dirs_exist_ok=True)
|
||||
runner_log = jobs_root / "runner.log"
|
||||
if runner_log.is_file():
|
||||
shutil.copy2(runner_log, archive.parent / "runner.log")
|
||||
traces = self._completed_traces(archive, batch_id)
|
||||
if len(traces) < count:
|
||||
raise RuntimeError(
|
||||
f"BenchFlow produced {len(traces)}/{count} candidate traces; see {log_path}"
|
||||
)
|
||||
traces = traces[:count]
|
||||
for index, trace in enumerate(traces, 1):
|
||||
trace.trace_id = f"{batch_id}-{index:03d}"
|
||||
trace.test_name = trace.trace_id
|
||||
if progress is not None:
|
||||
progress(index, len(traces), trace.trace_id)
|
||||
return traces
|
||||
@@ -0,0 +1,197 @@
|
||||
"""语义模型评分与局部编辑适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
|
||||
|
||||
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
|
||||
|
||||
|
||||
RUBRICS = {
|
||||
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
|
||||
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
|
||||
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
|
||||
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
|
||||
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
|
||||
}
|
||||
|
||||
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
|
||||
|
||||
|
||||
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
|
||||
return [
|
||||
value for value in values
|
||||
if isinstance(value, str)
|
||||
and _EVIDENCE.fullmatch(value)
|
||||
and value.startswith(prefixes)
|
||||
]
|
||||
|
||||
|
||||
class DeepAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
@staticmethod
|
||||
def _trace_payload(traces: list[RolloutTrace]) -> str:
|
||||
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
|
||||
|
||||
def score_column(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, CellScore]:
|
||||
unit_payload = [
|
||||
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
|
||||
for unit in units
|
||||
]
|
||||
result = self.client.json(
|
||||
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
|
||||
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Current units:
|
||||
{json.dumps(unit_payload, ensure_ascii=False)}
|
||||
Agent traces:
|
||||
{self._trace_payload(traces)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
|
||||
raise ValueError(f"judge returned an invalid {dimension} column")
|
||||
expected = {unit.unit_id for unit in units}
|
||||
column: dict[str, CellScore] = {}
|
||||
for item in result["scores"]:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("judge score entries must be objects")
|
||||
unit_id = str(item.get("unit_id", ""))
|
||||
score = float(item.get("score"))
|
||||
evidence = item.get("evidence", [])
|
||||
if unit_id not in expected or unit_id in column:
|
||||
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
|
||||
evidence = valid_evidence(evidence, traces)
|
||||
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
|
||||
if set(column) != expected:
|
||||
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
|
||||
return column
|
||||
|
||||
def compare_cell(
|
||||
self,
|
||||
task: str,
|
||||
incumbent_unit: SkillUnit,
|
||||
candidate_unit: SkillUnit,
|
||||
incumbent_traces: list[RolloutTrace],
|
||||
candidate_traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Compare one incumbent and candidate skill unit. Return JSON only.",
|
||||
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Incumbent unit:
|
||||
{incumbent_unit.text}
|
||||
Candidate unit:
|
||||
{candidate_unit.text}
|
||||
Incumbent runtime facts:
|
||||
{self._trace_payload(incumbent_traces)}
|
||||
Candidate runtime facts:
|
||||
{self._trace_payload(candidate_traces)}""",
|
||||
)
|
||||
if not isinstance(result.get("task_relevant"), bool):
|
||||
raise ValueError("judge returned invalid task relevance")
|
||||
incumbent_score = float(result.get("incumbent_score"))
|
||||
candidate_score = float(result.get("candidate_score"))
|
||||
for score in (incumbent_score, candidate_score):
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid paired score: {score}")
|
||||
regressions = result.get("runtime_regressions")
|
||||
allowed = {
|
||||
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
|
||||
}
|
||||
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
|
||||
raise ValueError("judge returned invalid runtime regressions")
|
||||
return {
|
||||
"task_relevant": result["task_relevant"],
|
||||
"incumbent_score": incumbent_score,
|
||||
"candidate_score": candidate_score,
|
||||
"candidate_evidence": valid_evidence(
|
||||
result.get("candidate_evidence"), candidate_traces
|
||||
),
|
||||
"runtime_regressions": list(dict.fromkeys(regressions)),
|
||||
"reason": str(result.get("reason", "")),
|
||||
}
|
||||
|
||||
def score_matrix(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
level: str,
|
||||
existing_columns: dict[str, dict[str, CellScore]] | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> ScoreMatrix:
|
||||
columns = dict(existing_columns or {})
|
||||
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {
|
||||
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
|
||||
for dimension in missing
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
dimension = futures[future]
|
||||
columns[dimension] = future.result()
|
||||
if result_callback is not None:
|
||||
result_callback(dimension, columns[dimension])
|
||||
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
|
||||
|
||||
def generate_edit(
|
||||
self,
|
||||
coordinate: Coordinate,
|
||||
unit: SkillUnit,
|
||||
cell: CellScore,
|
||||
rejected: list[dict[str, Any]],
|
||||
task_prompt: str,
|
||||
) -> LocalEdit:
|
||||
result = self.client.json(
|
||||
"Generate one bounded local edit for an agent skill. Return JSON only.",
|
||||
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
|
||||
|
||||
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
|
||||
|
||||
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
|
||||
|
||||
Target dimension rubric: {RUBRICS[coordinate.dimension]}
|
||||
Task prompt:
|
||||
{task_prompt}
|
||||
Target unit:
|
||||
{unit.text}
|
||||
Score: {cell.score}
|
||||
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
|
||||
Reason: {cell.reason}
|
||||
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
|
||||
)
|
||||
edit = LocalEdit.from_dict(result)
|
||||
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
|
||||
raise ValueError("edit generator changed the target coordinate")
|
||||
return edit
|
||||
@@ -0,0 +1,195 @@
|
||||
"""Markdown 单元解析与编辑边界校验。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
import re
|
||||
|
||||
from .models import SkillUnit
|
||||
|
||||
|
||||
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
|
||||
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
|
||||
|
||||
|
||||
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
|
||||
rows: list[tuple[int, int, str]] = []
|
||||
offset = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
rows.append((offset, offset + len(line), line))
|
||||
offset += len(line)
|
||||
return rows
|
||||
|
||||
|
||||
def _headings(text: str) -> list[tuple[int, int, int, str]]:
|
||||
found = []
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for start, end, line in _line_offsets(text):
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if fence_char:
|
||||
continue
|
||||
match = _HEADING.match(line)
|
||||
if match:
|
||||
found.append((start, end, len(match.group(1)), match.group(2).strip()))
|
||||
return found
|
||||
|
||||
|
||||
def _frontmatter_end(text: str) -> int:
|
||||
if not text.startswith("---"):
|
||||
return 0
|
||||
lines = text.splitlines(keepends=True)
|
||||
offset = len(lines[0]) if lines else 0
|
||||
for line in lines[1:]:
|
||||
offset += len(line)
|
||||
if line.strip() == "---":
|
||||
return offset
|
||||
return 0
|
||||
|
||||
|
||||
def parse_sections(text: str) -> list[SkillUnit]:
|
||||
headings = _headings(text)
|
||||
body_start = _frontmatter_end(text)
|
||||
body_headings = [item for item in headings if item[0] >= body_start]
|
||||
title = body_headings[0] if body_headings else None
|
||||
after_title = title[1] if title else body_start
|
||||
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
|
||||
if not candidates:
|
||||
body = text[body_start:]
|
||||
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
|
||||
section_depth = min(item[2] for item in candidates)
|
||||
peers = [item for item in candidates if item[2] == section_depth]
|
||||
spans: list[tuple[int, int, str, int | None]] = []
|
||||
preamble = text[after_title:peers[0][0]]
|
||||
if preamble.strip():
|
||||
spans.append((after_title, peers[0][0], "Preamble", section_depth))
|
||||
for index, heading in enumerate(peers):
|
||||
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
|
||||
spans.append((heading[0], end, heading[3], heading[2]))
|
||||
return [
|
||||
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
|
||||
for index, (start, end, heading, depth) in enumerate(spans)
|
||||
]
|
||||
|
||||
|
||||
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
|
||||
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
|
||||
|
||||
|
||||
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
|
||||
replacement = preserve_unit_boundary(unit, new_text)
|
||||
return document[:unit.start] + replacement + document[unit.end:]
|
||||
|
||||
|
||||
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
|
||||
text = section.text
|
||||
base = section.start
|
||||
rows = _line_offsets(text)
|
||||
blocks: list[tuple[int, int]] = []
|
||||
start: int | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for row_start, row_end, line in rows:
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if start is None:
|
||||
start = row_start
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if not fence_char and not line.strip():
|
||||
if start is not None:
|
||||
blocks.append((start, row_start))
|
||||
start = None
|
||||
continue
|
||||
if start is None:
|
||||
start = row_start
|
||||
if start is not None:
|
||||
blocks.append((start, len(text)))
|
||||
merged: list[tuple[int, int]] = []
|
||||
index = 0
|
||||
while index < len(blocks):
|
||||
start, end = blocks[index]
|
||||
block = text[start:end]
|
||||
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
|
||||
merged.append((start, blocks[index + 1][1]))
|
||||
index += 2
|
||||
else:
|
||||
merged.append((start, end))
|
||||
index += 1
|
||||
blocks = merged
|
||||
units = []
|
||||
for index, (start, end) in enumerate(blocks):
|
||||
block = text[start:end]
|
||||
heading_match = next((item for item in _headings(block)), None)
|
||||
units.append(SkillUnit(
|
||||
f"{section.unit_id}.P{index + 1:03d}",
|
||||
"paragraph",
|
||||
section.unit_id,
|
||||
heading_match[3] if heading_match else "",
|
||||
block,
|
||||
index,
|
||||
base + start,
|
||||
base + end,
|
||||
heading_match[2] if heading_match else None,
|
||||
))
|
||||
return units
|
||||
|
||||
|
||||
def fenced_blocks(text: str) -> Counter[str]:
|
||||
blocks: list[str] = []
|
||||
current: list[str] | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
fence = _FENCE.match(line)
|
||||
if current is None:
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
current = [line]
|
||||
continue
|
||||
current.append(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if marker[0] == fence_char and len(marker) >= fence_size:
|
||||
blocks.append("".join(current))
|
||||
current = None
|
||||
fence_char, fence_size = "", 0
|
||||
return Counter(blocks)
|
||||
|
||||
|
||||
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
|
||||
if not new_text.strip() or new_text == unit.text:
|
||||
raise ValueError("local edit must produce non-empty changed text")
|
||||
if unit.level == "section":
|
||||
old_headings = _headings(unit.text)
|
||||
new_headings = _headings(new_text)
|
||||
depth = unit.heading_depth
|
||||
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
|
||||
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("section edit must preserve its peer heading")
|
||||
if fenced_blocks(unit.text) != fenced_blocks(new_text):
|
||||
raise ValueError("section edit must preserve fenced code contents")
|
||||
elif section_depth is not None:
|
||||
old_peers = [
|
||||
(item[2], item[3]) for item in _headings(unit.text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
new_peers = [
|
||||
(item[2], item[3]) for item in _headings(new_text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("paragraph edit must not add or change a section heading")
|
||||
@@ -0,0 +1,134 @@
|
||||
"""领域模型。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
DIMENSIONS = (
|
||||
"Clarity",
|
||||
"Structure",
|
||||
"Executability",
|
||||
"Completeness",
|
||||
"Constraint Salience",
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkillUnit:
|
||||
unit_id: str
|
||||
level: str
|
||||
parent_id: str | None
|
||||
heading: str
|
||||
text: str
|
||||
order: int
|
||||
start: int
|
||||
end: int
|
||||
heading_depth: int | None = None
|
||||
|
||||
@dataclass
|
||||
class CellScore:
|
||||
score: float
|
||||
evidence: list[str]
|
||||
reason: str
|
||||
|
||||
@dataclass
|
||||
class Coordinate:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
normalized_gap: float
|
||||
|
||||
@dataclass
|
||||
class LocalEdit:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
new_text: str
|
||||
edit_summary: str
|
||||
reason: str
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "LocalEdit":
|
||||
required = {"unit_id", "dimension", "new_text", "edit_summary", "reason"}
|
||||
missing = sorted(required - value.keys())
|
||||
if missing:
|
||||
raise ValueError(f"local edit missing fields: {', '.join(missing)}")
|
||||
if not all(isinstance(value[key], str) for key in required):
|
||||
raise ValueError("local edit fields must be strings")
|
||||
return cls(**{key: value[key] for key in cls.__dataclass_fields__})
|
||||
|
||||
|
||||
@dataclass
|
||||
class ScoreMatrix:
|
||||
level: str
|
||||
units: list[SkillUnit]
|
||||
columns: dict[str, dict[str, CellScore]]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"level": self.level,
|
||||
"units": [asdict(unit) for unit in self.units],
|
||||
"columns": {
|
||||
dimension: {unit_id: asdict(cell) for unit_id, cell in column.items()}
|
||||
for dimension, column in self.columns.items()
|
||||
},
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "ScoreMatrix":
|
||||
return cls(
|
||||
level=str(value["level"]),
|
||||
units=[SkillUnit(**item) for item in value["units"]],
|
||||
columns={
|
||||
dimension: {
|
||||
unit_id: CellScore(float(cell["score"]), list(cell["evidence"]), str(cell["reason"]))
|
||||
for unit_id, cell in column.items()
|
||||
}
|
||||
for dimension, column in value["columns"].items()
|
||||
},
|
||||
)
|
||||
|
||||
def unit(self, unit_id: str) -> SkillUnit:
|
||||
return next(unit for unit in self.units if unit.unit_id == unit_id)
|
||||
|
||||
def normalized_gaps(self) -> dict[str, float]:
|
||||
gaps: dict[str, float] = {}
|
||||
for dimension in DIMENSIONS:
|
||||
values = [self.columns[dimension][unit.unit_id].score for unit in self.units]
|
||||
gaps[dimension] = (max(values) - min(values)) / 4.0 if values else 0.0
|
||||
return gaps
|
||||
|
||||
def select_coordinate(
|
||||
self,
|
||||
threshold: float,
|
||||
dimension: str | None = None,
|
||||
excluded: set[tuple[str, str]] | None = None,
|
||||
) -> Coordinate | None:
|
||||
gaps = self.normalized_gaps()
|
||||
excluded = excluded or set()
|
||||
|
||||
def weak_units(item: str) -> list[SkillUnit]:
|
||||
column = self.columns[item]
|
||||
maximum = max((cell.score for cell in column.values()), default=0.0)
|
||||
return [
|
||||
unit for unit in self.units
|
||||
if (unit.unit_id, item) not in excluded
|
||||
and (maximum - column[unit.unit_id].score) / 4.0 > threshold
|
||||
]
|
||||
|
||||
available = [
|
||||
item for item in ([dimension] if dimension else DIMENSIONS)
|
||||
if item is not None and gaps[item] > threshold and weak_units(item)
|
||||
]
|
||||
if not available:
|
||||
return None
|
||||
dimension = max(available, key=lambda item: gaps[item])
|
||||
target = min(
|
||||
weak_units(dimension),
|
||||
key=lambda unit: (self.columns[dimension][unit.unit_id].score, unit.order),
|
||||
)
|
||||
return Coordinate(
|
||||
target.unit_id,
|
||||
dimension,
|
||||
gaps[dimension],
|
||||
)
|
||||
@@ -0,0 +1,826 @@
|
||||
"""Deep 编译流水线编排。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import uuid
|
||||
from dataclasses import asdict, replace
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.storage import (
|
||||
atomic_write_json,
|
||||
atomic_write_jsonl,
|
||||
atomic_write_text,
|
||||
load_json,
|
||||
package_hash,
|
||||
read_jsonl,
|
||||
sha256_file,
|
||||
sha256_text,
|
||||
)
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
|
||||
from .adapters.benchflow import (
|
||||
BenchFlowInput,
|
||||
SkillsBenchDevelopmentRolloutRunner,
|
||||
load_benchflow_input,
|
||||
)
|
||||
from .adapters.semantic import DeepAnalyzer, valid_evidence
|
||||
from .core.models import (
|
||||
CellScore,
|
||||
Coordinate,
|
||||
DIMENSIONS,
|
||||
LocalEdit,
|
||||
ScoreMatrix,
|
||||
SkillUnit,
|
||||
)
|
||||
from .core.markdown import (
|
||||
parse_paragraphs,
|
||||
parse_sections,
|
||||
preserve_unit_boundary,
|
||||
replace_unit_text,
|
||||
validate_edit,
|
||||
)
|
||||
|
||||
|
||||
ROLLOUTS = 3
|
||||
MAX_SECTION_ITERATIONS = 6
|
||||
MAX_PARAGRAPH_ITERATIONS = 3
|
||||
MAX_EDIT_GENERATION_ATTEMPTS = 3
|
||||
GAP_THRESHOLD = 0.375
|
||||
REJECTION_LIMIT = 2
|
||||
MAX_PARALLEL = 3
|
||||
DEFAULT_MODEL = "ali/deepseek-v4-pro-0813"
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"[deep] {message}", file=sys.stderr, flush=True)
|
||||
|
||||
|
||||
def _trace_from_dict(value: dict[str, Any]) -> RolloutTrace:
|
||||
return RolloutTrace(**value)
|
||||
|
||||
|
||||
def _column_to_dict(column: dict[str, CellScore]) -> dict[str, Any]:
|
||||
return {unit_id: asdict(cell) for unit_id, cell in column.items()}
|
||||
|
||||
|
||||
def _column_from_dict(
|
||||
value: dict[str, Any], traces: list[RolloutTrace]
|
||||
) -> dict[str, CellScore]:
|
||||
return {
|
||||
unit_id: CellScore(
|
||||
float(cell["score"]), valid_evidence(cell.get("evidence"), traces), str(cell["reason"])
|
||||
)
|
||||
for unit_id, cell in value.items()
|
||||
}
|
||||
|
||||
|
||||
class DeepLoop:
|
||||
def __init__(
|
||||
self,
|
||||
run_dir: Path,
|
||||
analyzer: DeepAnalyzer | None = None,
|
||||
runner: Any | None = None,
|
||||
):
|
||||
self.run_dir = run_dir.resolve()
|
||||
self.state_path = self.run_dir / "run.json"
|
||||
state = load_json(self.state_path)
|
||||
if not isinstance(state, dict):
|
||||
raise ValueError(f"invalid or missing run state: {self.state_path}")
|
||||
self.state = state
|
||||
self.temp = self.run_dir / ".tmp"
|
||||
self.current = self.temp / "current"
|
||||
self.model = state.get("model", state.get("semantic_model", DEFAULT_MODEL))
|
||||
self.analyzer = analyzer or DeepAnalyzer(
|
||||
SemanticClient(self.model), MAX_PARALLEL,
|
||||
)
|
||||
context = BenchFlowInput(
|
||||
state["task"]["name"],
|
||||
Path(state["task"]["directory"]),
|
||||
state["task"]["agent"],
|
||||
state["task"]["model"],
|
||||
state["task"]["prompt"],
|
||||
[],
|
||||
)
|
||||
self.runner = runner or SkillsBenchDevelopmentRolloutRunner(
|
||||
context,
|
||||
Path(tempfile.gettempdir()) / "skill-compiler-deep" / state["run_id"],
|
||||
MAX_PARALLEL,
|
||||
archive_root=self.run_dir / "rollouts",
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
skill: Path,
|
||||
traces: Path,
|
||||
output: Path | None = None,
|
||||
*,
|
||||
model: str = DEFAULT_MODEL,
|
||||
analyzer: DeepAnalyzer | None = None,
|
||||
runner: Any | None = None,
|
||||
) -> "DeepLoop":
|
||||
skill = skill.resolve()
|
||||
if not (skill / "SKILL.md").is_file():
|
||||
raise ValueError("--skill must be a skill package containing SKILL.md")
|
||||
context = load_benchflow_input(traces)
|
||||
target = (output or skill.parent / f"{skill.name}-deep").resolve()
|
||||
if target.exists() and any(target.iterdir()):
|
||||
raise ValueError(f"deep run directory is not empty: {target}")
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(skill, target / "S_fast")
|
||||
(target / ".tmp").mkdir()
|
||||
(target / "levels").mkdir()
|
||||
shutil.copytree(target / "S_fast", target / ".tmp" / "current")
|
||||
atomic_write_jsonl(target / "input-traces.jsonl", [asdict(trace) for trace in context.traces])
|
||||
state = {
|
||||
"run_id": uuid.uuid4().hex[:12],
|
||||
"status": "created",
|
||||
"model": model,
|
||||
"skill_name": skill.name,
|
||||
"task": {
|
||||
"name": context.task_name,
|
||||
"directory": str(context.task_dir),
|
||||
"agent": context.agent,
|
||||
"model": context.model,
|
||||
"prompt": context.prompt,
|
||||
},
|
||||
"current_rollouts": str(traces.resolve()),
|
||||
"current_rollout_skill_sha256": sha256_file(skill / "SKILL.md"),
|
||||
"levels": {},
|
||||
}
|
||||
atomic_write_json(target / "run.json", state)
|
||||
return cls(target, analyzer=analyzer, runner=runner)
|
||||
|
||||
def _save(self) -> None:
|
||||
atomic_write_json(self.state_path, self.state)
|
||||
|
||||
def _prompt(self) -> str:
|
||||
return str(self.state["task"]["prompt"])
|
||||
|
||||
def _current_text(self) -> str:
|
||||
return (self.current / "SKILL.md").read_text(encoding="utf-8")
|
||||
|
||||
def _load_or_rollout(
|
||||
self,
|
||||
package: Path,
|
||||
trace_path: Path,
|
||||
batch_id: str,
|
||||
seed: list[RolloutTrace] | None = None,
|
||||
) -> list[RolloutTrace]:
|
||||
if trace_path.is_file():
|
||||
return [_trace_from_dict(item) for item in read_jsonl(trace_path)]
|
||||
if seed is not None:
|
||||
traces = seed
|
||||
else:
|
||||
_log(f"Starting {ROLLOUTS} rollouts for {batch_id}")
|
||||
traces = self.runner.run_batch(
|
||||
package,
|
||||
self._prompt(),
|
||||
batch_id,
|
||||
str(self.state["task"]["name"]),
|
||||
ROLLOUTS,
|
||||
progress=lambda done, total, trace_id: _log(
|
||||
f"Rollout {done}/{total} complete: {trace_id}"
|
||||
),
|
||||
)
|
||||
atomic_write_jsonl(trace_path, [asdict(trace) for trace in traces])
|
||||
return traces
|
||||
|
||||
def _load_or_matrix(
|
||||
self,
|
||||
path: Path,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
) -> ScoreMatrix:
|
||||
value = load_json(path)
|
||||
if isinstance(value, dict):
|
||||
return ScoreMatrix.from_dict(value)
|
||||
columns_dir = path.parent / "matrix-columns"
|
||||
cached_columns: dict[str, dict[str, CellScore]] = {}
|
||||
for dimension in DIMENSIONS:
|
||||
cached = load_json(columns_dir / f"{dimension.lower().replace(' ', '-')}.json")
|
||||
if isinstance(cached, dict):
|
||||
cached_columns[dimension] = _column_from_dict(cached, traces)
|
||||
|
||||
def save_column(dimension: str, column: dict[str, CellScore]) -> None:
|
||||
atomic_write_json(
|
||||
columns_dir / f"{dimension.lower().replace(' ', '-')}.json",
|
||||
_column_to_dict(column),
|
||||
)
|
||||
|
||||
_log(f"Scoring full {level} matrix with {self.model}")
|
||||
matrix = self.analyzer.score_matrix(
|
||||
self._current_text(), self._prompt(), units, traces, level,
|
||||
existing_columns=cached_columns,
|
||||
result_callback=save_column,
|
||||
)
|
||||
atomic_write_json(path, matrix.to_dict())
|
||||
return matrix
|
||||
|
||||
def _refresh_matrix(
|
||||
self,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
) -> ScoreMatrix:
|
||||
level_state = self.state["levels"][level]
|
||||
number = int(level_state.get("refreshes", 0)) + 1
|
||||
path = (
|
||||
self.run_dir
|
||||
/ "levels"
|
||||
/ level
|
||||
/ "refreshes"
|
||||
/ f"refresh-{number:02d}"
|
||||
/ "matrix.json"
|
||||
)
|
||||
_log(f"Refreshing full {level} matrix")
|
||||
matrix = self._load_or_matrix(path, level, units, traces)
|
||||
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
|
||||
level_state["refreshes"] = number
|
||||
level_state.pop("active_dimension", None)
|
||||
self._save()
|
||||
return matrix
|
||||
|
||||
def _decisions(self, level: str | None = None) -> list[dict[str, Any]]:
|
||||
roots = (
|
||||
[self.run_dir / "levels" / level]
|
||||
if level else list((self.run_dir / "levels").glob("*"))
|
||||
)
|
||||
decisions = []
|
||||
for root in roots:
|
||||
for path in sorted((root / "iterations").glob("iteration-*/decision.json")):
|
||||
value = load_json(path)
|
||||
if isinstance(value, dict):
|
||||
decisions.append(value)
|
||||
return decisions
|
||||
|
||||
def _rejected(self, coordinate: Coordinate) -> list[dict[str, Any]]:
|
||||
return [
|
||||
decision["rejected_edit"]
|
||||
for decision in self._decisions()
|
||||
if not decision["accepted"]
|
||||
and decision["rejected_edit"]["unit_id"] == coordinate.unit_id
|
||||
and decision["rejected_edit"]["dimension"] == coordinate.dimension
|
||||
]
|
||||
|
||||
def _exhausted(self, level: str) -> set[tuple[str, str]]:
|
||||
counts: dict[tuple[str, str], int] = {}
|
||||
for decision in self._decisions(level):
|
||||
if decision["accepted"]:
|
||||
continue
|
||||
rejected = decision["rejected_edit"]
|
||||
key = (rejected["unit_id"], rejected["dimension"])
|
||||
counts[key] = counts.get(key, 0) + 1
|
||||
return {key for key, count in counts.items() if count >= REJECTION_LIMIT}
|
||||
|
||||
@staticmethod
|
||||
def _block_dimension(level_state: dict[str, Any], dimension: str) -> None:
|
||||
blocked = set(level_state.get("blocked_dimensions", []))
|
||||
blocked.add(dimension)
|
||||
level_state["blocked_dimensions"] = sorted(blocked)
|
||||
|
||||
@staticmethod
|
||||
def _candidate_units(
|
||||
units: list[SkillUnit], target: SkillUnit, new_text: str
|
||||
) -> list[SkillUnit]:
|
||||
delta = len(new_text) - len(target.text)
|
||||
updated = []
|
||||
for unit in units:
|
||||
value = replace(unit)
|
||||
if unit.unit_id == target.unit_id:
|
||||
value.text = new_text
|
||||
value.end = value.start + len(new_text)
|
||||
elif unit.start >= target.end:
|
||||
value.start += delta
|
||||
value.end += delta
|
||||
updated.append(value)
|
||||
return updated
|
||||
|
||||
def _apply_edit(
|
||||
self,
|
||||
unit: SkillUnit,
|
||||
edit: LocalEdit,
|
||||
candidate: Path,
|
||||
section_depth: int | None,
|
||||
) -> None:
|
||||
self._validate_edit_candidate(unit, edit, section_depth)
|
||||
text = self._current_text()
|
||||
changed = replace_unit_text(text, unit, edit.new_text)
|
||||
temporary = candidate.with_name(f".{candidate.name}.{uuid.uuid4().hex}.tmp")
|
||||
shutil.copytree(self.current, temporary)
|
||||
atomic_write_text(temporary / "SKILL.md", changed)
|
||||
if candidate.exists():
|
||||
shutil.rmtree(candidate)
|
||||
candidate.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary.rename(candidate)
|
||||
|
||||
def _validate_edit_candidate(
|
||||
self,
|
||||
unit: SkillUnit,
|
||||
edit: LocalEdit,
|
||||
section_depth: int | None,
|
||||
) -> None:
|
||||
"""Validate a local edit against both unit and whole-document invariants."""
|
||||
|
||||
validate_edit(unit, edit.new_text, section_depth)
|
||||
text = self._current_text()
|
||||
if text[unit.start:unit.end] != unit.text:
|
||||
raise ValueError("target unit no longer matches current SKILL.md")
|
||||
changed = replace_unit_text(text, unit, edit.new_text)
|
||||
before = [(item.heading, item.heading_depth) for item in parse_sections(text)]
|
||||
after = [(item.heading, item.heading_depth) for item in parse_sections(changed)]
|
||||
if before != after:
|
||||
raise ValueError("local edit changed section boundaries")
|
||||
|
||||
@staticmethod
|
||||
def _validation_feedback(
|
||||
attempts: list[dict[str, Any]],
|
||||
) -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"edit_summary": str(item.get("edit_summary", "invalid generated edit")),
|
||||
"reject_reason": f"structural_validation_failed: {item['error']}",
|
||||
"new_text_hash": str(item.get("new_text_hash", "")),
|
||||
}
|
||||
for item in attempts
|
||||
]
|
||||
|
||||
def _valid_edit_or_rejection(
|
||||
self,
|
||||
*,
|
||||
level: str,
|
||||
number: int,
|
||||
iteration_dir: Path,
|
||||
unit: SkillUnit,
|
||||
coordinate: Coordinate,
|
||||
cell: CellScore,
|
||||
section_depth: int | None,
|
||||
) -> tuple[LocalEdit | None, bool, list[dict[str, Any]]]:
|
||||
"""Load or generate a valid edit, feeding structural failures back to the model."""
|
||||
|
||||
edit_path = iteration_dir / "edit.json"
|
||||
attempts_path = iteration_dir / "edit-attempts.json"
|
||||
attempts_value = load_json(attempts_path, [])
|
||||
attempts = attempts_value if isinstance(attempts_value, list) else []
|
||||
cached_value = load_json(edit_path)
|
||||
|
||||
if isinstance(cached_value, dict):
|
||||
try:
|
||||
cached = LocalEdit.from_dict(cached_value)
|
||||
cached.new_text = preserve_unit_boundary(unit, cached.new_text)
|
||||
self._validate_edit_candidate(unit, cached, section_depth)
|
||||
return cached, False, attempts
|
||||
except ValueError as exc:
|
||||
text = str(cached_value.get("new_text", ""))
|
||||
attempts.append({
|
||||
"source": "cached",
|
||||
"error": str(exc),
|
||||
"edit_summary": str(cached_value.get("edit_summary", "")),
|
||||
"new_text_hash": sha256_text(text) if text else "",
|
||||
})
|
||||
atomic_write_json(attempts_path, attempts)
|
||||
_log(
|
||||
f"{level} iteration {number}: cached local edit is invalid: {exc}; "
|
||||
"regenerating"
|
||||
)
|
||||
|
||||
for attempt in range(1, MAX_EDIT_GENERATION_ATTEMPTS + 1):
|
||||
_log(
|
||||
f"{level} iteration {number}: generating local edit for "
|
||||
f"{coordinate.unit_id}/{coordinate.dimension} "
|
||||
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS})"
|
||||
)
|
||||
feedback = self._rejected(coordinate) + self._validation_feedback(attempts)
|
||||
edit: LocalEdit | None = None
|
||||
try:
|
||||
edit = self.analyzer.generate_edit(
|
||||
coordinate,
|
||||
unit,
|
||||
cell,
|
||||
feedback,
|
||||
self._prompt(),
|
||||
)
|
||||
edit.new_text = preserve_unit_boundary(unit, edit.new_text)
|
||||
self._validate_edit_candidate(unit, edit, section_depth)
|
||||
except ValueError as exc:
|
||||
text = edit.new_text if edit is not None else ""
|
||||
attempts.append({
|
||||
"source": "generated",
|
||||
"generation_attempt": attempt,
|
||||
"error": str(exc),
|
||||
"edit_summary": edit.edit_summary if edit is not None else "",
|
||||
"new_text_hash": sha256_text(text) if text else "",
|
||||
})
|
||||
atomic_write_json(attempts_path, attempts)
|
||||
_log(
|
||||
f"{level} iteration {number}: local edit validation failed "
|
||||
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS}): {exc}"
|
||||
)
|
||||
continue
|
||||
|
||||
assert edit is not None
|
||||
atomic_write_json(edit_path, asdict(edit))
|
||||
return edit, True, attempts
|
||||
|
||||
return None, True, attempts
|
||||
|
||||
def _commit_iteration(
|
||||
self,
|
||||
level: str,
|
||||
number: int,
|
||||
iteration_dir: Path,
|
||||
matrix: ScoreMatrix,
|
||||
current_trace_path: Path,
|
||||
) -> ScoreMatrix:
|
||||
decision = load_json(iteration_dir / "decision.json")
|
||||
if not isinstance(decision, dict):
|
||||
raise ValueError("missing iteration decision")
|
||||
if decision["accepted"]:
|
||||
candidate = iteration_dir / "candidate" / self.state["skill_name"]
|
||||
replacement = self.temp / "next-current"
|
||||
shutil.rmtree(replacement, ignore_errors=True)
|
||||
shutil.copytree(candidate, replacement)
|
||||
shutil.rmtree(self.current)
|
||||
replacement.rename(self.current)
|
||||
matrix = ScoreMatrix.from_dict(decision["matrix_after"])
|
||||
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
|
||||
candidate_traces = read_jsonl(iteration_dir / "candidate-traces.jsonl")
|
||||
atomic_write_jsonl(current_trace_path, candidate_traces)
|
||||
rollout_output = decision.get("rollout_output")
|
||||
rollout_skill_sha256 = decision.get("rollout_skill_sha256")
|
||||
if isinstance(rollout_output, str) and isinstance(rollout_skill_sha256, str):
|
||||
self.state["current_rollouts"] = rollout_output
|
||||
self.state["current_rollout_skill_sha256"] = rollout_skill_sha256
|
||||
else:
|
||||
self.state.pop("current_rollouts", None)
|
||||
self.state.pop("current_rollout_skill_sha256", None)
|
||||
level_state = self.state["levels"][level]
|
||||
if int(level_state.get("iterations", 0)) < number:
|
||||
level_state["iterations"] = number
|
||||
self._save()
|
||||
return matrix
|
||||
|
||||
def _run_iteration(
|
||||
self,
|
||||
level: str,
|
||||
number: int,
|
||||
level_dir: Path,
|
||||
matrix: ScoreMatrix,
|
||||
coordinate: Coordinate,
|
||||
current_trace_path: Path,
|
||||
section_depth: int | None,
|
||||
) -> ScoreMatrix:
|
||||
iteration_dir = level_dir / "iterations" / f"iteration-{number:02d}"
|
||||
iteration_dir.mkdir(parents=True, exist_ok=True)
|
||||
unit = matrix.unit(coordinate.unit_id)
|
||||
edit, regenerated, validation_attempts = self._valid_edit_or_rejection(
|
||||
level=level,
|
||||
number=number,
|
||||
iteration_dir=iteration_dir,
|
||||
unit=unit,
|
||||
coordinate=coordinate,
|
||||
cell=matrix.columns[coordinate.dimension][coordinate.unit_id],
|
||||
section_depth=section_depth,
|
||||
)
|
||||
if edit is None:
|
||||
decision = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": False,
|
||||
"reason": "edit_validation_exhausted",
|
||||
"target_delta": 0.0,
|
||||
"validation_attempts": validation_attempts,
|
||||
"rejected_edit": {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": (
|
||||
"Could not generate a structurally valid local edit after "
|
||||
f"{MAX_EDIT_GENERATION_ATTEMPTS} attempts."
|
||||
),
|
||||
"score_change": 0.0,
|
||||
"reject_reason": "edit_validation_exhausted",
|
||||
"new_text_hash": str(
|
||||
validation_attempts[-1].get("new_text_hash", "")
|
||||
) if validation_attempts else "",
|
||||
},
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
_log(
|
||||
f"{level} iteration {number}: local edit validation exhausted; "
|
||||
"recording rejection and continuing"
|
||||
)
|
||||
return self._commit_iteration(
|
||||
level, number, iteration_dir, matrix, current_trace_path
|
||||
)
|
||||
edit_hash = sha256_text(edit.new_text)
|
||||
if any(item["new_text_hash"] == edit_hash for item in self._rejected(coordinate)):
|
||||
decision = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": False,
|
||||
"reason": "exact_duplicate_rejected_edit",
|
||||
"target_delta": 0.0,
|
||||
"rejected_edit": {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": edit.edit_summary,
|
||||
"score_change": 0.0,
|
||||
"reject_reason": "exact_duplicate_rejected_edit",
|
||||
"new_text_hash": edit_hash,
|
||||
},
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
|
||||
candidate = iteration_dir / "candidate" / self.state["skill_name"]
|
||||
if regenerated:
|
||||
shutil.rmtree(candidate, ignore_errors=True)
|
||||
for stale in (
|
||||
iteration_dir / "candidate-traces.jsonl",
|
||||
iteration_dir / "comparison.json",
|
||||
):
|
||||
if stale.exists():
|
||||
stale.unlink()
|
||||
if not (candidate / "SKILL.md").is_file():
|
||||
self._apply_edit(unit, edit, candidate, section_depth)
|
||||
candidate_units = self._candidate_units(matrix.units, unit, edit.new_text)
|
||||
candidate_skill_sha256 = sha256_file(candidate / "SKILL.md")
|
||||
batch_id = (
|
||||
f"{self.state['run_id']}-deep-{level}-i{number:02d}-"
|
||||
f"{candidate_skill_sha256[:12]}"
|
||||
)
|
||||
traces = self._load_or_rollout(
|
||||
candidate,
|
||||
iteration_dir / "candidate-traces.jsonl",
|
||||
batch_id,
|
||||
)
|
||||
comparison_path = iteration_dir / "comparison.json"
|
||||
comparison = load_json(comparison_path)
|
||||
if not isinstance(comparison, dict):
|
||||
_log(
|
||||
f"{level} iteration {number}: comparing "
|
||||
f"{coordinate.unit_id}/{coordinate.dimension}"
|
||||
)
|
||||
incumbent_traces = [
|
||||
_trace_from_dict(item) for item in read_jsonl(current_trace_path)
|
||||
]
|
||||
comparison = self.analyzer.compare_cell(
|
||||
self._prompt(),
|
||||
unit,
|
||||
next(item for item in candidate_units if item.unit_id == unit.unit_id),
|
||||
incumbent_traces,
|
||||
traces,
|
||||
coordinate.dimension,
|
||||
)
|
||||
incumbent_timeout_rate = (
|
||||
sum(trace.timed_out is True for trace in incumbent_traces)
|
||||
/ len(incumbent_traces)
|
||||
)
|
||||
candidate_timeout_rate = (
|
||||
sum(trace.timed_out is True for trace in traces) / len(traces)
|
||||
)
|
||||
if (
|
||||
candidate_timeout_rate > incumbent_timeout_rate
|
||||
and "timeout" not in comparison["runtime_regressions"]
|
||||
):
|
||||
comparison["runtime_regressions"].append("timeout")
|
||||
atomic_write_json(comparison_path, comparison)
|
||||
|
||||
delta = float(comparison["candidate_score"]) - float(
|
||||
comparison["incumbent_score"]
|
||||
)
|
||||
if not comparison["task_relevant"]:
|
||||
accepted, reason = False, "target_unit_not_task_relevant"
|
||||
elif comparison["runtime_regressions"]:
|
||||
accepted = False
|
||||
reason = "runtime_regressed:" + ",".join(comparison["runtime_regressions"])
|
||||
elif delta < 0.5:
|
||||
accepted, reason = False, "target_cell_did_not_improve"
|
||||
else:
|
||||
accepted, reason = True, "target_improved_without_runtime_regression"
|
||||
decision: dict[str, Any] = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": accepted,
|
||||
"reason": reason,
|
||||
"target_delta": delta,
|
||||
"rollout_skill_sha256": candidate_skill_sha256,
|
||||
}
|
||||
artifacts_dir = getattr(self.runner, "artifacts_dir", None)
|
||||
rollout_output = artifacts_dir(batch_id) if callable(artifacts_dir) else None
|
||||
if isinstance(rollout_output, Path):
|
||||
decision["rollout_output"] = str(rollout_output.resolve())
|
||||
if accepted:
|
||||
updated = ScoreMatrix(matrix.level, candidate_units, dict(matrix.columns))
|
||||
updated.columns[coordinate.dimension] = dict(
|
||||
matrix.columns[coordinate.dimension]
|
||||
)
|
||||
updated.columns[coordinate.dimension][coordinate.unit_id] = CellScore(
|
||||
float(comparison["candidate_score"]),
|
||||
list(comparison["candidate_evidence"]),
|
||||
str(comparison["reason"]),
|
||||
)
|
||||
decision["matrix_after"] = updated.to_dict()
|
||||
else:
|
||||
decision["rejected_edit"] = {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": edit.edit_summary,
|
||||
"score_change": delta,
|
||||
"reject_reason": reason,
|
||||
"new_text_hash": edit_hash,
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
|
||||
|
||||
def _run_level(
|
||||
self,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
seed_traces: list[RolloutTrace] | None = None,
|
||||
section_depth: int | None = None,
|
||||
) -> tuple[ScoreMatrix, list[RolloutTrace], Coordinate | None]:
|
||||
level_dir = self.run_dir / "levels" / level
|
||||
level_dir.mkdir(parents=True, exist_ok=True)
|
||||
level_state = self.state["levels"].setdefault(level, {
|
||||
"iterations": 0, "completed": False,
|
||||
})
|
||||
current_trace_path = level_dir / "current-traces.jsonl"
|
||||
traces = self._load_or_rollout(
|
||||
self.current,
|
||||
current_trace_path,
|
||||
f"{self.state['run_id']}-deep-{level}-initial",
|
||||
seed=seed_traces,
|
||||
)
|
||||
matrix = self._load_or_matrix(level_dir / "matrix.json", level, units, traces)
|
||||
if level_state.get("completed"):
|
||||
exhausted_value = level_state.get("exhausted_coordinate")
|
||||
exhausted = Coordinate(**exhausted_value) if isinstance(exhausted_value, dict) else None
|
||||
return matrix, traces, exhausted
|
||||
exhausted_coordinate: Coordinate | None = None
|
||||
max_iterations = (
|
||||
MAX_SECTION_ITERATIONS if level == "section" else MAX_PARAGRAPH_ITERATIONS
|
||||
)
|
||||
while int(level_state["iterations"]) < max_iterations:
|
||||
excluded = self._exhausted(level)
|
||||
excluded.update(
|
||||
(unit.unit_id, dimension)
|
||||
for dimension in level_state.get("blocked_dimensions", [])
|
||||
for unit in matrix.units
|
||||
)
|
||||
pending_number = int(level_state["iterations"]) + 1
|
||||
pending_dir = level_dir / "iterations" / f"iteration-{pending_number:02d}"
|
||||
pending_decision = load_json(pending_dir / "decision.json")
|
||||
if isinstance(pending_decision, dict):
|
||||
pending_coordinate = Coordinate(**pending_decision["coordinate"])
|
||||
level_state.setdefault("active_dimension", pending_coordinate.dimension)
|
||||
matrix = self._commit_iteration(
|
||||
level, pending_number, pending_dir, matrix, current_trace_path
|
||||
)
|
||||
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
|
||||
if len(self._rejected(pending_coordinate)) >= REJECTION_LIMIT:
|
||||
exhausted_coordinate = pending_coordinate
|
||||
level_state["stop_reason"] = "coordinate_exhausted"
|
||||
level_state["exhausted_coordinate"] = asdict(pending_coordinate)
|
||||
break
|
||||
if matrix.select_coordinate(
|
||||
GAP_THRESHOLD, level_state.get("active_dimension"), excluded
|
||||
) is None:
|
||||
matrix = self._refresh_matrix(level, matrix.units, traces)
|
||||
continue
|
||||
active_dimension = level_state.get("active_dimension")
|
||||
coordinate = matrix.select_coordinate(
|
||||
GAP_THRESHOLD, active_dimension, excluded
|
||||
)
|
||||
if coordinate is None:
|
||||
if active_dimension is None:
|
||||
level_state["stop_reason"] = (
|
||||
"normalized_gap_converged"
|
||||
if max(matrix.normalized_gaps().values(), default=0.0) <= GAP_THRESHOLD
|
||||
else "available_coordinates_exhausted"
|
||||
)
|
||||
break
|
||||
matrix = self._refresh_matrix(level, matrix.units, traces)
|
||||
continue
|
||||
if active_dimension is None:
|
||||
level_state["active_dimension"] = coordinate.dimension
|
||||
self._save()
|
||||
number = int(level_state["iterations"]) + 1
|
||||
matrix = self._run_iteration(
|
||||
level, number, level_dir, matrix, coordinate,
|
||||
current_trace_path, section_depth,
|
||||
)
|
||||
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
|
||||
if len(self._rejected(coordinate)) >= REJECTION_LIMIT:
|
||||
exhausted_coordinate = coordinate
|
||||
level_state["stop_reason"] = "coordinate_exhausted"
|
||||
level_state["exhausted_coordinate"] = asdict(coordinate)
|
||||
break
|
||||
else:
|
||||
decisions = self._decisions(level)
|
||||
last = decisions[-1] if decisions else {}
|
||||
if matrix.select_coordinate(GAP_THRESHOLD) is None:
|
||||
level_state["stop_reason"] = "normalized_gap_converged"
|
||||
else:
|
||||
level_state["stop_reason"] = (
|
||||
"max_iterations_after_accept" if last.get("accepted") else "max_iterations"
|
||||
)
|
||||
level_state["completed"] = True
|
||||
self._save()
|
||||
return matrix, traces, exhausted_coordinate
|
||||
|
||||
def drive(self) -> Path:
|
||||
if self.state.get("status") == "complete" and (self.run_dir / "S_final").is_dir():
|
||||
return self.run_dir / "S_final"
|
||||
_log(f"Deep Loop start/resume: {self.run_dir}")
|
||||
input_traces = [
|
||||
_trace_from_dict(item) for item in read_jsonl(self.run_dir / "input-traces.jsonl")
|
||||
]
|
||||
while True:
|
||||
section_units = parse_sections(self._current_text())
|
||||
section_matrix, traces, exhausted = self._run_level(
|
||||
"section", section_units, seed_traces=input_traces
|
||||
)
|
||||
if exhausted is None:
|
||||
break
|
||||
target_score = section_matrix.columns[exhausted.dimension][exhausted.unit_id].score
|
||||
current_sections = parse_sections(self._current_text())
|
||||
section = next((item for item in current_sections if item.unit_id == exhausted.unit_id), None)
|
||||
paragraphs = parse_paragraphs(section) if section is not None else []
|
||||
section_state = self.state["levels"]["section"]
|
||||
if (
|
||||
target_score > 3.5
|
||||
or len(paragraphs) < 2
|
||||
or section_state.get("paragraph_returned")
|
||||
):
|
||||
self._block_dimension(section_state, exhausted.dimension)
|
||||
section_state["completed"] = False
|
||||
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
|
||||
section_state.pop(key, None)
|
||||
self._save()
|
||||
continue
|
||||
_log(f"Descending into paragraphs of {exhausted.unit_id}")
|
||||
self.state["levels"].setdefault(
|
||||
"paragraph", {"iterations": 0, "completed": False}
|
||||
)["active_dimension"] = exhausted.dimension
|
||||
self._save()
|
||||
_, paragraph_traces, _ = self._run_level(
|
||||
"paragraph", paragraphs, seed_traces=traces,
|
||||
section_depth=section.heading_depth if section else None,
|
||||
)
|
||||
atomic_write_jsonl(
|
||||
self.run_dir / "levels" / "section" / "current-traces.jsonl",
|
||||
[asdict(trace) for trace in paragraph_traces],
|
||||
)
|
||||
section_state["completed"] = False
|
||||
section_state["paragraph_returned"] = True
|
||||
self._block_dimension(section_state, exhausted.dimension)
|
||||
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
|
||||
section_state.pop(key, None)
|
||||
self._refresh_matrix(
|
||||
"section", parse_sections(self._current_text()), paragraph_traces
|
||||
)
|
||||
return self._complete()
|
||||
|
||||
def _complete(self) -> Path:
|
||||
target = self.run_dir / "S_final"
|
||||
if target.exists():
|
||||
shutil.rmtree(target)
|
||||
shutil.copytree(self.current, target)
|
||||
final_skill_hash = sha256_file(target / "SKILL.md")
|
||||
final_rollouts = self.state.get("current_rollouts")
|
||||
final_rollout_hash = self.state.get("current_rollout_skill_sha256")
|
||||
if (
|
||||
not isinstance(final_rollouts, str)
|
||||
or not Path(final_rollouts).is_dir()
|
||||
or final_rollout_hash != final_skill_hash
|
||||
):
|
||||
final_rollouts = None
|
||||
report = {
|
||||
"status": "complete",
|
||||
"input_package_hash": package_hash(self.run_dir / "S_fast"),
|
||||
"final_package_hash": package_hash(target),
|
||||
"final_rollouts": final_rollouts,
|
||||
"final_rollout_skill_sha256": final_skill_hash if final_rollouts else None,
|
||||
"task": self.state["task"]["name"],
|
||||
"agent": self.state["task"]["agent"],
|
||||
"target_model": self.state["task"]["model"],
|
||||
"model": self.model,
|
||||
"rollout_backend": "skillsbench_development",
|
||||
"production_rollout_backend": "blank_container_required",
|
||||
"verifier_signal_used": False,
|
||||
"levels": self.state["levels"],
|
||||
"decisions": {
|
||||
name: self._decisions(name)
|
||||
for name in ("section", "paragraph")
|
||||
if name in self.state["levels"]
|
||||
},
|
||||
}
|
||||
atomic_write_json(self.run_dir / "report.json", report)
|
||||
self.state["status"] = "complete"
|
||||
self._save()
|
||||
shutil.rmtree(self.temp, ignore_errors=True)
|
||||
_log(f"Deep Loop complete: {target}")
|
||||
return target
|
||||
@@ -0,0 +1,3 @@
|
||||
from .cli import main
|
||||
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,300 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
from .models import Patch, RolloutTrace
|
||||
from .scoring.pre_score import PRE_SCORE_VERSION, PreScoreResult
|
||||
from .storage import (
|
||||
atomic_write_json,
|
||||
load_json,
|
||||
package_hash,
|
||||
read_jsonl,
|
||||
sha256_file,
|
||||
sha256_text,
|
||||
)
|
||||
|
||||
|
||||
CACHE_SCHEMA_VERSION = 1
|
||||
CACHE_PRODUCER = "dynamic-compile-fast"
|
||||
SCORE_ARTIFACTS = (
|
||||
"pre_scores.jsonl",
|
||||
"agentrm_scores.jsonl",
|
||||
"effective_scores.jsonl",
|
||||
)
|
||||
|
||||
|
||||
def fingerprint(value: Any) -> str:
|
||||
payload = json.dumps(
|
||||
value, ensure_ascii=False, sort_keys=True, separators=(",", ":")
|
||||
)
|
||||
return sha256_text(payload)
|
||||
|
||||
|
||||
def trace_fingerprint(traces: Iterable[RolloutTrace]) -> str:
|
||||
return fingerprint([asdict(trace) for trace in traces])
|
||||
|
||||
|
||||
def _artifact_hash(path: Path) -> str:
|
||||
if path.is_file():
|
||||
return sha256_file(path)
|
||||
if path.is_dir():
|
||||
return package_hash(path)
|
||||
raise OSError(f"cache artifact does not exist: {path}")
|
||||
|
||||
|
||||
def write_manifest(
|
||||
root: Path,
|
||||
name: str,
|
||||
*,
|
||||
stage: str,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
artifacts: Iterable[str],
|
||||
) -> None:
|
||||
artifact_names = tuple(artifacts)
|
||||
hashes = {item: _artifact_hash(root / item) for item in artifact_names}
|
||||
atomic_write_json(
|
||||
root / name,
|
||||
{
|
||||
"schema_version": CACHE_SCHEMA_VERSION,
|
||||
"producer": CACHE_PRODUCER,
|
||||
"stage": stage,
|
||||
"status": "complete",
|
||||
"input_hash": input_hash,
|
||||
"config_hash": fingerprint(config),
|
||||
"artifacts": hashes,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def valid_manifest(
|
||||
root: Path,
|
||||
name: str,
|
||||
*,
|
||||
stage: str,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
artifacts: Iterable[str],
|
||||
) -> bool:
|
||||
try:
|
||||
value = load_json(root / name)
|
||||
expected = tuple(artifacts)
|
||||
if not isinstance(value, dict):
|
||||
return False
|
||||
if value.get("schema_version") != CACHE_SCHEMA_VERSION:
|
||||
return False
|
||||
if value.get("producer") != CACHE_PRODUCER:
|
||||
return False
|
||||
if value.get("stage") != stage or value.get("status") != "complete":
|
||||
return False
|
||||
if value.get("input_hash") != input_hash:
|
||||
return False
|
||||
if value.get("config_hash") != fingerprint(config):
|
||||
return False
|
||||
hashes = value.get("artifacts")
|
||||
if not isinstance(hashes, dict) or set(hashes) != set(expected):
|
||||
return False
|
||||
return all(
|
||||
isinstance(hashes[item], str)
|
||||
and hashes[item] == _artifact_hash(root / item)
|
||||
for item in expected
|
||||
)
|
||||
except (OSError, TypeError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def load_score_cache(
|
||||
traces: list[RolloutTrace],
|
||||
score_dir: Path,
|
||||
*,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
) -> tuple[dict[str, float], list[dict[str, Any]]] | None:
|
||||
if not valid_manifest(
|
||||
score_dir,
|
||||
".score-cache.json",
|
||||
stage="score",
|
||||
input_hash=input_hash,
|
||||
config=config,
|
||||
artifacts=SCORE_ARTIFACTS,
|
||||
):
|
||||
return None
|
||||
try:
|
||||
pre_rows = read_jsonl(score_dir / "pre_scores.jsonl")
|
||||
agentrm_rows = read_jsonl(score_dir / "agentrm_scores.jsonl")
|
||||
score_rows = read_jsonl(score_dir / "effective_scores.jsonl")
|
||||
trace_by_id = {trace.trace_id: trace for trace in traces}
|
||||
if len(trace_by_id) != len(traces):
|
||||
return None
|
||||
pre_by_id: dict[str, PreScoreResult] = {}
|
||||
for row in pre_rows:
|
||||
result = PreScoreResult.from_dict(row)
|
||||
if result.trace_id in pre_by_id:
|
||||
return None
|
||||
if result.scoring_version != PRE_SCORE_VERSION:
|
||||
return None
|
||||
if result.route not in {"agentrm", "fixed_score"}:
|
||||
return None
|
||||
if result.route == "fixed_score":
|
||||
if result.score is None or not math.isfinite(result.score):
|
||||
return None
|
||||
pre_by_id[result.trace_id] = result
|
||||
if set(pre_by_id) != set(trace_by_id):
|
||||
return None
|
||||
|
||||
expected_agentrm = {
|
||||
trace.key.as_tuple()
|
||||
for trace in traces
|
||||
if pre_by_id[trace.trace_id].route == "agentrm"
|
||||
}
|
||||
actual_agentrm: set[tuple[str, str, str]] = set()
|
||||
for row in agentrm_rows:
|
||||
key = (
|
||||
str(row["task_name"]),
|
||||
str(row["compile_type"]),
|
||||
str(row["test_name"]),
|
||||
)
|
||||
score = float(row["score"])
|
||||
if key in actual_agentrm or not math.isfinite(score):
|
||||
return None
|
||||
if int(row["n_tokens"]) < 0:
|
||||
return None
|
||||
actual_agentrm.add(key)
|
||||
if actual_agentrm != expected_agentrm:
|
||||
return None
|
||||
|
||||
by_id: dict[str, dict[str, Any]] = {}
|
||||
scores: dict[str, float] = {}
|
||||
for row in score_rows:
|
||||
trace_id = str(row["trace_id"])
|
||||
score = float(row["effective_score"])
|
||||
if trace_id in by_id or not math.isfinite(score):
|
||||
return None
|
||||
trace = trace_by_id.get(trace_id)
|
||||
if trace is None or (
|
||||
str(row["task_name"]),
|
||||
str(row["compile_type"]),
|
||||
str(row["test_name"]),
|
||||
) != trace.key.as_tuple():
|
||||
return None
|
||||
expected_source = (
|
||||
"agentrm"
|
||||
if pre_by_id[trace_id].route == "agentrm"
|
||||
else "pre_score"
|
||||
)
|
||||
if row.get("score_source") != expected_source:
|
||||
return None
|
||||
by_id[trace_id] = row
|
||||
scores[trace_id] = score
|
||||
if set(by_id) != set(trace_by_id):
|
||||
return None
|
||||
return scores, [by_id[trace.trace_id] for trace in traces]
|
||||
except (KeyError, OSError, TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def load_maps_cache(
|
||||
output_dir: Path,
|
||||
selected: list[str],
|
||||
*,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
) -> list[dict[str, Any]] | None:
|
||||
if not valid_manifest(
|
||||
output_dir,
|
||||
".maps-cache.json",
|
||||
stage="maps",
|
||||
input_hash=input_hash,
|
||||
config=config,
|
||||
artifacts=("maps.json",),
|
||||
):
|
||||
return None
|
||||
try:
|
||||
value = load_json(output_dir / "maps.json")
|
||||
if not isinstance(value, list) or not all(isinstance(item, dict) for item in value):
|
||||
return None
|
||||
by_id = {str(item.get("trace_id", "")): item for item in value}
|
||||
if len(by_id) != len(value) or set(by_id) != set(selected):
|
||||
return None
|
||||
if any(
|
||||
not isinstance(item.get("patterns"), list)
|
||||
or not all(isinstance(pattern, dict) for pattern in item["patterns"])
|
||||
for item in value
|
||||
):
|
||||
return None
|
||||
return [by_id[trace_id] for trace_id in selected]
|
||||
except (OSError, TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def load_reduction_cache(
|
||||
output_dir: Path,
|
||||
*,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
) -> dict[str, Any] | None:
|
||||
if not valid_manifest(
|
||||
output_dir,
|
||||
".reduction-cache.json",
|
||||
stage="reduction",
|
||||
input_hash=input_hash,
|
||||
config=config,
|
||||
artifacts=("reduction.json",),
|
||||
):
|
||||
return None
|
||||
try:
|
||||
value = load_json(output_dir / "reduction.json")
|
||||
if not isinstance(value, dict):
|
||||
return None
|
||||
for key in ("successful_pattern", "failure_pattern", "selected_gap"):
|
||||
item = value.get(key)
|
||||
if not isinstance(item, dict) or not str(item.get("description", "")).strip():
|
||||
return None
|
||||
return value
|
||||
except (OSError, TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def load_candidate_cache(
|
||||
output_dir: Path,
|
||||
candidate_name: str,
|
||||
skill_hash: str,
|
||||
*,
|
||||
input_hash: str,
|
||||
config: dict[str, Any],
|
||||
) -> Path | None:
|
||||
candidate_relative = f"candidate-skill/{candidate_name}"
|
||||
if not valid_manifest(
|
||||
output_dir,
|
||||
".candidate-cache.json",
|
||||
stage="candidate",
|
||||
input_hash=input_hash,
|
||||
config=config,
|
||||
artifacts=("patch.json", candidate_relative),
|
||||
):
|
||||
return None
|
||||
candidate = output_dir / candidate_relative
|
||||
try:
|
||||
value = load_json(output_dir / "patch.json")
|
||||
if not isinstance(value, dict) or value.get("skill_hash") != skill_hash:
|
||||
return None
|
||||
patches = value.get("patches")
|
||||
if not isinstance(patches, list) or not 1 <= len(patches) <= 2:
|
||||
return None
|
||||
parsed = [Patch.from_dict(item) for item in patches if isinstance(item, dict)]
|
||||
if len(parsed) != len(patches) or any(patch.skill_hash != skill_hash for patch in parsed):
|
||||
return None
|
||||
if [patch.role for patch in parsed] != ["promote_success", "mitigate_failure"][:len(parsed)]:
|
||||
return None
|
||||
candidate_skill = candidate / "SKILL.md"
|
||||
if not candidate_skill.is_file():
|
||||
return None
|
||||
if value.get("candidate_skill_hash") != sha256_file(candidate_skill):
|
||||
return None
|
||||
return candidate
|
||||
except (OSError, TypeError, ValueError):
|
||||
return None
|
||||
@@ -0,0 +1,97 @@
|
||||
"""动态编译唯一命令行入口。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.provider_router import parse_model_reference
|
||||
|
||||
from .pipeline import run_pipeline
|
||||
from .scoring.agentrm import (
|
||||
DEFAULT_BATCH_SIZE,
|
||||
DEFAULT_CONCURRENCY,
|
||||
DEFAULT_MAX_LENGTH,
|
||||
DEFAULT_RM_API_URL,
|
||||
DEFAULT_TIMEOUT,
|
||||
)
|
||||
|
||||
|
||||
def _provider_model(value: str) -> str:
|
||||
try:
|
||||
return parse_model_reference(value).value
|
||||
except ValueError as exc:
|
||||
raise argparse.ArgumentTypeError(str(exc)) from exc
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="python -m scripts.dynamic_compile.fast",
|
||||
description="从一组 BenchFlow 历史轨迹生成一个动态编译候选 Skill。",
|
||||
)
|
||||
parser.add_argument("--traces", required=True, type=Path, help="BenchFlow 轨迹目录")
|
||||
parser.add_argument("--skill", required=True, type=Path, help="含根 SKILL.md 的 Skill 包")
|
||||
parser.add_argument("--score-output", type=Path, help="评分阶段产物目录")
|
||||
parser.add_argument("--output", type=Path, help="分析与候选 Skill 产物目录")
|
||||
parser.add_argument(
|
||||
"--model",
|
||||
required=True,
|
||||
type=_provider_model,
|
||||
help="所有外部模型调用使用的 provider/model",
|
||||
)
|
||||
parser.add_argument("--max-parallel", type=int, default=3)
|
||||
parser.add_argument(
|
||||
"--rm-api-url", default=os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
|
||||
)
|
||||
parser.add_argument(
|
||||
"--rm-max-length",
|
||||
type=int,
|
||||
default=os.environ.get("RM_MAX_LENGTH", str(DEFAULT_MAX_LENGTH)),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--rm-timeout",
|
||||
type=float,
|
||||
default=os.environ.get("RM_TIMEOUT", str(DEFAULT_TIMEOUT)),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--rm-concurrency",
|
||||
type=int,
|
||||
default=os.environ.get("RM_CONCURRENCY", str(DEFAULT_CONCURRENCY)),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--rm-batch-size",
|
||||
type=int,
|
||||
default=os.environ.get("RM_BATCH_SIZE", str(DEFAULT_BATCH_SIZE)),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--force",
|
||||
action="store_true",
|
||||
help="复用有效评分/Map 缓存,强制重建 Reduce、Patch 和候选 Skill",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = build_parser().parse_args(argv)
|
||||
try:
|
||||
result = run_pipeline(
|
||||
args.traces,
|
||||
args.skill,
|
||||
score_output=args.score_output,
|
||||
output=args.output,
|
||||
model=args.model,
|
||||
max_parallel=args.max_parallel,
|
||||
rm_api_url=args.rm_api_url,
|
||||
rm_max_length=args.rm_max_length,
|
||||
rm_timeout=args.rm_timeout,
|
||||
rm_concurrency=args.rm_concurrency,
|
||||
rm_batch_size=args.rm_batch_size,
|
||||
force=args.force,
|
||||
)
|
||||
except (OSError, RuntimeError, ValueError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
print(result)
|
||||
return 0
|
||||
@@ -0,0 +1,89 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from typing import Any, Mapping
|
||||
|
||||
|
||||
@dataclass(frozen=True, order=True)
|
||||
class TraceKey:
|
||||
task_name: str
|
||||
compile_type: str
|
||||
test_name: str
|
||||
|
||||
@classmethod
|
||||
def from_record(cls, value: Mapping[str, Any]) -> "TraceKey":
|
||||
try:
|
||||
return cls(
|
||||
str(value["task_name"]),
|
||||
str(value["compile_type"]),
|
||||
str(value["test_name"]),
|
||||
)
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"missing trace identity field: {exc.args[0]}") from exc
|
||||
|
||||
def as_tuple(self) -> tuple[str, str, str]:
|
||||
return self.task_name, self.compile_type, self.test_name
|
||||
|
||||
def __str__(self) -> str:
|
||||
return "/".join(self.as_tuple())
|
||||
|
||||
|
||||
@dataclass
|
||||
class Patch:
|
||||
edit_type: str
|
||||
target_heading: str
|
||||
old_text: str
|
||||
new_text: str
|
||||
evidence_ids: list[str]
|
||||
evidence_type: str
|
||||
confidence: str
|
||||
reason: str
|
||||
role: str = ""
|
||||
skill_hash: str = ""
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "Patch":
|
||||
required = {
|
||||
"edit_type", "target_heading", "old_text", "new_text", "evidence_ids",
|
||||
"evidence_type", "confidence", "reason",
|
||||
}
|
||||
missing = sorted(required - value.keys())
|
||||
if missing:
|
||||
raise ValueError(f"patch missing fields: {', '.join(missing)}")
|
||||
if not isinstance(value["evidence_ids"], list):
|
||||
raise ValueError("patch evidence_ids must be a list")
|
||||
if "role" in value and not isinstance(value["role"], str):
|
||||
raise ValueError("patch role must be a string")
|
||||
for key in required - {"evidence_ids"}:
|
||||
if not isinstance(value[key], str):
|
||||
raise ValueError(f"patch {key} must be a string")
|
||||
return cls(**{key: value.get(key, "") for key in cls.__dataclass_fields__})
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RolloutTrace:
|
||||
trace_id: str
|
||||
task_name: str
|
||||
compile_type: str
|
||||
test_name: str
|
||||
state: list[dict[str, Any]]
|
||||
skill_invoked: bool
|
||||
exit_code: int | None = None
|
||||
timed_out: bool | None = None
|
||||
metadata: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def key(self) -> TraceKey:
|
||||
return TraceKey(self.task_name, self.compile_type, self.test_name)
|
||||
|
||||
def agentrm_request(self) -> dict[str, Any]:
|
||||
"""Project a rich runtime trace onto AgentRM's stable input schema."""
|
||||
return {
|
||||
"state": self.state,
|
||||
"task_name": self.task_name,
|
||||
"compile_type": self.compile_type,
|
||||
"test_name": self.test_name,
|
||||
}
|
||||
@@ -0,0 +1,239 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from ..models import Patch, RolloutTrace
|
||||
from scripts.provider_router import resolve_model_route
|
||||
|
||||
from ..storage import package_manifest, sha256_file
|
||||
from .trace_format import compact_trace
|
||||
|
||||
|
||||
class SemanticClient:
|
||||
def __init__(
|
||||
self,
|
||||
model: str = "opencode/deepseek-v4-pro",
|
||||
timeout: int = 900,
|
||||
):
|
||||
route = resolve_model_route(model)
|
||||
assert route is not None
|
||||
base_url = route.url.removesuffix("/chat/completions").rstrip("/")
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("the openai package is required for semantic calls") from exc
|
||||
self.client = OpenAI(base_url=base_url, api_key=route.api_key, timeout=timeout, max_retries=0)
|
||||
self.model = route.reference.model_id
|
||||
self.model_reference = route.reference.value
|
||||
|
||||
def json(self, system: str, user: str, attempts: int = 1) -> dict[str, Any]:
|
||||
error: Exception | None = None
|
||||
for attempt in range(attempts):
|
||||
try:
|
||||
response = self.client.chat.completions.create(
|
||||
model=self.model,
|
||||
temperature=0,
|
||||
response_format={"type": "json_object"},
|
||||
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
stream=True,
|
||||
)
|
||||
content = "".join(
|
||||
choice.delta.content or ""
|
||||
for chunk in response
|
||||
for choice in chunk.choices
|
||||
)
|
||||
value = json.loads(content or "{}")
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError("semantic response must be a JSON object")
|
||||
return value
|
||||
except Exception as exc:
|
||||
error = exc
|
||||
if attempt + 1 < attempts:
|
||||
print(
|
||||
f"[semantic] request attempt {attempt + 1}/{attempts} failed: "
|
||||
f"{type(exc).__name__}: {exc}; retrying",
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
time.sleep(1 + attempt)
|
||||
raise RuntimeError(f"semantic call failed after {attempts} attempts: {error}")
|
||||
|
||||
|
||||
class SemanticAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
def generate_probe(self, skill_package: Path) -> dict[str, str]:
|
||||
skill = (skill_package / "SKILL.md").read_text(encoding="utf-8")
|
||||
files = [item["path"] for item in package_manifest(skill_package)]
|
||||
result = self.client.json(
|
||||
"You create realistic probe tasks for testing an agent skill. Return JSON only.",
|
||||
f"""Create one task prompt for the skill below. The task must explicitly tell the agent to invoke this skill, exercise its core workflow, and remain solvable in an empty workspace with network access. Do not copy the skill's operation steps, create a verifier, or create a test environment. Return {{"prompt": string, "rationale": string}}.
|
||||
|
||||
Package files: {json.dumps(files, ensure_ascii=False)}
|
||||
SKILL.md:
|
||||
{skill}""",
|
||||
)
|
||||
prompt = result.get("prompt")
|
||||
if not isinstance(prompt, str) or not prompt.strip():
|
||||
raise ValueError("probe generator returned no prompt")
|
||||
return {"prompt": prompt.strip(), "rationale": str(result.get("rationale", ""))}
|
||||
|
||||
def map_trace(self, trace: RolloutTrace, score: float, bucket: str) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Analyze agent behavior from a scored trace. Return concise JSON only.",
|
||||
f"""Analyze this {bucket} trace (AgentRM score {score}) as an action-level workflow. Identify what the agent did, action ordering, stopping behavior, error recovery, and whether required outputs were persisted promptly. Runtime facts report observable execution only; do not infer external verification outcomes or correctness that is not visible in the trace. Use long payload details only when they are necessary to explain a behavioral effect. Return:
|
||||
{{"trace_id":"{trace.trace_id}","patterns":[{{"description":string,"condition":string,"effect":string,"recovered":boolean,"final_quality_impact":string,"evidence_ids":[string]}}]}}.
|
||||
Use E### or E###.T## event IDs as evidence. Skill invocation is evidence, not a quality gate.
|
||||
|
||||
{compact_trace(trace)}""",
|
||||
)
|
||||
result["runtime_facts"] = {
|
||||
"termination": trace.metadata.get("termination"),
|
||||
"timed_out": trace.timed_out,
|
||||
"exit_code": trace.exit_code,
|
||||
"duration_seconds": trace.metadata.get("agent_execution_seconds"),
|
||||
"tool_calls": trace.metadata.get("tool_calls"),
|
||||
"skill_invoked": trace.skill_invoked,
|
||||
"timeout_reason": trace.metadata.get("timeout_reason"),
|
||||
"error_category": trace.metadata.get("error_category"),
|
||||
"partial_trajectory": trace.metadata.get("partial_trajectory"),
|
||||
}
|
||||
result.setdefault("trace_id", trace.trace_id)
|
||||
result.setdefault("patterns", [])
|
||||
return result
|
||||
|
||||
def map_all(
|
||||
self,
|
||||
traces: list[RolloutTrace],
|
||||
scores: dict[str, float],
|
||||
high_ids: set[str],
|
||||
low_ids: set[str] | None = None,
|
||||
progress: Any | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
def work(trace: RolloutTrace) -> dict[str, Any]:
|
||||
bucket = "High" if trace.trace_id in high_ids else "Low" if low_ids is None or trace.trace_id in low_ids else "Neutral"
|
||||
return self.map_trace(trace, scores[trace.trace_id], bucket)
|
||||
|
||||
results: list[dict[str, Any] | None] = [None] * len(traces)
|
||||
errors: list[tuple[str, Exception]] = []
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {pool.submit(work, trace): index for index, trace in enumerate(traces)}
|
||||
completed = 0
|
||||
for future in as_completed(futures):
|
||||
index = futures[future]
|
||||
try:
|
||||
result = future.result()
|
||||
except Exception as exc:
|
||||
errors.append((traces[index].trace_id, exc))
|
||||
continue
|
||||
results[index] = result
|
||||
if result_callback is not None:
|
||||
result_callback(result)
|
||||
completed += 1
|
||||
if progress is not None:
|
||||
progress(completed, len(traces), traces[index].trace_id)
|
||||
if errors:
|
||||
details = "; ".join(
|
||||
f"{trace_id}: {type(error).__name__}: {error}"
|
||||
for trace_id, error in errors
|
||||
)
|
||||
raise RuntimeError(f"{len(errors)} Map trace(s) failed; successful results were preserved: {details}")
|
||||
return [result for result in results if result is not None]
|
||||
|
||||
def reduce(
|
||||
self,
|
||||
skill_text: str,
|
||||
maps: list[dict[str, Any]],
|
||||
score_rows: list[dict[str, Any]],
|
||||
high_ids: list[str],
|
||||
low_ids: list[str],
|
||||
history: list[dict[str, Any]],
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Contrast exactly one successful pattern with exactly one failure pattern to identify one bounded skill-improvement gap. Return JSON only with every requested field populated; never return an empty object.",
|
||||
f"""Compare the fixed relative High and Low groups. Select exactly one successful behavioral pattern from the High traces that the skill should preserve or promote, and exactly one contrasting failure pattern from the Low traces that the skill should mitigate. Then select exactly one local, generalizable gap in the current skill that connects those two patterns and has not already been addressed in Patch history. Do not select two unrelated improvements.
|
||||
|
||||
The selected gap must be one behavioral clarification or stopping decision, expressed in at most two sentences, not a multi-step policy. It must explicitly preserve the selected successful behavior and mitigate the selected failure behavior, remain general to the skill, and avoid turning the failure into an exhaustive or universal obligation. Do not mention benchmark-specific files or labels, invent exact counts, thresholds, quotas, or mandatory tool sequences, add significant tool work, broaden external search, or delay a required deliverable.
|
||||
|
||||
Infer the contrast before proposing the remedy. Compare runtime_facts for completion, duration, tool calls, and timeouts; use Map patterns and their evidence to explain the behavior. Effective score indicates relative outcome, not a root cause. Describe group tendencies only to the extent supported, cite the supporting trace and event IDs, and reflect exceptions in confidence.
|
||||
|
||||
If High traces show several successful behaviors, choose the one with the clearest evidence and strongest direct contrast with the selected Low failure. If Low traces show opposing failure modes, choose only the strongest failure that can be addressed by the same qualitative decision boundary as the selected success. For incomplete work and overwork, prefer prioritization, evidence-based stopping, and timely persistence over additional checking.
|
||||
|
||||
You must still select one success, one failure, and one gap, each with a non-empty description. Each pattern must cite evidence from its corresponding group. If contrast is weak, use evidence_type=weak_contrast_fallback and confidence=low, and choose the most conservative supported pair and clarification.
|
||||
Return every field in this exact shape: {{"successful_pattern":{{"description":"one High-group behavior to preserve or promote","evidence_ids":["trace_id:E###"]}},"failure_pattern":{{"description":"one contrasting Low-group behavior to mitigate","evidence_ids":["trace_id:E###"]}},"contrast":"direct relationship between the selected success and failure","root_cause":"skill-level cause","source":"compared evidence","skill_mitigatable":true,"confidence":"low|medium|high","selected_gap":{{"description":"one supported behavioral clarification","success_behavior_to_preserve":"the selected successful behavior","failure_behavior_to_mitigate":"the selected failure behavior","target_heading":"existing skill heading","evidence_type":"contrast type","confidence":"low|medium|high","evidence_ids":["trace_id:E###"],"reason":"why this one gap preserves the success while mitigating the failure"}}}}.
|
||||
|
||||
High IDs: {json.dumps(high_ids)}
|
||||
Low IDs: {json.dumps(low_ids)}
|
||||
Scores: {json.dumps(score_rows, ensure_ascii=False)}
|
||||
Map results: {json.dumps(maps, ensure_ascii=False)}
|
||||
Patch history: {json.dumps(history, ensure_ascii=False)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
for field, label in (
|
||||
("successful_pattern", "successful pattern"),
|
||||
("failure_pattern", "failure pattern"),
|
||||
):
|
||||
pattern = result.get(field)
|
||||
if not isinstance(pattern, dict) or not str(pattern.get("description", "")).strip():
|
||||
raise ValueError(f"reducer returned no usable {label}")
|
||||
evidence_ids = pattern.get("evidence_ids")
|
||||
if not isinstance(evidence_ids, list) or not evidence_ids:
|
||||
raise ValueError(f"reducer returned no evidence for {label}")
|
||||
gap = result.get("selected_gap")
|
||||
if not isinstance(gap, dict) or not str(gap.get("description", "")).strip():
|
||||
raise ValueError("reducer returned no usable contrastive gap")
|
||||
for field in ("success_behavior_to_preserve", "failure_behavior_to_mitigate"):
|
||||
if not str(gap.get(field, "")).strip():
|
||||
raise ValueError(f"reducer selected_gap missing {field}")
|
||||
return result
|
||||
|
||||
def generate_patches(
|
||||
self, skill_path: Path, reduction: dict[str, Any], history: list[dict[str, Any]], error: str = ""
|
||||
) -> list[Patch]:
|
||||
text = skill_path.read_text(encoding="utf-8")
|
||||
result = self.client.json(
|
||||
"Generate a small ordered patch bundle for a skill document. Return JSON only.",
|
||||
f"""Generate one required success-oriented patch and, only when it adds distinct value, one optional failure-oriented patch. Both patches must address the same selected_gap; do not introduce unrelated improvements.
|
||||
|
||||
The required promote_success patch must express the selected successful behavior as a clear, actionable recommended workflow or stopping condition in the most appropriate existing section.
|
||||
|
||||
The optional mitigate_failure patch is allowed only when it adds non-duplicative detection, recovery, or exception-handling guidance. Omit it when it would merely negate, restate, or cross-reference the promote_success patch. If included, it must remain useful independently rather than existing only to repeat the preferred path.
|
||||
|
||||
Return patches in application order: promote_success first, then optional mitigate_failure. Each patch is one contiguous text replacement. For every patch, old_text must be a non-empty, uniquely occurring verbatim substring of the original SKILL.md and patches must target non-overlapping substrings so they can be applied sequentially. new_text must replace old_text locally and preserve general applicability. Do not rewrite the whole document; keep textual growth and behavioral scope minimal.
|
||||
|
||||
The patch bundle must preserve efficient successful behavior. It must not add significant tool cost, introduce mandatory tool or API calls, broaden the existing external search scope, require exhaustive checking when targeted checking is sufficient, or delay creation of a required deliverable. Prefer prioritization, bounded stopping criteria, and writing or updating required outputs as soon as the core result is supported. Do not turn a trace-specific failure into an unconditional every/all/always/never/only-after rule unless the task itself inherently requires that rule.
|
||||
|
||||
Return {{"patches":[{{"role":"promote_success|mitigate_failure","edit_type":string,"target_heading":string,"old_text":string,"new_text":string,"evidence_ids":[string],"evidence_type":string,"confidence":string,"reason":string}}]}}. The patches array must contain one or two items and must always begin with promote_success.
|
||||
Selected analysis: {json.dumps(reduction, ensure_ascii=False)}
|
||||
History: {json.dumps(history, ensure_ascii=False)}
|
||||
Previous application error: {error}
|
||||
SKILL.md:
|
||||
{text}""",
|
||||
)
|
||||
values = result.get("patches")
|
||||
if not isinstance(values, list) or not 1 <= len(values) <= 2:
|
||||
raise ValueError("patch generator must return one or two patches")
|
||||
if not all(isinstance(value, dict) for value in values):
|
||||
raise ValueError("every generated patch must be an object")
|
||||
expected_roles = ["promote_success", "mitigate_failure"]
|
||||
roles = [value.get("role") for value in values]
|
||||
if roles != expected_roles[: len(values)]:
|
||||
raise ValueError(
|
||||
"patch roles must be promote_success followed by optional mitigate_failure"
|
||||
)
|
||||
patches = [Patch.from_dict(value) for value in values]
|
||||
if len(patches) == 2 and patches[0].new_text.strip() == patches[1].new_text.strip():
|
||||
raise ValueError("failure patch duplicates the success patch")
|
||||
skill_hash = sha256_file(skill_path)
|
||||
for patch in patches:
|
||||
patch.skill_hash = skill_hash
|
||||
return patches
|
||||
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import Patch
|
||||
from ..storage import atomic_write_text, sha256_file
|
||||
|
||||
|
||||
def _validate_skill_text(text: str) -> None:
|
||||
if not text.strip():
|
||||
raise ValueError("SKILL.md is empty")
|
||||
if text.startswith("---"):
|
||||
end = text.find("\n---", 3)
|
||||
if end < 0:
|
||||
raise ValueError("SKILL.md has an unterminated YAML frontmatter")
|
||||
frontmatter = text[3:end].strip()
|
||||
try:
|
||||
import yaml
|
||||
|
||||
parsed = yaml.safe_load(frontmatter) if frontmatter else {}
|
||||
except Exception as exc:
|
||||
raise ValueError(f"invalid SKILL.md frontmatter: {exc}") from exc
|
||||
if parsed is not None and not isinstance(parsed, dict):
|
||||
raise ValueError("SKILL.md frontmatter must be a mapping")
|
||||
|
||||
|
||||
def apply_patches(
|
||||
current_package: Path, candidate_package: Path, patches: list[Patch]
|
||||
) -> str:
|
||||
skill = current_package / "SKILL.md"
|
||||
if not skill.is_file():
|
||||
raise ValueError(f"missing {skill}")
|
||||
if not patches:
|
||||
raise ValueError("at least one patch is required")
|
||||
actual_hash = sha256_file(skill)
|
||||
text = skill.read_text(encoding="utf-8")
|
||||
for index, patch in enumerate(patches, start=1):
|
||||
if not patch.skill_hash:
|
||||
raise ValueError(f"patch {index} missing generation-time skill_hash")
|
||||
if actual_hash != patch.skill_hash:
|
||||
raise ValueError(
|
||||
f"patch {index} was not generated from the current SKILL.md"
|
||||
)
|
||||
if not patch.old_text:
|
||||
raise ValueError(f"patch {index} old_text must not be empty")
|
||||
matches = text.count(patch.old_text)
|
||||
if matches != 1:
|
||||
raise ValueError(
|
||||
f"patch {index} old_text must match exactly once; found {matches}"
|
||||
)
|
||||
changed = text.replace(patch.old_text, patch.new_text, 1)
|
||||
if changed == text:
|
||||
raise ValueError(f"patch {index} does not change SKILL.md")
|
||||
_validate_skill_text(changed)
|
||||
text = changed
|
||||
token = uuid.uuid4().hex
|
||||
temporary = candidate_package.with_name(f".{candidate_package.name}.{token}.tmp")
|
||||
backup = candidate_package.with_name(f".{candidate_package.name}.{token}.bak")
|
||||
shutil.copytree(current_package, temporary)
|
||||
try:
|
||||
atomic_write_text(temporary / "SKILL.md", text)
|
||||
if candidate_package.exists():
|
||||
candidate_package.rename(backup)
|
||||
try:
|
||||
temporary.rename(candidate_package)
|
||||
except Exception:
|
||||
if backup.exists() and not candidate_package.exists():
|
||||
backup.rename(candidate_package)
|
||||
raise
|
||||
finally:
|
||||
if temporary.exists():
|
||||
shutil.rmtree(temporary)
|
||||
if backup.exists() and candidate_package.exists():
|
||||
shutil.rmtree(backup)
|
||||
return sha256_file(candidate_package / "SKILL.md")
|
||||
@@ -0,0 +1,21 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
|
||||
|
||||
def relative_high_low(
|
||||
score_by_id: dict[str, float], count: int = 3, seed: str | int = 0
|
||||
) -> tuple[list[str], list[str]]:
|
||||
"""稳定选择互不重叠的相对高分组和低分组。"""
|
||||
|
||||
if len(score_by_id) < count * 2:
|
||||
raise ValueError(f"need at least {count * 2} traces for disjoint High/Low groups")
|
||||
trace_ids = list(score_by_id)
|
||||
random.Random(str(seed)).shuffle(trace_ids)
|
||||
high = sorted(trace_ids, key=score_by_id.__getitem__, reverse=True)[:count]
|
||||
high_set = set(high)
|
||||
low = sorted(
|
||||
(trace_id for trace_id in trace_ids if trace_id not in high_set),
|
||||
key=score_by_id.__getitem__,
|
||||
)[:count]
|
||||
return high, low
|
||||
@@ -0,0 +1,234 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
from ..models import RolloutTrace
|
||||
|
||||
|
||||
_TOOL_CALL_RE = re.compile(r"^Tool call (?P<name>[^:\n]+):[ \t]*", re.MULTILINE)
|
||||
_TOOL_RESULT_RE = re.compile(
|
||||
r"^Tool result(?: \((?P<name>[^;\n)]+)(?:;[ \t]*(?P<status>[^)\n]+))?\))?:?[ \t]*",
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
|
||||
def _content(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return json.dumps(value, ensure_ascii=False)
|
||||
|
||||
|
||||
def opencode_events_to_state(
|
||||
lines: Iterable[str], probe: str
|
||||
) -> tuple[list[dict[str, str]], bool, int]:
|
||||
state: list[dict[str, str]] = [{"role": "user", "content": probe}]
|
||||
invoked = False
|
||||
parsed = 0
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not isinstance(event, dict):
|
||||
continue
|
||||
parsed += 1
|
||||
kind = str(event.get("type", event.get("event", ""))).lower()
|
||||
part = event.get("part") if isinstance(event.get("part"), dict) else event
|
||||
tool = part.get("tool") or part.get("name") or event.get("tool") or event.get("name")
|
||||
title = part.get("title") or event.get("title") or ""
|
||||
state_data = part.get("state") if isinstance(part.get("state"), dict) else {}
|
||||
haystack = " ".join([str(kind), str(tool or ""), str(title), _content(part)])
|
||||
if str(tool or "").lower() == "skill" or "<skill_content" in haystack.lower():
|
||||
invoked = True
|
||||
if tool or "tool" in kind:
|
||||
arguments = state_data.get("input", part.get("input", part.get("arguments", {})))
|
||||
output = state_data.get("output", part.get("output", part.get("result", "")))
|
||||
state.append({"role": "assistant", "content": f"Tool call {tool or title}: {_content(arguments)}"})
|
||||
if output not in (None, ""):
|
||||
state.append({"role": "user", "content": f"Tool result: {_content(output)}"})
|
||||
continue
|
||||
text = part.get("text", part.get("content", event.get("message", "")))
|
||||
if text not in (None, ""):
|
||||
role = str(event.get("role", part.get("role", "assistant")))
|
||||
if role not in {"assistant", "user", "system"}:
|
||||
role = "assistant"
|
||||
state.append({"role": role, "content": _content(text)})
|
||||
return state, invoked, parsed
|
||||
|
||||
|
||||
def read_event_file(path: Path, probe: str) -> tuple[list[dict[str, str]], bool, int]:
|
||||
with path.open(encoding="utf-8", errors="replace") as handle:
|
||||
return opencode_events_to_state(handle, probe)
|
||||
|
||||
|
||||
def _split_tool_results(content: str) -> tuple[str, list[tuple[str, str, str]]]:
|
||||
matches = list(_TOOL_RESULT_RE.finditer(content))
|
||||
if not matches:
|
||||
return content, []
|
||||
prefix = content[:matches[0].start()].strip()
|
||||
results = []
|
||||
for index, match in enumerate(matches):
|
||||
end = matches[index + 1].start() if index + 1 < len(matches) else len(content)
|
||||
results.append((
|
||||
(match.group("name") or "unknown").strip(),
|
||||
(match.group("status") or "unknown").strip(),
|
||||
content[match.end():end].strip(),
|
||||
))
|
||||
return prefix, results
|
||||
|
||||
|
||||
def _excerpt(content: str, limit: int) -> str:
|
||||
content = content.strip()
|
||||
if len(content) <= limit:
|
||||
return content
|
||||
marker = "\n[... content omitted ...]\n"
|
||||
if limit <= len(marker) + 2:
|
||||
return content[:limit]
|
||||
omitted = len(content) - (limit - len(marker))
|
||||
while True:
|
||||
marker = f"\n[... {omitted} chars omitted ...]\n"
|
||||
available = limit - len(marker)
|
||||
updated = len(content) - available
|
||||
if updated == omitted:
|
||||
break
|
||||
omitted = updated
|
||||
head = (available + 1) // 2
|
||||
tail = available // 2
|
||||
return content[:head] + marker + content[-tail:]
|
||||
|
||||
|
||||
def _termination(trace: RolloutTrace) -> str:
|
||||
value = trace.metadata.get("termination")
|
||||
if value not in (None, ""):
|
||||
return str(value)
|
||||
if trace.timed_out is True:
|
||||
return "timeout"
|
||||
if trace.exit_code == 0:
|
||||
return "completed"
|
||||
if trace.exit_code is not None:
|
||||
return "error"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _runtime_facts(trace: RolloutTrace) -> str:
|
||||
metadata = trace.metadata
|
||||
facts = {
|
||||
"trace_id": trace.trace_id,
|
||||
"termination": _termination(trace),
|
||||
"timed_out": trace.timed_out,
|
||||
"exit_code": trace.exit_code,
|
||||
"agent_execution_seconds": metadata.get("agent_execution_seconds"),
|
||||
"tool_calls": metadata.get("tool_calls"),
|
||||
"skill_invoked": trace.skill_invoked,
|
||||
"timeout_reason": metadata.get("timeout_reason"),
|
||||
"error_category": metadata.get("error_category"),
|
||||
"partial_trajectory": metadata.get("partial_trajectory"),
|
||||
}
|
||||
return "RUNTIME_FACTS " + json.dumps(facts, ensure_ascii=False, separators=(",", ":"))
|
||||
|
||||
|
||||
def _allocate_excerpt_budget(caps: list[int], available: int) -> list[int]:
|
||||
allocations = [0] * len(caps)
|
||||
active = [index for index, cap in enumerate(caps) if cap > 0]
|
||||
while active and available > 0:
|
||||
share = max(1, available // len(active))
|
||||
progressed = False
|
||||
for index in active.copy():
|
||||
amount = min(share, caps[index] - allocations[index], available)
|
||||
allocations[index] += amount
|
||||
available -= amount
|
||||
progressed = progressed or amount > 0
|
||||
if allocations[index] >= caps[index]:
|
||||
active.remove(index)
|
||||
if available == 0:
|
||||
break
|
||||
if not progressed:
|
||||
break
|
||||
return allocations
|
||||
|
||||
|
||||
def compact_trace(trace: RolloutTrace, total: int = 30000) -> str:
|
||||
"""Render runtime facts and a complete action ledger for one Map call."""
|
||||
entries: list[dict[str, Any]] = []
|
||||
for index, message in enumerate(trace.state):
|
||||
event_id = f"E{index:03d}"
|
||||
role = str(message.get("role", "unknown"))
|
||||
content = _content(message.get("content", ""))
|
||||
|
||||
tool_call = _TOOL_CALL_RE.match(content)
|
||||
if tool_call:
|
||||
entries.append({
|
||||
"id": f"{event_id}.T01",
|
||||
"kind": "tool_call",
|
||||
"role": role,
|
||||
"name": tool_call.group("name").strip(),
|
||||
"status": "unknown",
|
||||
"content": content[tool_call.end():].strip(),
|
||||
})
|
||||
continue
|
||||
|
||||
prefix, tool_results = _split_tool_results(content) if index > 0 else (content, [])
|
||||
if prefix:
|
||||
entries.append({
|
||||
"id": event_id,
|
||||
"kind": "message",
|
||||
"role": role,
|
||||
"content": prefix,
|
||||
})
|
||||
for tool_index, (name, status, result) in enumerate(tool_results, 1):
|
||||
entries.append({
|
||||
"id": f"{event_id}.T{tool_index:02d}",
|
||||
"kind": "tool_result",
|
||||
"role": role,
|
||||
"name": name,
|
||||
"status": status,
|
||||
"content": result,
|
||||
})
|
||||
|
||||
message_entries = [entry for entry in entries if entry["kind"] == "message"]
|
||||
first_user = next((entry for entry in message_entries if entry["role"] == "user"), None)
|
||||
final_assistant = next(
|
||||
(entry for entry in reversed(message_entries) if entry["role"] == "assistant"),
|
||||
None,
|
||||
)
|
||||
|
||||
skeletons = []
|
||||
caps = []
|
||||
for entry in entries:
|
||||
if entry["kind"] == "message":
|
||||
labels = ["message", f"role={entry['role']}"]
|
||||
if entry is first_user:
|
||||
labels.append("task")
|
||||
if entry is final_assistant:
|
||||
labels.append("final")
|
||||
skeleton = f"{entry['id']} [" + " ".join(labels) + "]"
|
||||
cap = 2500 if entry is first_user else 3000 if entry is final_assistant else 600
|
||||
else:
|
||||
name = str(entry["name"])[:80]
|
||||
status = str(entry["status"])[:40]
|
||||
skeleton = f"{entry['id']} [{entry['kind']} name={name} status={status}]"
|
||||
cap = 600
|
||||
skeletons.append(skeleton)
|
||||
caps.append(min(cap, len(str(entry["content"]))))
|
||||
|
||||
facts = _runtime_facts(trace)
|
||||
fixed_size = (
|
||||
len(facts)
|
||||
+ sum(len(skeleton) + 1 for skeleton in skeletons)
|
||||
+ sum(1 for cap in caps if cap > 0)
|
||||
)
|
||||
allocations = _allocate_excerpt_budget(caps, max(0, total - fixed_size))
|
||||
rendered = [facts]
|
||||
for entry, skeleton, allocation in zip(entries, skeletons, allocations):
|
||||
rendered.append(skeleton)
|
||||
if allocation:
|
||||
rendered.append(_excerpt(str(entry["content"]), allocation))
|
||||
return "\n".join(rendered)
|
||||
@@ -0,0 +1,50 @@
|
||||
"""动态编译流水线的全部路径规则。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _find_project_root(module_dir: Path) -> Path:
|
||||
"""通过项目标志定位根目录,不编码包目录深度。"""
|
||||
|
||||
for candidate in (module_dir, *module_dir.parents):
|
||||
if (
|
||||
(candidate / "provider_routes.json").is_file()
|
||||
and (candidate / "scripts").is_dir()
|
||||
and (candidate / "data").is_dir()
|
||||
):
|
||||
return candidate
|
||||
raise RuntimeError(f"cannot locate project root from {module_dir}")
|
||||
|
||||
|
||||
PROJECT_ROOT = _find_project_root(Path(__file__).resolve().parent)
|
||||
DATA_ROOT = PROJECT_ROOT / "data"
|
||||
RESULTS_ROOT = PROJECT_ROOT / "results"
|
||||
ENV_FILE = PROJECT_ROOT / ".env"
|
||||
|
||||
DYNAMIC_RESULTS_ROOT = RESULTS_ROOT / "dynamic-optimization"
|
||||
TRACE_ROOT = DYNAMIC_RESULTS_ROOT / "traces"
|
||||
RAW_TRACE_ROOT = TRACE_ROOT / "raw_agent_trace"
|
||||
FINAL_SCORE_ROOT = TRACE_ROOT / "final-score"
|
||||
COMPILED_SKILL_ROOT = DYNAMIC_RESULTS_ROOT / "compiled-skills"
|
||||
|
||||
|
||||
def project_path(value: Path) -> Path:
|
||||
"""将命令行相对路径稳定地解释为项目根目录下的路径。"""
|
||||
|
||||
expanded = value.expanduser()
|
||||
return (expanded if expanded.is_absolute() else PROJECT_ROOT / expanded).resolve()
|
||||
|
||||
|
||||
def default_outputs(trace_input: Path, compile_type: str) -> tuple[Path, Path]:
|
||||
"""按默认原始轨迹树中的身份生成评分和候选产物目录。"""
|
||||
|
||||
try:
|
||||
relative = trace_input.resolve().relative_to(RAW_TRACE_ROOT)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
"轨迹不在默认数据树中,请同时提供 --score-output 和 --output"
|
||||
) from exc
|
||||
identity = relative.parent / compile_type
|
||||
return FINAL_SCORE_ROOT / identity, COMPILED_SKILL_ROOT / identity
|
||||
@@ -0,0 +1,319 @@
|
||||
"""动态编译主流水线;本模块只负责阶段编排。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .cache import (
|
||||
SCORE_ARTIFACTS,
|
||||
fingerprint,
|
||||
load_candidate_cache,
|
||||
load_maps_cache,
|
||||
load_reduction_cache,
|
||||
load_score_cache,
|
||||
trace_fingerprint,
|
||||
write_manifest,
|
||||
)
|
||||
from .optimization.analyzer import SemanticAnalyzer, SemanticClient
|
||||
from .optimization.patch import apply_patches
|
||||
from .optimization.selection import relative_high_low
|
||||
from .paths import default_outputs, project_path
|
||||
from .scoring.agentrm import (
|
||||
DEFAULT_BATCH_SIZE,
|
||||
DEFAULT_CONCURRENCY,
|
||||
DEFAULT_MAX_LENGTH,
|
||||
DEFAULT_RM_API_URL,
|
||||
DEFAULT_TIMEOUT,
|
||||
AgentRM,
|
||||
)
|
||||
from .scoring.service import TraceScorer
|
||||
from .scoring.pre_score import PRE_SCORE_VERSION, PreScorer, RelevanceJudge
|
||||
from .storage import (
|
||||
atomic_write_json,
|
||||
package_hash,
|
||||
read_jsonl,
|
||||
sha256_file,
|
||||
)
|
||||
from .traces.benchflow import load_benchflow_traces
|
||||
|
||||
|
||||
GROUP_SIZE = 3
|
||||
SCORE_VERSION = 1
|
||||
MAP_VERSION = 1
|
||||
REDUCTION_VERSION = 1
|
||||
PATCH_VERSION = 1
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"[dynamic_compile.fast] {message}", file=sys.stderr, flush=True)
|
||||
|
||||
|
||||
def _implementation_name(value: object | None, default: str) -> str:
|
||||
if value is None:
|
||||
return default
|
||||
return type(value).__module__ + "." + type(value).__qualname__
|
||||
|
||||
|
||||
def run_pipeline(
|
||||
trace_input: Path,
|
||||
skill_package: Path,
|
||||
*,
|
||||
score_output: Path | None = None,
|
||||
output: Path | None = None,
|
||||
model: str = "opencode/deepseek-v4-pro",
|
||||
max_parallel: int = 3,
|
||||
rm_api_url: str | None = None,
|
||||
rm_max_length: int = DEFAULT_MAX_LENGTH,
|
||||
rm_timeout: float = DEFAULT_TIMEOUT,
|
||||
rm_concurrency: int = DEFAULT_CONCURRENCY,
|
||||
rm_batch_size: int = DEFAULT_BATCH_SIZE,
|
||||
force: bool = False,
|
||||
analyzer: SemanticAnalyzer | None = None,
|
||||
pre_scorer: PreScorer | None = None,
|
||||
agentrm: AgentRM | None = None,
|
||||
) -> Path:
|
||||
"""从历史 BenchFlow 轨迹生成一个候选 Skill 包。"""
|
||||
|
||||
trace_input = project_path(trace_input)
|
||||
skill_package = project_path(skill_package)
|
||||
if not (skill_package / "SKILL.md").is_file():
|
||||
raise ValueError("skill package must contain a root SKILL.md")
|
||||
if max_parallel < 1:
|
||||
raise ValueError("max_parallel must be at least 1")
|
||||
if agentrm is None and min(
|
||||
rm_max_length, rm_timeout, rm_concurrency, rm_batch_size
|
||||
) <= 0:
|
||||
raise ValueError("AgentRM numeric options must be positive")
|
||||
|
||||
traces = load_benchflow_traces(trace_input)
|
||||
identities = {(trace.task_name, trace.compile_type) for trace in traces}
|
||||
if len(identities) != 1:
|
||||
raise ValueError(
|
||||
f"trace input must contain one task and compile type: {sorted(identities)}"
|
||||
)
|
||||
if len(traces) < GROUP_SIZE * 2:
|
||||
raise ValueError(f"at least {GROUP_SIZE * 2} traces are required")
|
||||
if len({trace.test_name for trace in traces}) != len(traces):
|
||||
raise ValueError("trace input contains duplicate test names")
|
||||
|
||||
_, compile_type = next(iter(identities))
|
||||
if score_output is None or output is None:
|
||||
default_score, default_output = default_outputs(trace_input, compile_type)
|
||||
score_dir = project_path(score_output) if score_output else default_score
|
||||
output_dir = project_path(output) if output else default_output
|
||||
if output_dir == skill_package or output_dir.is_relative_to(skill_package):
|
||||
raise ValueError("output directory must not be inside the input skill package")
|
||||
score_dir.mkdir(parents=True, exist_ok=True)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
traces_hash = trace_fingerprint(traces)
|
||||
resolved_rm_url = rm_api_url or os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
|
||||
score_config: dict[str, Any] = {
|
||||
"score_version": SCORE_VERSION,
|
||||
"pre_score_version": PRE_SCORE_VERSION,
|
||||
"model": model,
|
||||
"pre_scorer": _implementation_name(pre_scorer, "PreScorer/RelevanceJudge"),
|
||||
"agentrm": _implementation_name(agentrm, "AgentRM/HttpAgentRMBackend"),
|
||||
"rm_api_url": resolved_rm_url,
|
||||
"rm_max_length": rm_max_length,
|
||||
"rm_timeout": rm_timeout,
|
||||
"rm_concurrency": rm_concurrency,
|
||||
"rm_batch_size": rm_batch_size,
|
||||
}
|
||||
score_input_hash = fingerprint({"traces": traces_hash, "config": score_config})
|
||||
|
||||
_log(f"loaded {len(traces)} traces from {trace_input}")
|
||||
cached_score = load_score_cache(
|
||||
traces, score_dir, input_hash=score_input_hash, config=score_config
|
||||
)
|
||||
if cached_score is not None:
|
||||
scores, score_rows = cached_score
|
||||
_log("reusing complete, input-matched scoring cache")
|
||||
else:
|
||||
scoring = TraceScorer(
|
||||
pre_scorer or PreScorer(RelevanceJudge(model=model), max_parallel),
|
||||
agentrm
|
||||
or AgentRM(
|
||||
api_url=resolved_rm_url,
|
||||
max_length=rm_max_length,
|
||||
timeout=rm_timeout,
|
||||
concurrency=rm_concurrency,
|
||||
batch_size=rm_batch_size,
|
||||
),
|
||||
)
|
||||
_log("scoring traces")
|
||||
scores = scoring.score_all(traces, score_dir)
|
||||
score_rows = read_jsonl(score_dir / "effective_scores.jsonl")
|
||||
write_manifest(
|
||||
score_dir,
|
||||
".score-cache.json",
|
||||
stage="score",
|
||||
input_hash=score_input_hash,
|
||||
config=score_config,
|
||||
artifacts=SCORE_ARTIFACTS,
|
||||
)
|
||||
|
||||
high, low = relative_high_low(scores, count=GROUP_SIZE)
|
||||
selected = high + low
|
||||
semantic: SemanticAnalyzer | None = analyzer
|
||||
|
||||
def get_semantic() -> SemanticAnalyzer:
|
||||
nonlocal semantic
|
||||
if semantic is None:
|
||||
semantic = SemanticAnalyzer(SemanticClient(model), max_parallel)
|
||||
return semantic
|
||||
|
||||
semantic_implementation = _implementation_name(analyzer, "SemanticAnalyzer/SemanticClient")
|
||||
map_config = {
|
||||
"map_version": MAP_VERSION,
|
||||
"model": model,
|
||||
"semantic_implementation": semantic_implementation,
|
||||
"max_parallel": max_parallel,
|
||||
}
|
||||
map_input_hash = fingerprint(
|
||||
{
|
||||
"traces": traces_hash,
|
||||
"selected": selected,
|
||||
"scores": {trace_id: scores[trace_id] for trace_id in selected},
|
||||
}
|
||||
)
|
||||
maps = load_maps_cache(
|
||||
output_dir,
|
||||
selected,
|
||||
input_hash=map_input_hash,
|
||||
config=map_config,
|
||||
)
|
||||
if maps is not None:
|
||||
_log(f"reusing complete Top {GROUP_SIZE} / Bottom {GROUP_SIZE} Map cache")
|
||||
else:
|
||||
_log(f"mapping Top {GROUP_SIZE} / Bottom {GROUP_SIZE} traces")
|
||||
selected_set = set(selected)
|
||||
maps = get_semantic().map_all(
|
||||
[trace for trace in traces if trace.trace_id in selected_set],
|
||||
scores,
|
||||
set(high),
|
||||
set(low),
|
||||
progress=lambda done, total, trace_id: _log(
|
||||
f"Map {done}/{total}: {trace_id}"
|
||||
),
|
||||
)
|
||||
atomic_write_json(output_dir / "maps.json", maps)
|
||||
write_manifest(
|
||||
output_dir,
|
||||
".maps-cache.json",
|
||||
stage="maps",
|
||||
input_hash=map_input_hash,
|
||||
config=map_config,
|
||||
artifacts=("maps.json",),
|
||||
)
|
||||
|
||||
maps_by_id = {str(item["trace_id"]): item for item in maps}
|
||||
scores_by_id = {str(item["trace_id"]): item for item in score_rows}
|
||||
skill_path = skill_package / "SKILL.md"
|
||||
skill_text = skill_path.read_text(encoding="utf-8")
|
||||
skill_hash = sha256_file(skill_path)
|
||||
skill_package_hash = package_hash(skill_package)
|
||||
reduction_config = {
|
||||
"reduction_version": REDUCTION_VERSION,
|
||||
"model": model,
|
||||
"semantic_implementation": semantic_implementation,
|
||||
}
|
||||
reduction_input_hash = fingerprint(
|
||||
{
|
||||
"traces": traces_hash,
|
||||
"package": skill_package_hash,
|
||||
"maps": [maps_by_id[trace_id] for trace_id in selected],
|
||||
"score_rows": [scores_by_id[trace_id] for trace_id in selected],
|
||||
"high": high,
|
||||
"low": low,
|
||||
}
|
||||
)
|
||||
reduction = None if force else load_reduction_cache(
|
||||
output_dir,
|
||||
input_hash=reduction_input_hash,
|
||||
config=reduction_config,
|
||||
)
|
||||
if reduction is not None:
|
||||
_log("reusing complete Top/Bottom reduction cache")
|
||||
else:
|
||||
_log("reducing Top/Bottom contrast")
|
||||
reduction = get_semantic().reduce(
|
||||
skill_text,
|
||||
[maps_by_id[trace_id] for trace_id in selected],
|
||||
[scores_by_id[trace_id] for trace_id in selected],
|
||||
high,
|
||||
low,
|
||||
[],
|
||||
)
|
||||
atomic_write_json(output_dir / "reduction.json", reduction)
|
||||
write_manifest(
|
||||
output_dir,
|
||||
".reduction-cache.json",
|
||||
stage="reduction",
|
||||
input_hash=reduction_input_hash,
|
||||
config=reduction_config,
|
||||
artifacts=("reduction.json",),
|
||||
)
|
||||
|
||||
candidate = output_dir / "candidate-skill" / skill_package.name
|
||||
candidate_config = {
|
||||
"patch_version": PATCH_VERSION,
|
||||
"model": model,
|
||||
"semantic_implementation": semantic_implementation,
|
||||
"attempts": 3,
|
||||
}
|
||||
candidate_input_hash = fingerprint(
|
||||
{
|
||||
"traces": traces_hash,
|
||||
"package": skill_package_hash,
|
||||
"reduction": reduction,
|
||||
}
|
||||
)
|
||||
cached_candidate = None if force else load_candidate_cache(
|
||||
output_dir,
|
||||
skill_package.name,
|
||||
skill_hash,
|
||||
input_hash=candidate_input_hash,
|
||||
config=candidate_config,
|
||||
)
|
||||
if cached_candidate is not None:
|
||||
_log(f"reusing complete candidate skill: {cached_candidate}")
|
||||
return cached_candidate
|
||||
|
||||
error = ""
|
||||
for attempt in range(3):
|
||||
try:
|
||||
_log(f"generating patch bundle ({attempt + 1}/3)")
|
||||
patches = get_semantic().generate_patches(skill_path, reduction, [], error)
|
||||
candidate_hash = apply_patches(skill_package, candidate, patches)
|
||||
atomic_write_json(
|
||||
output_dir / "patch.json",
|
||||
{
|
||||
"patches": [patch.to_dict() for patch in patches],
|
||||
"skill_hash": skill_hash,
|
||||
"candidate_skill_hash": candidate_hash,
|
||||
},
|
||||
)
|
||||
write_manifest(
|
||||
output_dir,
|
||||
".candidate-cache.json",
|
||||
stage="candidate",
|
||||
input_hash=candidate_input_hash,
|
||||
config=candidate_config,
|
||||
artifacts=(
|
||||
"patch.json",
|
||||
f"candidate-skill/{skill_package.name}",
|
||||
),
|
||||
)
|
||||
_log(f"candidate skill ready: {candidate}")
|
||||
return candidate
|
||||
except (OSError, ValueError, RuntimeError) as exc:
|
||||
error = str(exc)
|
||||
if attempt == 2:
|
||||
raise RuntimeError(
|
||||
f"could not generate an applicable patch: {error}"
|
||||
) from exc
|
||||
raise AssertionError("unreachable")
|
||||
@@ -0,0 +1,191 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
import math
|
||||
import os
|
||||
import time
|
||||
from typing import Any, Protocol
|
||||
|
||||
import requests
|
||||
|
||||
from .identity import composite_id
|
||||
|
||||
|
||||
DEFAULT_MAX_LENGTH = 8192
|
||||
DEFAULT_RM_API_URL = "http://127.0.0.1:28080"
|
||||
DEFAULT_TIMEOUT = 300.0
|
||||
DEFAULT_CONCURRENCY = 8
|
||||
DEFAULT_BATCH_SIZE = 32
|
||||
RETRY_ATTEMPTS = 3
|
||||
|
||||
|
||||
class AgentRMBackend(Protocol):
|
||||
def score(self, requests: list[dict[str, Any]]) -> list[dict[str, Any]]: ...
|
||||
|
||||
|
||||
class HttpAgentRMBackend:
|
||||
"""AgentRM backend backed by the remote ``/score_batch`` API."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
api_url: str | None = None,
|
||||
*,
|
||||
max_length: int = DEFAULT_MAX_LENGTH,
|
||||
timeout: float = DEFAULT_TIMEOUT,
|
||||
concurrency: int = DEFAULT_CONCURRENCY,
|
||||
batch_size: int = DEFAULT_BATCH_SIZE,
|
||||
) -> None:
|
||||
self.api_url = (
|
||||
api_url or os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
|
||||
).rstrip("/")
|
||||
self.max_length = max_length
|
||||
self.timeout = timeout
|
||||
self.concurrency = concurrency
|
||||
self.batch_size = batch_size
|
||||
if not self.api_url:
|
||||
raise ValueError("AgentRM API URL cannot be empty")
|
||||
if max_length <= 0 or timeout <= 0 or concurrency <= 0 or batch_size <= 0:
|
||||
raise ValueError(
|
||||
"AgentRM max_length, timeout, concurrency, and batch_size must be positive"
|
||||
)
|
||||
|
||||
def _post_batch(self, batch: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
payload = {
|
||||
"states": [request["state"] for request in batch],
|
||||
"max_length": self.max_length,
|
||||
}
|
||||
last_error: BaseException | None = None
|
||||
for attempt in range(RETRY_ATTEMPTS):
|
||||
try:
|
||||
response = requests.post(
|
||||
f"{self.api_url}/score_batch",
|
||||
json=payload,
|
||||
timeout=self.timeout,
|
||||
)
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
scores = body.get("scores") if isinstance(body, dict) else None
|
||||
if not isinstance(scores, list) or len(scores) != len(batch):
|
||||
count = len(scores) if isinstance(scores, list) else "invalid"
|
||||
raise ValueError(
|
||||
f"AgentRM returned {count} scores for {len(batch)} states"
|
||||
)
|
||||
if any(not isinstance(score, dict) for score in scores):
|
||||
raise ValueError("AgentRM returned a non-object score item")
|
||||
if any(
|
||||
"score" not in score or "n_tokens" not in score
|
||||
for score in scores
|
||||
):
|
||||
raise ValueError("AgentRM returned a score item with missing fields")
|
||||
return [
|
||||
{
|
||||
"task_name": request["task_name"],
|
||||
"compile_type": request["compile_type"],
|
||||
"test_name": request["test_name"],
|
||||
**score,
|
||||
}
|
||||
for request, score in zip(batch, scores)
|
||||
]
|
||||
except (requests.RequestException, ValueError) as exc:
|
||||
last_error = exc
|
||||
if attempt + 1 < RETRY_ATTEMPTS:
|
||||
time.sleep(2**attempt)
|
||||
assert last_error is not None
|
||||
raise RuntimeError(
|
||||
f"AgentRM request failed after {RETRY_ATTEMPTS} attempts: {last_error}"
|
||||
)
|
||||
|
||||
def score(self, requests_to_score: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
if not requests_to_score:
|
||||
return []
|
||||
batches = [
|
||||
requests_to_score[index : index + self.batch_size]
|
||||
for index in range(0, len(requests_to_score), self.batch_size)
|
||||
]
|
||||
ordered: list[list[dict[str, Any]] | None] = [None] * len(batches)
|
||||
with ThreadPoolExecutor(max_workers=self.concurrency) as pool:
|
||||
futures = {
|
||||
pool.submit(self._post_batch, batch): index
|
||||
for index, batch in enumerate(batches)
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
ordered[futures[future]] = future.result()
|
||||
return [row for batch in ordered if batch is not None for row in batch]
|
||||
|
||||
|
||||
class AgentRM:
|
||||
"""Score AgentRM requests through a validated, replaceable backend."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
backend: AgentRMBackend | None = None,
|
||||
*,
|
||||
api_url: str | None = None,
|
||||
max_length: int = DEFAULT_MAX_LENGTH,
|
||||
timeout: float = DEFAULT_TIMEOUT,
|
||||
concurrency: int = DEFAULT_CONCURRENCY,
|
||||
batch_size: int = DEFAULT_BATCH_SIZE,
|
||||
) -> None:
|
||||
self.backend = (
|
||||
backend
|
||||
if backend is not None
|
||||
else HttpAgentRMBackend(
|
||||
api_url,
|
||||
max_length=max_length,
|
||||
timeout=timeout,
|
||||
concurrency=concurrency,
|
||||
batch_size=batch_size,
|
||||
)
|
||||
)
|
||||
|
||||
def score_requests(
|
||||
self, requests_to_score: list[dict[str, Any]]
|
||||
) -> list[dict[str, Any]]:
|
||||
request_keys = [composite_id(request) for request in requests_to_score]
|
||||
if len(set(request_keys)) != len(request_keys):
|
||||
raise ValueError("AgentRM requests contain duplicate identities")
|
||||
|
||||
responses = self.backend.score(requests_to_score)
|
||||
response_by_key: dict[tuple[str, str, str], dict[str, Any]] = {}
|
||||
for response in responses:
|
||||
key = composite_id(response)
|
||||
if key in response_by_key:
|
||||
raise ValueError(f"AgentRM returned duplicate score identity: {key}")
|
||||
response_by_key[key] = response
|
||||
|
||||
requested = set(request_keys)
|
||||
unexpected = set(response_by_key) - requested
|
||||
missing = requested - set(response_by_key)
|
||||
if unexpected:
|
||||
raise ValueError(
|
||||
f"AgentRM returned unexpected score identities: {sorted(unexpected)}"
|
||||
)
|
||||
if missing:
|
||||
raise ValueError(f"AgentRM returned incomplete scores: {sorted(missing)}")
|
||||
|
||||
result = []
|
||||
for request, key in zip(requests_to_score, request_keys):
|
||||
response = response_by_key[key]
|
||||
try:
|
||||
score = float(response["score"])
|
||||
except (KeyError, TypeError, ValueError) as exc:
|
||||
raise ValueError(f"AgentRM returned an invalid score for {key}") from exc
|
||||
if not math.isfinite(score):
|
||||
raise ValueError(f"AgentRM returned a non-finite score for {key}")
|
||||
try:
|
||||
n_tokens = int(response["n_tokens"])
|
||||
except (KeyError, TypeError, ValueError) as exc:
|
||||
raise ValueError(
|
||||
f"AgentRM returned an invalid n_tokens for {key}"
|
||||
) from exc
|
||||
if n_tokens < 0:
|
||||
raise ValueError(f"AgentRM returned a negative n_tokens for {key}")
|
||||
row = {
|
||||
"task_name": str(request["task_name"]),
|
||||
"compile_type": str(request["compile_type"]),
|
||||
"test_name": str(request["test_name"]),
|
||||
"score": score,
|
||||
"n_tokens": n_tokens,
|
||||
}
|
||||
result.append(row)
|
||||
return result
|
||||
@@ -0,0 +1,13 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from ..models import TraceKey
|
||||
|
||||
|
||||
ScoreKey = tuple[str, str, str]
|
||||
|
||||
|
||||
def composite_id(row: dict[str, Any]) -> ScoreKey:
|
||||
return TraceKey.from_record(row).as_tuple()
|
||||
return result
|
||||
@@ -0,0 +1,254 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from dataclasses import asdict, dataclass, replace
|
||||
from typing import Any, Protocol
|
||||
|
||||
from ..models import RolloutTrace
|
||||
from ..optimization.analyzer import SemanticClient
|
||||
|
||||
|
||||
RELEVANCE_MODEL = "opencode/deepseek-v4-pro"
|
||||
RELEVANCE_CONFIDENCE_THRESHOLD = 0.80
|
||||
TIMEOUT_SCORE = 0.0
|
||||
IRRELEVANT_SCORE = 0.1
|
||||
MAX_EFFICIENCY_PENALTY = 0.08
|
||||
TIME_COST_WEIGHT = 0.80
|
||||
TOOL_COST_WEIGHT = 0.20
|
||||
EFFICIENCY_WINSOR_QUANTILE = 0.90
|
||||
PRE_SCORE_VERSION = 3
|
||||
|
||||
|
||||
class RelevanceClient(Protocol):
|
||||
def json(self, system: str, user: str, attempts: int = 3) -> dict[str, Any]: ...
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PreScoreResult:
|
||||
trace_id: str
|
||||
route: str
|
||||
score: float | None
|
||||
reason: str
|
||||
relevance_label: str | None = None
|
||||
relevance_confidence: float | None = None
|
||||
efficiency_cost: float = 0.0
|
||||
efficiency_penalty: float = 0.0
|
||||
scoring_version: int = PRE_SCORE_VERSION
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "PreScoreResult":
|
||||
return cls(
|
||||
trace_id=str(value["trace_id"]),
|
||||
route=str(value["route"]),
|
||||
score=float(value["score"]) if value.get("score") is not None else None,
|
||||
reason=str(value["reason"]),
|
||||
relevance_label=(
|
||||
str(value["relevance_label"])
|
||||
if value.get("relevance_label") is not None
|
||||
else None
|
||||
),
|
||||
relevance_confidence=(
|
||||
float(value["relevance_confidence"])
|
||||
if value.get("relevance_confidence") is not None
|
||||
else None
|
||||
),
|
||||
efficiency_cost=float(value.get("efficiency_cost", 0.0)),
|
||||
efficiency_penalty=float(value.get("efficiency_penalty", 0.0)),
|
||||
scoring_version=int(value.get("scoring_version", 1)),
|
||||
)
|
||||
|
||||
|
||||
def adjust_agent_score(score: float, pre_score: PreScoreResult) -> float:
|
||||
"""Apply the batch-relative efficiency penalty to an AgentRM quality score."""
|
||||
if pre_score.route != "agentrm":
|
||||
return score
|
||||
return max(IRRELEVANT_SCORE, score - pre_score.efficiency_penalty)
|
||||
|
||||
|
||||
def _quantile(values: list[float], quantile: float) -> float:
|
||||
ordered = sorted(values)
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
position = (len(ordered) - 1) * quantile
|
||||
lower = math.floor(position)
|
||||
upper = math.ceil(position)
|
||||
if lower == upper:
|
||||
return ordered[lower]
|
||||
fraction = position - lower
|
||||
return ordered[lower] + fraction * (ordered[upper] - ordered[lower])
|
||||
|
||||
|
||||
def _magnitude_costs(
|
||||
values: dict[str, float],
|
||||
*,
|
||||
transform=lambda value: value,
|
||||
power: float = 1.0,
|
||||
) -> dict[str, float]:
|
||||
if len(values) < 2:
|
||||
return {trace_id: 0.0 for trace_id in values}
|
||||
transformed = {trace_id: float(transform(value)) for trace_id, value in values.items()}
|
||||
floor = min(transformed.values())
|
||||
ceiling = _quantile(list(transformed.values()), EFFICIENCY_WINSOR_QUANTILE)
|
||||
if ceiling <= floor:
|
||||
return {trace_id: 0.0 for trace_id in values}
|
||||
scale = ceiling - floor
|
||||
return {
|
||||
trace_id: min(1.0, max(0.0, (value - floor) / scale)) ** power
|
||||
for trace_id, value in transformed.items()
|
||||
}
|
||||
|
||||
|
||||
class RelevanceJudge:
|
||||
def __init__(
|
||||
self,
|
||||
client: RelevanceClient | None = None,
|
||||
model: str = RELEVANCE_MODEL,
|
||||
):
|
||||
self.model = model
|
||||
self.client = client or SemanticClient(model)
|
||||
|
||||
def judge(self, task_prompt: str, final_output: str) -> tuple[str, float]:
|
||||
result = self.client.json(
|
||||
"Judge only whether an agent's last output is relevant to its task. Return JSON only.",
|
||||
f"""Classify the last agent output as relevant or irrelevant to the task.
|
||||
|
||||
An output is relevant if it attempts, plans, discusses, or reports work on the requested task, even when it is wrong, incomplete, brief, malformed, or lacks a final answer. Mark it irrelevant only when it clearly addresses a materially different task or topic. Do not judge correctness, completeness, or answer quality.
|
||||
|
||||
Return exactly {{"label":"relevant"|"irrelevant","confidence":number}} where confidence is between 0 and 1.
|
||||
|
||||
Input:
|
||||
{json.dumps({"task_prompt": task_prompt, "last_agent_output": final_output}, ensure_ascii=False)}""",
|
||||
)
|
||||
if not isinstance(result, dict) or set(result) != {"label", "confidence"}:
|
||||
raise ValueError("relevance response must contain exactly label and confidence")
|
||||
label = result.get("label")
|
||||
confidence = result.get("confidence")
|
||||
if label not in {"relevant", "irrelevant"}:
|
||||
raise ValueError("relevance label must be relevant or irrelevant")
|
||||
if isinstance(confidence, bool) or not isinstance(confidence, (int, float)):
|
||||
raise ValueError("relevance confidence must be a number")
|
||||
confidence = float(confidence)
|
||||
if not math.isfinite(confidence) or not 0.0 <= confidence <= 1.0:
|
||||
raise ValueError("relevance confidence must be between 0 and 1")
|
||||
return label, confidence
|
||||
|
||||
|
||||
class PreScorer:
|
||||
def __init__(
|
||||
self,
|
||||
judge: RelevanceJudge,
|
||||
max_parallel: int = 3,
|
||||
confidence_threshold: float = RELEVANCE_CONFIDENCE_THRESHOLD,
|
||||
):
|
||||
self.judge = judge
|
||||
self.max_parallel = max(1, max_parallel)
|
||||
self.confidence_threshold = confidence_threshold
|
||||
|
||||
@staticmethod
|
||||
def _first_user_message(trace: RolloutTrace) -> str:
|
||||
return next(
|
||||
(
|
||||
str(message.get("content", "")).strip()
|
||||
for message in trace.state
|
||||
if isinstance(message, dict)
|
||||
and message.get("role") == "user"
|
||||
and str(message.get("content", "")).strip()
|
||||
),
|
||||
"",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _last_assistant_message(trace: RolloutTrace) -> str:
|
||||
return next(
|
||||
(
|
||||
str(message.get("content", "")).strip()
|
||||
for message in reversed(trace.state)
|
||||
if isinstance(message, dict)
|
||||
and message.get("role") == "assistant"
|
||||
and str(message.get("content", "")).strip()
|
||||
),
|
||||
"",
|
||||
)
|
||||
|
||||
def score_one(self, trace: RolloutTrace) -> PreScoreResult:
|
||||
if trace.timed_out:
|
||||
return PreScoreResult(trace.trace_id, "fixed_score", TIMEOUT_SCORE, "hard_timeout")
|
||||
|
||||
final_output = self._last_assistant_message(trace)
|
||||
if not final_output:
|
||||
return PreScoreResult(trace.trace_id, "agentrm", None, "no_assistant_output")
|
||||
|
||||
task_prompt = self._first_user_message(trace)
|
||||
try:
|
||||
label, confidence = self.judge.judge(task_prompt, final_output)
|
||||
except Exception:
|
||||
return PreScoreResult(trace.trace_id, "agentrm", None, "judge_failed")
|
||||
|
||||
if label == "irrelevant" and confidence >= self.confidence_threshold:
|
||||
return PreScoreResult(
|
||||
trace.trace_id,
|
||||
"fixed_score",
|
||||
IRRELEVANT_SCORE,
|
||||
"strongly_irrelevant",
|
||||
label,
|
||||
confidence,
|
||||
)
|
||||
reason = "low_confidence" if label == "irrelevant" else "relevant"
|
||||
return PreScoreResult(
|
||||
trace.trace_id, "agentrm", None, reason, label, confidence
|
||||
)
|
||||
|
||||
def score_all(self, traces: list[RolloutTrace]) -> list[PreScoreResult]:
|
||||
results: list[PreScoreResult | None] = [None] * len(traces)
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {
|
||||
pool.submit(self.score_one, trace): index
|
||||
for index, trace in enumerate(traces)
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
results[futures[future]] = future.result()
|
||||
completed = [result for result in results if result is not None]
|
||||
by_id = {trace.trace_id: trace for trace in traces}
|
||||
eligible = {
|
||||
result.trace_id for result in completed if result.route == "agentrm"
|
||||
}
|
||||
durations = {
|
||||
trace_id: float(by_id[trace_id].metadata["agent_execution_seconds"])
|
||||
for trace_id in eligible
|
||||
if isinstance(
|
||||
by_id[trace_id].metadata.get("agent_execution_seconds"), (int, float)
|
||||
)
|
||||
}
|
||||
tool_calls = {
|
||||
trace_id: float(by_id[trace_id].metadata["tool_calls"])
|
||||
for trace_id in eligible
|
||||
if isinstance(by_id[trace_id].metadata.get("tool_calls"), (int, float))
|
||||
}
|
||||
duration_cost = _magnitude_costs(durations, power=2.0)
|
||||
tool_cost = _magnitude_costs(tool_calls, transform=math.log1p)
|
||||
adjusted = []
|
||||
for result in completed:
|
||||
components = []
|
||||
if result.trace_id in duration_cost:
|
||||
components.append((TIME_COST_WEIGHT, duration_cost[result.trace_id]))
|
||||
if result.trace_id in tool_cost:
|
||||
components.append((TOOL_COST_WEIGHT, tool_cost[result.trace_id]))
|
||||
total_weight = sum(weight for weight, _ in components)
|
||||
cost = (
|
||||
sum(weight * value for weight, value in components) / total_weight
|
||||
if total_weight
|
||||
else 0.0
|
||||
)
|
||||
adjusted.append(
|
||||
replace(
|
||||
result,
|
||||
efficiency_cost=cost,
|
||||
efficiency_penalty=MAX_EFFICIENCY_PENALTY * cost,
|
||||
)
|
||||
)
|
||||
return adjusted
|
||||
@@ -0,0 +1,75 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import RolloutTrace
|
||||
from ..storage import atomic_write_jsonl
|
||||
from .agentrm import AgentRM
|
||||
from .pre_score import PreScorer, adjust_agent_score
|
||||
from .identity import composite_id
|
||||
|
||||
|
||||
class TraceScorer:
|
||||
"""Resolve pre-scores and AgentRM scores into one effective score per trace."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pre_scorer: PreScorer,
|
||||
agentrm: AgentRM | None = None,
|
||||
):
|
||||
self.pre_scorer = pre_scorer
|
||||
self.agentrm = agentrm if agentrm is not None else AgentRM()
|
||||
|
||||
def score_all(
|
||||
self,
|
||||
traces: list[RolloutTrace],
|
||||
final_score_dir: Path,
|
||||
agentrm_dir: Path | None = None,
|
||||
) -> dict[str, float]:
|
||||
final_score_dir.mkdir(parents=True, exist_ok=True)
|
||||
agentrm_dir = agentrm_dir or final_score_dir
|
||||
agentrm_dir.mkdir(parents=True, exist_ok=True)
|
||||
results = self.pre_scorer.score_all(traces)
|
||||
pre_scores = {result.trace_id: result for result in results}
|
||||
if set(pre_scores) != {trace.trace_id for trace in traces}:
|
||||
raise ValueError("pre-scorer returned incomplete or duplicate results")
|
||||
atomic_write_jsonl(
|
||||
final_score_dir / "pre_scores.jsonl",
|
||||
[pre_scores[trace.trace_id].to_dict() for trace in traces],
|
||||
)
|
||||
agentrm_traces = [
|
||||
trace for trace in traces if pre_scores[trace.trace_id].route == "agentrm"
|
||||
]
|
||||
requests = [trace.agentrm_request() for trace in agentrm_traces]
|
||||
local_scores = agentrm_dir / "agentrm_scores.jsonl"
|
||||
found = self.agentrm.score_requests(requests)
|
||||
atomic_write_jsonl(local_scores, found)
|
||||
|
||||
by_key = {composite_id(row): float(row["score"]) for row in found}
|
||||
resolved = []
|
||||
scores: dict[str, float] = {}
|
||||
for trace in traces:
|
||||
pre_score = pre_scores[trace.trace_id]
|
||||
if pre_score.route == "fixed_score":
|
||||
if pre_score.score is None:
|
||||
raise ValueError(f"fixed pre-score is missing for {trace.trace_id}")
|
||||
score = pre_score.score
|
||||
source = "pre_score"
|
||||
else:
|
||||
score = adjust_agent_score(by_key[trace.key.as_tuple()], pre_score)
|
||||
source = "agentrm"
|
||||
scores[trace.trace_id] = score
|
||||
resolved.append({
|
||||
"task_name": trace.task_name,
|
||||
"compile_type": trace.compile_type,
|
||||
"test_name": trace.test_name,
|
||||
"trace_id": trace.trace_id,
|
||||
"effective_score": score,
|
||||
"score_source": source,
|
||||
"pre_score_reason": pre_score.reason,
|
||||
"agentrm_score": by_key.get(trace.key.as_tuple()),
|
||||
"efficiency_cost": pre_score.efficiency_cost,
|
||||
"efficiency_penalty": pre_score.efficiency_penalty,
|
||||
})
|
||||
atomic_write_jsonl(final_score_dir / "effective_scores.jsonl", resolved)
|
||||
return scores
|
||||
@@ -0,0 +1,121 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
MAX_CHARS = 2000
|
||||
LAST_MSG_BUDGET = 4000
|
||||
HEAD_RATIO = 0.5
|
||||
|
||||
|
||||
def compact(content: str, budget: int) -> str:
|
||||
if len(content) <= budget:
|
||||
return content
|
||||
head = int(budget * HEAD_RATIO)
|
||||
tail = budget - head - 50
|
||||
return (
|
||||
content[:head]
|
||||
+ f"\n...[truncated {len(content)-head-tail} chars]...\n"
|
||||
+ content[-tail:]
|
||||
)
|
||||
|
||||
|
||||
def _unwrap_text_content(value: str) -> str:
|
||||
prefix = "@{type=text; text="
|
||||
if value.startswith(prefix) and value.endswith("}"):
|
||||
return value[len(prefix) : -1]
|
||||
return value
|
||||
|
||||
|
||||
def _tool_result_text(event: dict[str, Any]) -> str:
|
||||
parts: list[str] = []
|
||||
content_items = event.get("content")
|
||||
if isinstance(content_items, list):
|
||||
for item in content_items:
|
||||
if not isinstance(item, dict) or item.get("type") != "content":
|
||||
continue
|
||||
value = item.get("content")
|
||||
if isinstance(value, str) and value:
|
||||
parts.append(_unwrap_text_content(value))
|
||||
elif (
|
||||
isinstance(value, dict)
|
||||
and value.get("type") == "text"
|
||||
and isinstance(value.get("text"), str)
|
||||
and value["text"]
|
||||
):
|
||||
parts.append(value["text"])
|
||||
return "\n".join(parts) if parts else "(no output)"
|
||||
|
||||
|
||||
def acp_events_to_state(events: Iterable[dict[str, Any]]) -> list[dict[str, str]]:
|
||||
merged: list[dict[str, str]] = []
|
||||
for event in events:
|
||||
event_type = event.get("type")
|
||||
role: str | None = None
|
||||
content: str | None = None
|
||||
if event_type == "user_message":
|
||||
role, content = "user", event.get("text")
|
||||
elif event_type == "agent_thought":
|
||||
role, content = "assistant", event.get("text")
|
||||
elif event_type == "agent_message":
|
||||
role, content = "assistant", event.get("text")
|
||||
if content == "":
|
||||
continue
|
||||
elif event_type == "tool_call":
|
||||
role = "user"
|
||||
kind = event.get("kind", "other")
|
||||
status = event.get("status", "unknown")
|
||||
content = f"Tool result ({kind}; {status}):\n{_tool_result_text(event)}"
|
||||
elif event_type == "agent_timeout":
|
||||
continue
|
||||
else:
|
||||
continue
|
||||
|
||||
if not isinstance(content, str):
|
||||
raise ValueError(f"{event_type} event has a non-string text/content value")
|
||||
if merged and merged[-1]["role"] == role:
|
||||
merged[-1]["content"] += "\n" + content
|
||||
else:
|
||||
merged.append({"role": role, "content": content})
|
||||
|
||||
if not merged:
|
||||
raise ValueError("trajectory produced an empty state")
|
||||
if merged[0]["role"] != "user":
|
||||
raise ValueError("state must start with a user message")
|
||||
|
||||
seen: set[str] = set()
|
||||
last_index = len(merged) - 1
|
||||
for index, message in enumerate(merged):
|
||||
budget = LAST_MSG_BUDGET if index == last_index else MAX_CHARS
|
||||
content = compact(message["content"], budget)
|
||||
if len(content) > 200:
|
||||
digest = hashlib.md5(content.encode("utf-8")).hexdigest()
|
||||
if digest in seen:
|
||||
content = "[same as previous tool result]"
|
||||
else:
|
||||
seen.add(digest)
|
||||
message["content"] = content
|
||||
return merged
|
||||
|
||||
|
||||
def read_acp_events(path: Path) -> list[dict[str, Any]]:
|
||||
events: list[dict[str, Any]] = []
|
||||
with path.open("r", encoding="utf-8") as handle:
|
||||
for line_number, line in enumerate(handle, 1):
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"invalid JSON on line {line_number}: {exc.msg}") from exc
|
||||
if not isinstance(event, dict):
|
||||
raise ValueError(f"line {line_number} is not a JSON object")
|
||||
events.append(event)
|
||||
return events
|
||||
|
||||
|
||||
def read_acp_state(path: Path) -> list[dict[str, str]]:
|
||||
return acp_events_to_state(read_acp_events(path))
|
||||
@@ -0,0 +1,88 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
from .paths import ENV_FILE
|
||||
|
||||
|
||||
def load_project_env() -> None:
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
except ImportError:
|
||||
return
|
||||
if ENV_FILE.is_file():
|
||||
load_dotenv(ENV_FILE, override=False)
|
||||
|
||||
|
||||
def atomic_write_text(path: Path, text: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd, tmp = tempfile.mkstemp(prefix=f".{path.name}.", dir=path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle:
|
||||
handle.write(text)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
os.replace(tmp, path)
|
||||
finally:
|
||||
if os.path.exists(tmp):
|
||||
os.unlink(tmp)
|
||||
|
||||
|
||||
def atomic_write_json(path: Path, value: Any) -> None:
|
||||
atomic_write_text(path, json.dumps(value, ensure_ascii=False, indent=2) + "\n")
|
||||
|
||||
|
||||
def atomic_write_jsonl(path: Path, rows: Iterable[dict[str, Any]]) -> None:
|
||||
atomic_write_text(path, "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows))
|
||||
|
||||
|
||||
def load_json(path: Path, default: Any = None) -> Any:
|
||||
if not path.exists():
|
||||
return default
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def read_jsonl(path: Path) -> list[dict[str, Any]]:
|
||||
if not path.exists():
|
||||
return []
|
||||
rows: list[dict[str, Any]] = []
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
for line_no, line in enumerate(handle, 1):
|
||||
if not line.strip():
|
||||
continue
|
||||
value = json.loads(line)
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError(f"{path}:{line_no}: expected a JSON object")
|
||||
rows.append(value)
|
||||
return rows
|
||||
|
||||
|
||||
def sha256_bytes(data: bytes) -> str:
|
||||
return hashlib.sha256(data).hexdigest()
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
return sha256_bytes(path.read_bytes())
|
||||
|
||||
|
||||
def sha256_text(text: str) -> str:
|
||||
return sha256_bytes(text.encode("utf-8"))
|
||||
|
||||
|
||||
def package_manifest(root: Path) -> list[dict[str, Any]]:
|
||||
return [
|
||||
{"path": str(p.relative_to(root)), "sha256": sha256_file(p), "bytes": p.stat().st_size}
|
||||
for p in sorted(root.rglob("*"))
|
||||
if p.is_file()
|
||||
]
|
||||
|
||||
|
||||
def package_hash(root: Path) -> str:
|
||||
payload = json.dumps(package_manifest(root), sort_keys=True, separators=(",", ":"))
|
||||
return sha256_text(payload)
|
||||
@@ -0,0 +1,133 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from ..models import RolloutTrace
|
||||
from ..scoring.state import acp_events_to_state, read_acp_events
|
||||
|
||||
|
||||
VARIANT_TO_COMPILE_TYPE = {
|
||||
"model_skill": "model_compile",
|
||||
"ori_skill": "ori",
|
||||
}
|
||||
|
||||
|
||||
def event_to_state(trajectory_path: Path) -> tuple[list[dict[str, str]], bool]:
|
||||
events = read_acp_events(trajectory_path)
|
||||
skill_invoked = any(
|
||||
event.get("type") == "tool_call"
|
||||
and any(
|
||||
str(event.get(field, "")).strip().lower() == "skill"
|
||||
for field in ("title", "kind")
|
||||
)
|
||||
for event in events
|
||||
)
|
||||
return acp_events_to_state(events), skill_invoked
|
||||
|
||||
|
||||
def trajectory_for_test(test_dir: Path) -> Path | None:
|
||||
candidates = sorted(test_dir.rglob("acp_trajectory.jsonl"))
|
||||
if not candidates:
|
||||
return None
|
||||
canonical = [path for path in candidates if "trajectory" in path.parts]
|
||||
return canonical[0] if canonical else candidates[0]
|
||||
|
||||
|
||||
def _result_for_trajectory(trajectory_path: Path) -> dict[str, Any]:
|
||||
result_path = trajectory_path.parent.parent / "result.json"
|
||||
if not result_path.is_file():
|
||||
raise ValueError(f"missing structured BenchFlow result: {result_path}")
|
||||
value = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError(f"BenchFlow result must be a JSON object: {result_path}")
|
||||
return value
|
||||
|
||||
|
||||
def load_benchflow_trace(
|
||||
test_dir: Path,
|
||||
task_name: str,
|
||||
compile_type: str,
|
||||
) -> RolloutTrace:
|
||||
trajectory = trajectory_for_test(test_dir)
|
||||
if trajectory is None:
|
||||
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
|
||||
result = _result_for_trajectory(trajectory)
|
||||
agent_timeout = result.get("agent_timeout_info")
|
||||
idle_timeout = result.get("idle_timeout_info")
|
||||
timeout_info = (
|
||||
agent_timeout
|
||||
if isinstance(agent_timeout, dict)
|
||||
else idle_timeout if isinstance(idle_timeout, dict) else None
|
||||
)
|
||||
timed_out = timeout_info is not None
|
||||
metadata: dict[str, Any] = {
|
||||
"source": str(test_dir.resolve()),
|
||||
"termination": "timeout" if timed_out else "completed",
|
||||
}
|
||||
timing = result.get("timing") if isinstance(result.get("timing"), dict) else {}
|
||||
execution_seconds = timing.get("agent_execution")
|
||||
if execution_seconds is None and isinstance(timeout_info, dict):
|
||||
execution_seconds = timeout_info.get(
|
||||
"wall_clock_elapsed_sec", timeout_info.get("timeout_sec")
|
||||
)
|
||||
if isinstance(execution_seconds, (int, float)):
|
||||
metadata["agent_execution_seconds"] = float(execution_seconds)
|
||||
if isinstance(result.get("n_tool_calls"), int):
|
||||
metadata["tool_calls"] = result["n_tool_calls"]
|
||||
if timed_out:
|
||||
metadata["timeout_reason"] = timeout_info.get("reason")
|
||||
metadata["timeout_seconds"] = timeout_info.get(
|
||||
"timeout_sec", timeout_info.get("idle_timeout_sec")
|
||||
)
|
||||
metadata["partial_trajectory"] = bool(result.get("partial_trajectory", False))
|
||||
metadata["error_category"] = result.get("error_category")
|
||||
state, skill_invoked = event_to_state(trajectory)
|
||||
return RolloutTrace(
|
||||
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
|
||||
task_name=task_name,
|
||||
compile_type=compile_type,
|
||||
test_name=test_dir.name,
|
||||
state=state,
|
||||
skill_invoked=skill_invoked,
|
||||
timed_out=timed_out,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
|
||||
def _variant_dirs(input_path: Path) -> list[tuple[Path, str, str]]:
|
||||
if input_path.name in VARIANT_TO_COMPILE_TYPE:
|
||||
return [(
|
||||
input_path,
|
||||
input_path.parent.name,
|
||||
VARIANT_TO_COMPILE_TYPE[input_path.name],
|
||||
)]
|
||||
direct_variants = [
|
||||
(input_path / variant_name, input_path.name, compile_type)
|
||||
for variant_name, compile_type in VARIANT_TO_COMPILE_TYPE.items()
|
||||
if (input_path / variant_name).is_dir()
|
||||
]
|
||||
if direct_variants:
|
||||
return direct_variants
|
||||
variants = []
|
||||
for task_dir in sorted(path for path in input_path.iterdir() if path.is_dir()):
|
||||
for variant_name, compile_type in VARIANT_TO_COMPILE_TYPE.items():
|
||||
variant_dir = task_dir / variant_name
|
||||
if variant_dir.is_dir():
|
||||
variants.append((variant_dir, task_dir.name, compile_type))
|
||||
return variants
|
||||
|
||||
|
||||
def load_benchflow_traces(input_path: Path) -> list[RolloutTrace]:
|
||||
input_path = input_path.resolve()
|
||||
if not input_path.is_dir():
|
||||
raise ValueError(f"BenchFlow input does not exist: {input_path}")
|
||||
variants = _variant_dirs(input_path)
|
||||
if not variants:
|
||||
raise ValueError(f"no model_skill or ori_skill directories under {input_path}")
|
||||
traces = []
|
||||
for variant_dir, task_name, compile_type in variants:
|
||||
for test_dir in sorted(path for path in variant_dir.glob("test-*") if path.is_dir()):
|
||||
traces.append(load_benchflow_trace(test_dir, task_name, compile_type))
|
||||
return traces
|
||||
Reference in New Issue
Block a user