Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+53
View File
@@ -0,0 +1,53 @@
"""Deep 编译流水线的唯一命令行入口。"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from scripts.provider_router import parse_model_reference
from .pipeline import DeepLoop
def _provider_model(value: str) -> str:
try:
return parse_model_reference(value).value
except ValueError as exc:
raise argparse.ArgumentTypeError(str(exc)) from exc
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="python -m scripts.dynamic_compile.deep")
commands = parser.add_subparsers(dest="command", required=True)
run = commands.add_parser("run", help="run Deep Loop from a skill and its BenchFlow traces")
run.add_argument("--skill", type=Path, required=True)
run.add_argument("--traces", type=Path, required=True)
run.add_argument("--output", type=Path)
run.add_argument(
"--model",
required=True,
type=_provider_model,
help="all external model calls use this provider/model",
)
resume = commands.add_parser("resume", help="resume a Deep Loop run")
resume.add_argument("--run", type=Path, required=True)
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
loop = (
DeepLoop.create(args.skill, args.traces, args.output, model=args.model)
if args.command == "run"
else DeepLoop(args.run)
)
print(loop.drive())
return 0
except (OSError, ValueError, RuntimeError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
raise SystemExit(main())
@@ -0,0 +1,216 @@
"""BenchFlow 输入与运行时轨迹适配。"""
from __future__ import annotations
import json
import re
import shutil
import subprocess
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.traces.benchflow import trajectory_for_test
from scripts.dynamic_compile.fast.scoring.state import acp_events_to_state, read_acp_events
@dataclass
class BenchFlowInput:
task_name: str
task_dir: Path
agent: str
model: str
prompt: str
traces: list[RolloutTrace]
def _project_root() -> Path:
return Path(__file__).resolve().parents[4]
def _run_config(test_dir: Path) -> dict[str, Any]:
paths = sorted(test_dir.rglob("config.json"))
if not paths:
raise ValueError(f"missing BenchFlow run config under {test_dir}")
value = json.loads(paths[0].read_text(encoding="utf-8"))
if not isinstance(value, dict):
raise ValueError(f"invalid BenchFlow run config: {paths[0]}")
return value
def _prompt(test_dir: Path) -> str:
paths = sorted(test_dir.rglob("prompts.json"))
if not paths:
raise ValueError(f"missing BenchFlow prompts.json under {test_dir}")
value = json.loads(paths[0].read_text(encoding="utf-8"))
if not isinstance(value, list) or not value or not isinstance(value[0], str):
raise ValueError(f"invalid BenchFlow prompts: {paths[0]}")
return value[0].strip()
def load_runtime_trace(test_dir: Path, task_name: str, compile_type: str) -> RolloutTrace:
"""Load agent runtime evidence without opening verifier/result artifacts."""
trajectory = trajectory_for_test(test_dir)
if trajectory is None:
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
events = read_acp_events(trajectory)
state = acp_events_to_state(events)
skill_invoked = any(
event.get("type") == "tool_call"
and event.get("status") == "completed"
and any(
str(event.get(field, "")).strip().lower() == "skill"
for field in ("title", "kind")
)
for event in events
)
timed_out = any(event.get("type") == "agent_timeout" for event in events)
return RolloutTrace(
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
task_name=task_name,
compile_type=compile_type,
test_name=test_dir.name,
state=state,
skill_invoked=skill_invoked,
timed_out=timed_out,
metadata={
"termination": "timeout" if timed_out else "completed",
"tool_calls": sum(event.get("type") == "tool_call" for event in events),
},
)
def load_benchflow_input(source: Path) -> BenchFlowInput:
source = source.resolve()
tests = sorted(path for path in source.glob("test-*") if path.is_dir())[-5:]
if not tests:
raise ValueError(f"no test-* BenchFlow traces under {source}")
task_name = source.parent.name
traces = [load_runtime_trace(test, task_name, "custom") for test in tests]
configs = [_run_config(test) for test in tests]
agents = {str(item.get("agent", "")).strip() for item in configs}
models = {str(item.get("model", "")).strip() for item in configs}
if "" in agents or len(agents) != 1:
raise ValueError(f"BenchFlow traces do not identify one agent: {sorted(agents)}")
if "" in models or len(models) != 1:
raise ValueError(f"BenchFlow traces do not identify one model: {sorted(models)}")
prompts = {_prompt(test) for test in tests}
if len(prompts) != 1:
raise ValueError(f"BenchFlow traces contain {len(prompts)} different task prompts")
task_dir = _project_root() / "data" / "skills-bench" / "tasks" / task_name
if not (task_dir / "task.md").is_file():
raise ValueError(f"cannot resolve SkillsBench task directory: {task_dir}")
return BenchFlowInput(
task_name, task_dir, next(iter(agents)), next(iter(models)),
next(iter(prompts)), traces,
)
def _slug(value: str) -> str:
return re.sub(r"^-+|-+$", "", re.sub(r"[^a-z0-9]+", "-", value.lower()))
class SkillsBenchDevelopmentRolloutRunner:
"""Development-only runner; verifier output is never projected into traces."""
def __init__(
self,
context: BenchFlowInput,
work_root: Path,
max_parallel: int = 3,
archive_root: Path | None = None,
):
self.context = context
self.work_root = work_root
self.max_parallel = max_parallel
self.archive_root = archive_root
def _variant_dir(self, jobs_root: Path) -> Path:
return (
jobs_root
/ _slug(f"{self.context.agent}-{self.context.model}")
/ self.context.task_name
/ "custom_skill"
)
def _completed_traces(self, variant: Path, batch_id: str) -> list[RolloutTrace]:
completed: list[RolloutTrace] = []
for test in sorted(path for path in variant.glob("test-*") if path.is_dir()):
try:
requirement = json.loads(
(test / "required-skill.json").read_text(encoding="utf-8")
)
if requirement.get("invoked") is not True:
continue
trace = load_runtime_trace(test, self.context.task_name, batch_id)
except (AttributeError, json.JSONDecodeError, OSError, ValueError):
continue
completed.append(trace)
return completed
def artifacts_dir(self, batch_id: str) -> Path | None:
if self.archive_root is None:
return None
return self.archive_root / batch_id / "custom_skill"
def run_batch(
self,
skill_package: Path,
_prompt: str,
batch_id: str,
_task_name: str,
count: int,
progress: Any | None = None,
) -> list[RolloutTrace]:
jobs_root = self.work_root / batch_id
log_path = jobs_root / "runner.log"
variant = self._variant_dir(jobs_root)
archive = self.artifacts_dir(batch_id)
if archive is not None and archive.is_dir():
archived_traces = self._completed_traces(archive, batch_id)
if len(archived_traces) >= count:
traces = archived_traces
missing = 0
else:
variant.parent.mkdir(parents=True, exist_ok=True)
shutil.copytree(archive, variant, dirs_exist_ok=True)
traces = self._completed_traces(variant, batch_id)
missing = max(0, count - len(traces))
else:
traces = self._completed_traces(variant, batch_id)
missing = max(0, count - len(traces))
if missing:
jobs_root.mkdir(parents=True, exist_ok=True)
command = [
"bash", str(_project_root() / "scripts" / "evaluate" / "run-raw-task.sh"),
"--harness", self.context.agent,
"--model", self.context.model,
"--task", str(self.context.task_dir),
"--skill-source", str(skill_package.resolve()),
"--require-skill",
"--repeat", str(missing),
"--max-parallel", str(self.max_parallel),
"--output", str(variant),
]
with log_path.open("a", encoding="utf-8") as handle:
subprocess.run(command, stdout=handle, stderr=subprocess.STDOUT, text=True)
traces = self._completed_traces(variant, batch_id)
if archive is not None and variant.is_dir():
archive.parent.mkdir(parents=True, exist_ok=True)
shutil.copytree(variant, archive, dirs_exist_ok=True)
runner_log = jobs_root / "runner.log"
if runner_log.is_file():
shutil.copy2(runner_log, archive.parent / "runner.log")
traces = self._completed_traces(archive, batch_id)
if len(traces) < count:
raise RuntimeError(
f"BenchFlow produced {len(traces)}/{count} candidate traces; see {log_path}"
)
traces = traces[:count]
for index, trace in enumerate(traces, 1):
trace.trace_id = f"{batch_id}-{index:03d}"
trace.test_name = trace.trace_id
if progress is not None:
progress(index, len(traces), trace.trace_id)
return traces
@@ -0,0 +1,197 @@
"""语义模型评分与局部编辑适配。"""
from __future__ import annotations
import json
import re
from concurrent.futures import ThreadPoolExecutor, as_completed
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
RUBRICS = {
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
}
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
if not isinstance(values, list):
return []
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
return [
value for value in values
if isinstance(value, str)
and _EVIDENCE.fullmatch(value)
and value.startswith(prefixes)
]
class DeepAnalyzer:
def __init__(self, client: SemanticClient, max_parallel: int = 3):
self.client = client
self.max_parallel = max_parallel
@staticmethod
def _trace_payload(traces: list[RolloutTrace]) -> str:
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
def score_column(
self,
skill_text: str,
task: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
dimension: str,
) -> dict[str, CellScore]:
unit_payload = [
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
for unit in units
]
result = self.client.json(
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
Rubric: {RUBRICS[dimension]}
Task prompt:
{task}
Current units:
{json.dumps(unit_payload, ensure_ascii=False)}
Agent traces:
{self._trace_payload(traces)}
Current SKILL.md:
{skill_text}""",
)
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
raise ValueError(f"judge returned an invalid {dimension} column")
expected = {unit.unit_id for unit in units}
column: dict[str, CellScore] = {}
for item in result["scores"]:
if not isinstance(item, dict):
raise ValueError("judge score entries must be objects")
unit_id = str(item.get("unit_id", ""))
score = float(item.get("score"))
evidence = item.get("evidence", [])
if unit_id not in expected or unit_id in column:
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
evidence = valid_evidence(evidence, traces)
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
if set(column) != expected:
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
return column
def compare_cell(
self,
task: str,
incumbent_unit: SkillUnit,
candidate_unit: SkillUnit,
incumbent_traces: list[RolloutTrace],
candidate_traces: list[RolloutTrace],
dimension: str,
) -> dict[str, Any]:
result = self.client.json(
"Compare one incumbent and candidate skill unit. Return JSON only.",
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
Rubric: {RUBRICS[dimension]}
Task prompt:
{task}
Incumbent unit:
{incumbent_unit.text}
Candidate unit:
{candidate_unit.text}
Incumbent runtime facts:
{self._trace_payload(incumbent_traces)}
Candidate runtime facts:
{self._trace_payload(candidate_traces)}""",
)
if not isinstance(result.get("task_relevant"), bool):
raise ValueError("judge returned invalid task relevance")
incumbent_score = float(result.get("incumbent_score"))
candidate_score = float(result.get("candidate_score"))
for score in (incumbent_score, candidate_score):
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
raise ValueError(f"judge returned an invalid paired score: {score}")
regressions = result.get("runtime_regressions")
allowed = {
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
}
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
raise ValueError("judge returned invalid runtime regressions")
return {
"task_relevant": result["task_relevant"],
"incumbent_score": incumbent_score,
"candidate_score": candidate_score,
"candidate_evidence": valid_evidence(
result.get("candidate_evidence"), candidate_traces
),
"runtime_regressions": list(dict.fromkeys(regressions)),
"reason": str(result.get("reason", "")),
}
def score_matrix(
self,
skill_text: str,
task: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
level: str,
existing_columns: dict[str, dict[str, CellScore]] | None = None,
result_callback: Any | None = None,
) -> ScoreMatrix:
columns = dict(existing_columns or {})
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
futures = {
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
for dimension in missing
}
for future in as_completed(futures):
dimension = futures[future]
columns[dimension] = future.result()
if result_callback is not None:
result_callback(dimension, columns[dimension])
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
def generate_edit(
self,
coordinate: Coordinate,
unit: SkillUnit,
cell: CellScore,
rejected: list[dict[str, Any]],
task_prompt: str,
) -> LocalEdit:
result = self.client.json(
"Generate one bounded local edit for an agent skill. Return JSON only.",
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
Target dimension rubric: {RUBRICS[coordinate.dimension]}
Task prompt:
{task_prompt}
Target unit:
{unit.text}
Score: {cell.score}
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
Reason: {cell.reason}
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
)
edit = LocalEdit.from_dict(result)
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
raise ValueError("edit generator changed the target coordinate")
return edit
@@ -0,0 +1,195 @@
"""Markdown 单元解析与编辑边界校验。"""
from __future__ import annotations
from collections import Counter
import re
from .models import SkillUnit
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
rows: list[tuple[int, int, str]] = []
offset = 0
for line in text.splitlines(keepends=True):
rows.append((offset, offset + len(line), line))
offset += len(line)
return rows
def _headings(text: str) -> list[tuple[int, int, int, str]]:
found = []
fence_char = ""
fence_size = 0
for start, end, line in _line_offsets(text):
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if fence_char:
continue
match = _HEADING.match(line)
if match:
found.append((start, end, len(match.group(1)), match.group(2).strip()))
return found
def _frontmatter_end(text: str) -> int:
if not text.startswith("---"):
return 0
lines = text.splitlines(keepends=True)
offset = len(lines[0]) if lines else 0
for line in lines[1:]:
offset += len(line)
if line.strip() == "---":
return offset
return 0
def parse_sections(text: str) -> list[SkillUnit]:
headings = _headings(text)
body_start = _frontmatter_end(text)
body_headings = [item for item in headings if item[0] >= body_start]
title = body_headings[0] if body_headings else None
after_title = title[1] if title else body_start
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
if not candidates:
body = text[body_start:]
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
section_depth = min(item[2] for item in candidates)
peers = [item for item in candidates if item[2] == section_depth]
spans: list[tuple[int, int, str, int | None]] = []
preamble = text[after_title:peers[0][0]]
if preamble.strip():
spans.append((after_title, peers[0][0], "Preamble", section_depth))
for index, heading in enumerate(peers):
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
spans.append((heading[0], end, heading[3], heading[2]))
return [
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
for index, (start, end, heading, depth) in enumerate(spans)
]
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
replacement = preserve_unit_boundary(unit, new_text)
return document[:unit.start] + replacement + document[unit.end:]
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
text = section.text
base = section.start
rows = _line_offsets(text)
blocks: list[tuple[int, int]] = []
start: int | None = None
fence_char = ""
fence_size = 0
for row_start, row_end, line in rows:
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if start is None:
start = row_start
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if not fence_char and not line.strip():
if start is not None:
blocks.append((start, row_start))
start = None
continue
if start is None:
start = row_start
if start is not None:
blocks.append((start, len(text)))
merged: list[tuple[int, int]] = []
index = 0
while index < len(blocks):
start, end = blocks[index]
block = text[start:end]
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
merged.append((start, blocks[index + 1][1]))
index += 2
else:
merged.append((start, end))
index += 1
blocks = merged
units = []
for index, (start, end) in enumerate(blocks):
block = text[start:end]
heading_match = next((item for item in _headings(block)), None)
units.append(SkillUnit(
f"{section.unit_id}.P{index + 1:03d}",
"paragraph",
section.unit_id,
heading_match[3] if heading_match else "",
block,
index,
base + start,
base + end,
heading_match[2] if heading_match else None,
))
return units
def fenced_blocks(text: str) -> Counter[str]:
blocks: list[str] = []
current: list[str] | None = None
fence_char = ""
fence_size = 0
for line in text.splitlines(keepends=True):
fence = _FENCE.match(line)
if current is None:
if fence:
marker = fence.group(1)
fence_char, fence_size = marker[0], len(marker)
current = [line]
continue
current.append(line)
if fence:
marker = fence.group(1)
if marker[0] == fence_char and len(marker) >= fence_size:
blocks.append("".join(current))
current = None
fence_char, fence_size = "", 0
return Counter(blocks)
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
if not new_text.strip() or new_text == unit.text:
raise ValueError("local edit must produce non-empty changed text")
if unit.level == "section":
old_headings = _headings(unit.text)
new_headings = _headings(new_text)
depth = unit.heading_depth
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
if old_peers != new_peers:
raise ValueError("section edit must preserve its peer heading")
if fenced_blocks(unit.text) != fenced_blocks(new_text):
raise ValueError("section edit must preserve fenced code contents")
elif section_depth is not None:
old_peers = [
(item[2], item[3]) for item in _headings(unit.text)
if item[2] <= section_depth
]
new_peers = [
(item[2], item[3]) for item in _headings(new_text)
if item[2] <= section_depth
]
if old_peers != new_peers:
raise ValueError("paragraph edit must not add or change a section heading")
+134
View File
@@ -0,0 +1,134 @@
"""领域模型。"""
from __future__ import annotations
from dataclasses import asdict, dataclass
from typing import Any
DIMENSIONS = (
"Clarity",
"Structure",
"Executability",
"Completeness",
"Constraint Salience",
)
@dataclass
class SkillUnit:
unit_id: str
level: str
parent_id: str | None
heading: str
text: str
order: int
start: int
end: int
heading_depth: int | None = None
@dataclass
class CellScore:
score: float
evidence: list[str]
reason: str
@dataclass
class Coordinate:
unit_id: str
dimension: str
normalized_gap: float
@dataclass
class LocalEdit:
unit_id: str
dimension: str
new_text: str
edit_summary: str
reason: str
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "LocalEdit":
required = {"unit_id", "dimension", "new_text", "edit_summary", "reason"}
missing = sorted(required - value.keys())
if missing:
raise ValueError(f"local edit missing fields: {', '.join(missing)}")
if not all(isinstance(value[key], str) for key in required):
raise ValueError("local edit fields must be strings")
return cls(**{key: value[key] for key in cls.__dataclass_fields__})
@dataclass
class ScoreMatrix:
level: str
units: list[SkillUnit]
columns: dict[str, dict[str, CellScore]]
def to_dict(self) -> dict[str, Any]:
return {
"level": self.level,
"units": [asdict(unit) for unit in self.units],
"columns": {
dimension: {unit_id: asdict(cell) for unit_id, cell in column.items()}
for dimension, column in self.columns.items()
},
}
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "ScoreMatrix":
return cls(
level=str(value["level"]),
units=[SkillUnit(**item) for item in value["units"]],
columns={
dimension: {
unit_id: CellScore(float(cell["score"]), list(cell["evidence"]), str(cell["reason"]))
for unit_id, cell in column.items()
}
for dimension, column in value["columns"].items()
},
)
def unit(self, unit_id: str) -> SkillUnit:
return next(unit for unit in self.units if unit.unit_id == unit_id)
def normalized_gaps(self) -> dict[str, float]:
gaps: dict[str, float] = {}
for dimension in DIMENSIONS:
values = [self.columns[dimension][unit.unit_id].score for unit in self.units]
gaps[dimension] = (max(values) - min(values)) / 4.0 if values else 0.0
return gaps
def select_coordinate(
self,
threshold: float,
dimension: str | None = None,
excluded: set[tuple[str, str]] | None = None,
) -> Coordinate | None:
gaps = self.normalized_gaps()
excluded = excluded or set()
def weak_units(item: str) -> list[SkillUnit]:
column = self.columns[item]
maximum = max((cell.score for cell in column.values()), default=0.0)
return [
unit for unit in self.units
if (unit.unit_id, item) not in excluded
and (maximum - column[unit.unit_id].score) / 4.0 > threshold
]
available = [
item for item in ([dimension] if dimension else DIMENSIONS)
if item is not None and gaps[item] > threshold and weak_units(item)
]
if not available:
return None
dimension = max(available, key=lambda item: gaps[item])
target = min(
weak_units(dimension),
key=lambda unit: (self.columns[dimension][unit.unit_id].score, unit.order),
)
return Coordinate(
target.unit_id,
dimension,
gaps[dimension],
)
+826
View File
@@ -0,0 +1,826 @@
"""Deep 编译流水线编排。"""
from __future__ import annotations
import shutil
import sys
import tempfile
import uuid
from dataclasses import asdict, replace
from pathlib import Path
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.storage import (
atomic_write_json,
atomic_write_jsonl,
atomic_write_text,
load_json,
package_hash,
read_jsonl,
sha256_file,
sha256_text,
)
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
from .adapters.benchflow import (
BenchFlowInput,
SkillsBenchDevelopmentRolloutRunner,
load_benchflow_input,
)
from .adapters.semantic import DeepAnalyzer, valid_evidence
from .core.models import (
CellScore,
Coordinate,
DIMENSIONS,
LocalEdit,
ScoreMatrix,
SkillUnit,
)
from .core.markdown import (
parse_paragraphs,
parse_sections,
preserve_unit_boundary,
replace_unit_text,
validate_edit,
)
ROLLOUTS = 3
MAX_SECTION_ITERATIONS = 6
MAX_PARAGRAPH_ITERATIONS = 3
MAX_EDIT_GENERATION_ATTEMPTS = 3
GAP_THRESHOLD = 0.375
REJECTION_LIMIT = 2
MAX_PARALLEL = 3
DEFAULT_MODEL = "ali/deepseek-v4-pro-0813"
def _log(message: str) -> None:
print(f"[deep] {message}", file=sys.stderr, flush=True)
def _trace_from_dict(value: dict[str, Any]) -> RolloutTrace:
return RolloutTrace(**value)
def _column_to_dict(column: dict[str, CellScore]) -> dict[str, Any]:
return {unit_id: asdict(cell) for unit_id, cell in column.items()}
def _column_from_dict(
value: dict[str, Any], traces: list[RolloutTrace]
) -> dict[str, CellScore]:
return {
unit_id: CellScore(
float(cell["score"]), valid_evidence(cell.get("evidence"), traces), str(cell["reason"])
)
for unit_id, cell in value.items()
}
class DeepLoop:
def __init__(
self,
run_dir: Path,
analyzer: DeepAnalyzer | None = None,
runner: Any | None = None,
):
self.run_dir = run_dir.resolve()
self.state_path = self.run_dir / "run.json"
state = load_json(self.state_path)
if not isinstance(state, dict):
raise ValueError(f"invalid or missing run state: {self.state_path}")
self.state = state
self.temp = self.run_dir / ".tmp"
self.current = self.temp / "current"
self.model = state.get("model", state.get("semantic_model", DEFAULT_MODEL))
self.analyzer = analyzer or DeepAnalyzer(
SemanticClient(self.model), MAX_PARALLEL,
)
context = BenchFlowInput(
state["task"]["name"],
Path(state["task"]["directory"]),
state["task"]["agent"],
state["task"]["model"],
state["task"]["prompt"],
[],
)
self.runner = runner or SkillsBenchDevelopmentRolloutRunner(
context,
Path(tempfile.gettempdir()) / "skill-compiler-deep" / state["run_id"],
MAX_PARALLEL,
archive_root=self.run_dir / "rollouts",
)
@classmethod
def create(
cls,
skill: Path,
traces: Path,
output: Path | None = None,
*,
model: str = DEFAULT_MODEL,
analyzer: DeepAnalyzer | None = None,
runner: Any | None = None,
) -> "DeepLoop":
skill = skill.resolve()
if not (skill / "SKILL.md").is_file():
raise ValueError("--skill must be a skill package containing SKILL.md")
context = load_benchflow_input(traces)
target = (output or skill.parent / f"{skill.name}-deep").resolve()
if target.exists() and any(target.iterdir()):
raise ValueError(f"deep run directory is not empty: {target}")
target.mkdir(parents=True, exist_ok=True)
shutil.copytree(skill, target / "S_fast")
(target / ".tmp").mkdir()
(target / "levels").mkdir()
shutil.copytree(target / "S_fast", target / ".tmp" / "current")
atomic_write_jsonl(target / "input-traces.jsonl", [asdict(trace) for trace in context.traces])
state = {
"run_id": uuid.uuid4().hex[:12],
"status": "created",
"model": model,
"skill_name": skill.name,
"task": {
"name": context.task_name,
"directory": str(context.task_dir),
"agent": context.agent,
"model": context.model,
"prompt": context.prompt,
},
"current_rollouts": str(traces.resolve()),
"current_rollout_skill_sha256": sha256_file(skill / "SKILL.md"),
"levels": {},
}
atomic_write_json(target / "run.json", state)
return cls(target, analyzer=analyzer, runner=runner)
def _save(self) -> None:
atomic_write_json(self.state_path, self.state)
def _prompt(self) -> str:
return str(self.state["task"]["prompt"])
def _current_text(self) -> str:
return (self.current / "SKILL.md").read_text(encoding="utf-8")
def _load_or_rollout(
self,
package: Path,
trace_path: Path,
batch_id: str,
seed: list[RolloutTrace] | None = None,
) -> list[RolloutTrace]:
if trace_path.is_file():
return [_trace_from_dict(item) for item in read_jsonl(trace_path)]
if seed is not None:
traces = seed
else:
_log(f"Starting {ROLLOUTS} rollouts for {batch_id}")
traces = self.runner.run_batch(
package,
self._prompt(),
batch_id,
str(self.state["task"]["name"]),
ROLLOUTS,
progress=lambda done, total, trace_id: _log(
f"Rollout {done}/{total} complete: {trace_id}"
),
)
atomic_write_jsonl(trace_path, [asdict(trace) for trace in traces])
return traces
def _load_or_matrix(
self,
path: Path,
level: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
) -> ScoreMatrix:
value = load_json(path)
if isinstance(value, dict):
return ScoreMatrix.from_dict(value)
columns_dir = path.parent / "matrix-columns"
cached_columns: dict[str, dict[str, CellScore]] = {}
for dimension in DIMENSIONS:
cached = load_json(columns_dir / f"{dimension.lower().replace(' ', '-')}.json")
if isinstance(cached, dict):
cached_columns[dimension] = _column_from_dict(cached, traces)
def save_column(dimension: str, column: dict[str, CellScore]) -> None:
atomic_write_json(
columns_dir / f"{dimension.lower().replace(' ', '-')}.json",
_column_to_dict(column),
)
_log(f"Scoring full {level} matrix with {self.model}")
matrix = self.analyzer.score_matrix(
self._current_text(), self._prompt(), units, traces, level,
existing_columns=cached_columns,
result_callback=save_column,
)
atomic_write_json(path, matrix.to_dict())
return matrix
def _refresh_matrix(
self,
level: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
) -> ScoreMatrix:
level_state = self.state["levels"][level]
number = int(level_state.get("refreshes", 0)) + 1
path = (
self.run_dir
/ "levels"
/ level
/ "refreshes"
/ f"refresh-{number:02d}"
/ "matrix.json"
)
_log(f"Refreshing full {level} matrix")
matrix = self._load_or_matrix(path, level, units, traces)
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
level_state["refreshes"] = number
level_state.pop("active_dimension", None)
self._save()
return matrix
def _decisions(self, level: str | None = None) -> list[dict[str, Any]]:
roots = (
[self.run_dir / "levels" / level]
if level else list((self.run_dir / "levels").glob("*"))
)
decisions = []
for root in roots:
for path in sorted((root / "iterations").glob("iteration-*/decision.json")):
value = load_json(path)
if isinstance(value, dict):
decisions.append(value)
return decisions
def _rejected(self, coordinate: Coordinate) -> list[dict[str, Any]]:
return [
decision["rejected_edit"]
for decision in self._decisions()
if not decision["accepted"]
and decision["rejected_edit"]["unit_id"] == coordinate.unit_id
and decision["rejected_edit"]["dimension"] == coordinate.dimension
]
def _exhausted(self, level: str) -> set[tuple[str, str]]:
counts: dict[tuple[str, str], int] = {}
for decision in self._decisions(level):
if decision["accepted"]:
continue
rejected = decision["rejected_edit"]
key = (rejected["unit_id"], rejected["dimension"])
counts[key] = counts.get(key, 0) + 1
return {key for key, count in counts.items() if count >= REJECTION_LIMIT}
@staticmethod
def _block_dimension(level_state: dict[str, Any], dimension: str) -> None:
blocked = set(level_state.get("blocked_dimensions", []))
blocked.add(dimension)
level_state["blocked_dimensions"] = sorted(blocked)
@staticmethod
def _candidate_units(
units: list[SkillUnit], target: SkillUnit, new_text: str
) -> list[SkillUnit]:
delta = len(new_text) - len(target.text)
updated = []
for unit in units:
value = replace(unit)
if unit.unit_id == target.unit_id:
value.text = new_text
value.end = value.start + len(new_text)
elif unit.start >= target.end:
value.start += delta
value.end += delta
updated.append(value)
return updated
def _apply_edit(
self,
unit: SkillUnit,
edit: LocalEdit,
candidate: Path,
section_depth: int | None,
) -> None:
self._validate_edit_candidate(unit, edit, section_depth)
text = self._current_text()
changed = replace_unit_text(text, unit, edit.new_text)
temporary = candidate.with_name(f".{candidate.name}.{uuid.uuid4().hex}.tmp")
shutil.copytree(self.current, temporary)
atomic_write_text(temporary / "SKILL.md", changed)
if candidate.exists():
shutil.rmtree(candidate)
candidate.parent.mkdir(parents=True, exist_ok=True)
temporary.rename(candidate)
def _validate_edit_candidate(
self,
unit: SkillUnit,
edit: LocalEdit,
section_depth: int | None,
) -> None:
"""Validate a local edit against both unit and whole-document invariants."""
validate_edit(unit, edit.new_text, section_depth)
text = self._current_text()
if text[unit.start:unit.end] != unit.text:
raise ValueError("target unit no longer matches current SKILL.md")
changed = replace_unit_text(text, unit, edit.new_text)
before = [(item.heading, item.heading_depth) for item in parse_sections(text)]
after = [(item.heading, item.heading_depth) for item in parse_sections(changed)]
if before != after:
raise ValueError("local edit changed section boundaries")
@staticmethod
def _validation_feedback(
attempts: list[dict[str, Any]],
) -> list[dict[str, Any]]:
return [
{
"edit_summary": str(item.get("edit_summary", "invalid generated edit")),
"reject_reason": f"structural_validation_failed: {item['error']}",
"new_text_hash": str(item.get("new_text_hash", "")),
}
for item in attempts
]
def _valid_edit_or_rejection(
self,
*,
level: str,
number: int,
iteration_dir: Path,
unit: SkillUnit,
coordinate: Coordinate,
cell: CellScore,
section_depth: int | None,
) -> tuple[LocalEdit | None, bool, list[dict[str, Any]]]:
"""Load or generate a valid edit, feeding structural failures back to the model."""
edit_path = iteration_dir / "edit.json"
attempts_path = iteration_dir / "edit-attempts.json"
attempts_value = load_json(attempts_path, [])
attempts = attempts_value if isinstance(attempts_value, list) else []
cached_value = load_json(edit_path)
if isinstance(cached_value, dict):
try:
cached = LocalEdit.from_dict(cached_value)
cached.new_text = preserve_unit_boundary(unit, cached.new_text)
self._validate_edit_candidate(unit, cached, section_depth)
return cached, False, attempts
except ValueError as exc:
text = str(cached_value.get("new_text", ""))
attempts.append({
"source": "cached",
"error": str(exc),
"edit_summary": str(cached_value.get("edit_summary", "")),
"new_text_hash": sha256_text(text) if text else "",
})
atomic_write_json(attempts_path, attempts)
_log(
f"{level} iteration {number}: cached local edit is invalid: {exc}; "
"regenerating"
)
for attempt in range(1, MAX_EDIT_GENERATION_ATTEMPTS + 1):
_log(
f"{level} iteration {number}: generating local edit for "
f"{coordinate.unit_id}/{coordinate.dimension} "
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS})"
)
feedback = self._rejected(coordinate) + self._validation_feedback(attempts)
edit: LocalEdit | None = None
try:
edit = self.analyzer.generate_edit(
coordinate,
unit,
cell,
feedback,
self._prompt(),
)
edit.new_text = preserve_unit_boundary(unit, edit.new_text)
self._validate_edit_candidate(unit, edit, section_depth)
except ValueError as exc:
text = edit.new_text if edit is not None else ""
attempts.append({
"source": "generated",
"generation_attempt": attempt,
"error": str(exc),
"edit_summary": edit.edit_summary if edit is not None else "",
"new_text_hash": sha256_text(text) if text else "",
})
atomic_write_json(attempts_path, attempts)
_log(
f"{level} iteration {number}: local edit validation failed "
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS}): {exc}"
)
continue
assert edit is not None
atomic_write_json(edit_path, asdict(edit))
return edit, True, attempts
return None, True, attempts
def _commit_iteration(
self,
level: str,
number: int,
iteration_dir: Path,
matrix: ScoreMatrix,
current_trace_path: Path,
) -> ScoreMatrix:
decision = load_json(iteration_dir / "decision.json")
if not isinstance(decision, dict):
raise ValueError("missing iteration decision")
if decision["accepted"]:
candidate = iteration_dir / "candidate" / self.state["skill_name"]
replacement = self.temp / "next-current"
shutil.rmtree(replacement, ignore_errors=True)
shutil.copytree(candidate, replacement)
shutil.rmtree(self.current)
replacement.rename(self.current)
matrix = ScoreMatrix.from_dict(decision["matrix_after"])
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
candidate_traces = read_jsonl(iteration_dir / "candidate-traces.jsonl")
atomic_write_jsonl(current_trace_path, candidate_traces)
rollout_output = decision.get("rollout_output")
rollout_skill_sha256 = decision.get("rollout_skill_sha256")
if isinstance(rollout_output, str) and isinstance(rollout_skill_sha256, str):
self.state["current_rollouts"] = rollout_output
self.state["current_rollout_skill_sha256"] = rollout_skill_sha256
else:
self.state.pop("current_rollouts", None)
self.state.pop("current_rollout_skill_sha256", None)
level_state = self.state["levels"][level]
if int(level_state.get("iterations", 0)) < number:
level_state["iterations"] = number
self._save()
return matrix
def _run_iteration(
self,
level: str,
number: int,
level_dir: Path,
matrix: ScoreMatrix,
coordinate: Coordinate,
current_trace_path: Path,
section_depth: int | None,
) -> ScoreMatrix:
iteration_dir = level_dir / "iterations" / f"iteration-{number:02d}"
iteration_dir.mkdir(parents=True, exist_ok=True)
unit = matrix.unit(coordinate.unit_id)
edit, regenerated, validation_attempts = self._valid_edit_or_rejection(
level=level,
number=number,
iteration_dir=iteration_dir,
unit=unit,
coordinate=coordinate,
cell=matrix.columns[coordinate.dimension][coordinate.unit_id],
section_depth=section_depth,
)
if edit is None:
decision = {
"coordinate": asdict(coordinate),
"accepted": False,
"reason": "edit_validation_exhausted",
"target_delta": 0.0,
"validation_attempts": validation_attempts,
"rejected_edit": {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": (
"Could not generate a structurally valid local edit after "
f"{MAX_EDIT_GENERATION_ATTEMPTS} attempts."
),
"score_change": 0.0,
"reject_reason": "edit_validation_exhausted",
"new_text_hash": str(
validation_attempts[-1].get("new_text_hash", "")
) if validation_attempts else "",
},
}
atomic_write_json(iteration_dir / "decision.json", decision)
_log(
f"{level} iteration {number}: local edit validation exhausted; "
"recording rejection and continuing"
)
return self._commit_iteration(
level, number, iteration_dir, matrix, current_trace_path
)
edit_hash = sha256_text(edit.new_text)
if any(item["new_text_hash"] == edit_hash for item in self._rejected(coordinate)):
decision = {
"coordinate": asdict(coordinate),
"accepted": False,
"reason": "exact_duplicate_rejected_edit",
"target_delta": 0.0,
"rejected_edit": {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": edit.edit_summary,
"score_change": 0.0,
"reject_reason": "exact_duplicate_rejected_edit",
"new_text_hash": edit_hash,
},
}
atomic_write_json(iteration_dir / "decision.json", decision)
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
candidate = iteration_dir / "candidate" / self.state["skill_name"]
if regenerated:
shutil.rmtree(candidate, ignore_errors=True)
for stale in (
iteration_dir / "candidate-traces.jsonl",
iteration_dir / "comparison.json",
):
if stale.exists():
stale.unlink()
if not (candidate / "SKILL.md").is_file():
self._apply_edit(unit, edit, candidate, section_depth)
candidate_units = self._candidate_units(matrix.units, unit, edit.new_text)
candidate_skill_sha256 = sha256_file(candidate / "SKILL.md")
batch_id = (
f"{self.state['run_id']}-deep-{level}-i{number:02d}-"
f"{candidate_skill_sha256[:12]}"
)
traces = self._load_or_rollout(
candidate,
iteration_dir / "candidate-traces.jsonl",
batch_id,
)
comparison_path = iteration_dir / "comparison.json"
comparison = load_json(comparison_path)
if not isinstance(comparison, dict):
_log(
f"{level} iteration {number}: comparing "
f"{coordinate.unit_id}/{coordinate.dimension}"
)
incumbent_traces = [
_trace_from_dict(item) for item in read_jsonl(current_trace_path)
]
comparison = self.analyzer.compare_cell(
self._prompt(),
unit,
next(item for item in candidate_units if item.unit_id == unit.unit_id),
incumbent_traces,
traces,
coordinate.dimension,
)
incumbent_timeout_rate = (
sum(trace.timed_out is True for trace in incumbent_traces)
/ len(incumbent_traces)
)
candidate_timeout_rate = (
sum(trace.timed_out is True for trace in traces) / len(traces)
)
if (
candidate_timeout_rate > incumbent_timeout_rate
and "timeout" not in comparison["runtime_regressions"]
):
comparison["runtime_regressions"].append("timeout")
atomic_write_json(comparison_path, comparison)
delta = float(comparison["candidate_score"]) - float(
comparison["incumbent_score"]
)
if not comparison["task_relevant"]:
accepted, reason = False, "target_unit_not_task_relevant"
elif comparison["runtime_regressions"]:
accepted = False
reason = "runtime_regressed:" + ",".join(comparison["runtime_regressions"])
elif delta < 0.5:
accepted, reason = False, "target_cell_did_not_improve"
else:
accepted, reason = True, "target_improved_without_runtime_regression"
decision: dict[str, Any] = {
"coordinate": asdict(coordinate),
"accepted": accepted,
"reason": reason,
"target_delta": delta,
"rollout_skill_sha256": candidate_skill_sha256,
}
artifacts_dir = getattr(self.runner, "artifacts_dir", None)
rollout_output = artifacts_dir(batch_id) if callable(artifacts_dir) else None
if isinstance(rollout_output, Path):
decision["rollout_output"] = str(rollout_output.resolve())
if accepted:
updated = ScoreMatrix(matrix.level, candidate_units, dict(matrix.columns))
updated.columns[coordinate.dimension] = dict(
matrix.columns[coordinate.dimension]
)
updated.columns[coordinate.dimension][coordinate.unit_id] = CellScore(
float(comparison["candidate_score"]),
list(comparison["candidate_evidence"]),
str(comparison["reason"]),
)
decision["matrix_after"] = updated.to_dict()
else:
decision["rejected_edit"] = {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": edit.edit_summary,
"score_change": delta,
"reject_reason": reason,
"new_text_hash": edit_hash,
}
atomic_write_json(iteration_dir / "decision.json", decision)
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
def _run_level(
self,
level: str,
units: list[SkillUnit],
seed_traces: list[RolloutTrace] | None = None,
section_depth: int | None = None,
) -> tuple[ScoreMatrix, list[RolloutTrace], Coordinate | None]:
level_dir = self.run_dir / "levels" / level
level_dir.mkdir(parents=True, exist_ok=True)
level_state = self.state["levels"].setdefault(level, {
"iterations": 0, "completed": False,
})
current_trace_path = level_dir / "current-traces.jsonl"
traces = self._load_or_rollout(
self.current,
current_trace_path,
f"{self.state['run_id']}-deep-{level}-initial",
seed=seed_traces,
)
matrix = self._load_or_matrix(level_dir / "matrix.json", level, units, traces)
if level_state.get("completed"):
exhausted_value = level_state.get("exhausted_coordinate")
exhausted = Coordinate(**exhausted_value) if isinstance(exhausted_value, dict) else None
return matrix, traces, exhausted
exhausted_coordinate: Coordinate | None = None
max_iterations = (
MAX_SECTION_ITERATIONS if level == "section" else MAX_PARAGRAPH_ITERATIONS
)
while int(level_state["iterations"]) < max_iterations:
excluded = self._exhausted(level)
excluded.update(
(unit.unit_id, dimension)
for dimension in level_state.get("blocked_dimensions", [])
for unit in matrix.units
)
pending_number = int(level_state["iterations"]) + 1
pending_dir = level_dir / "iterations" / f"iteration-{pending_number:02d}"
pending_decision = load_json(pending_dir / "decision.json")
if isinstance(pending_decision, dict):
pending_coordinate = Coordinate(**pending_decision["coordinate"])
level_state.setdefault("active_dimension", pending_coordinate.dimension)
matrix = self._commit_iteration(
level, pending_number, pending_dir, matrix, current_trace_path
)
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
if len(self._rejected(pending_coordinate)) >= REJECTION_LIMIT:
exhausted_coordinate = pending_coordinate
level_state["stop_reason"] = "coordinate_exhausted"
level_state["exhausted_coordinate"] = asdict(pending_coordinate)
break
if matrix.select_coordinate(
GAP_THRESHOLD, level_state.get("active_dimension"), excluded
) is None:
matrix = self._refresh_matrix(level, matrix.units, traces)
continue
active_dimension = level_state.get("active_dimension")
coordinate = matrix.select_coordinate(
GAP_THRESHOLD, active_dimension, excluded
)
if coordinate is None:
if active_dimension is None:
level_state["stop_reason"] = (
"normalized_gap_converged"
if max(matrix.normalized_gaps().values(), default=0.0) <= GAP_THRESHOLD
else "available_coordinates_exhausted"
)
break
matrix = self._refresh_matrix(level, matrix.units, traces)
continue
if active_dimension is None:
level_state["active_dimension"] = coordinate.dimension
self._save()
number = int(level_state["iterations"]) + 1
matrix = self._run_iteration(
level, number, level_dir, matrix, coordinate,
current_trace_path, section_depth,
)
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
if len(self._rejected(coordinate)) >= REJECTION_LIMIT:
exhausted_coordinate = coordinate
level_state["stop_reason"] = "coordinate_exhausted"
level_state["exhausted_coordinate"] = asdict(coordinate)
break
else:
decisions = self._decisions(level)
last = decisions[-1] if decisions else {}
if matrix.select_coordinate(GAP_THRESHOLD) is None:
level_state["stop_reason"] = "normalized_gap_converged"
else:
level_state["stop_reason"] = (
"max_iterations_after_accept" if last.get("accepted") else "max_iterations"
)
level_state["completed"] = True
self._save()
return matrix, traces, exhausted_coordinate
def drive(self) -> Path:
if self.state.get("status") == "complete" and (self.run_dir / "S_final").is_dir():
return self.run_dir / "S_final"
_log(f"Deep Loop start/resume: {self.run_dir}")
input_traces = [
_trace_from_dict(item) for item in read_jsonl(self.run_dir / "input-traces.jsonl")
]
while True:
section_units = parse_sections(self._current_text())
section_matrix, traces, exhausted = self._run_level(
"section", section_units, seed_traces=input_traces
)
if exhausted is None:
break
target_score = section_matrix.columns[exhausted.dimension][exhausted.unit_id].score
current_sections = parse_sections(self._current_text())
section = next((item for item in current_sections if item.unit_id == exhausted.unit_id), None)
paragraphs = parse_paragraphs(section) if section is not None else []
section_state = self.state["levels"]["section"]
if (
target_score > 3.5
or len(paragraphs) < 2
or section_state.get("paragraph_returned")
):
self._block_dimension(section_state, exhausted.dimension)
section_state["completed"] = False
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
section_state.pop(key, None)
self._save()
continue
_log(f"Descending into paragraphs of {exhausted.unit_id}")
self.state["levels"].setdefault(
"paragraph", {"iterations": 0, "completed": False}
)["active_dimension"] = exhausted.dimension
self._save()
_, paragraph_traces, _ = self._run_level(
"paragraph", paragraphs, seed_traces=traces,
section_depth=section.heading_depth if section else None,
)
atomic_write_jsonl(
self.run_dir / "levels" / "section" / "current-traces.jsonl",
[asdict(trace) for trace in paragraph_traces],
)
section_state["completed"] = False
section_state["paragraph_returned"] = True
self._block_dimension(section_state, exhausted.dimension)
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
section_state.pop(key, None)
self._refresh_matrix(
"section", parse_sections(self._current_text()), paragraph_traces
)
return self._complete()
def _complete(self) -> Path:
target = self.run_dir / "S_final"
if target.exists():
shutil.rmtree(target)
shutil.copytree(self.current, target)
final_skill_hash = sha256_file(target / "SKILL.md")
final_rollouts = self.state.get("current_rollouts")
final_rollout_hash = self.state.get("current_rollout_skill_sha256")
if (
not isinstance(final_rollouts, str)
or not Path(final_rollouts).is_dir()
or final_rollout_hash != final_skill_hash
):
final_rollouts = None
report = {
"status": "complete",
"input_package_hash": package_hash(self.run_dir / "S_fast"),
"final_package_hash": package_hash(target),
"final_rollouts": final_rollouts,
"final_rollout_skill_sha256": final_skill_hash if final_rollouts else None,
"task": self.state["task"]["name"],
"agent": self.state["task"]["agent"],
"target_model": self.state["task"]["model"],
"model": self.model,
"rollout_backend": "skillsbench_development",
"production_rollout_backend": "blank_container_required",
"verifier_signal_used": False,
"levels": self.state["levels"],
"decisions": {
name: self._decisions(name)
for name in ("section", "paragraph")
if name in self.state["levels"]
},
}
atomic_write_json(self.run_dir / "report.json", report)
self.state["status"] = "complete"
self._save()
shutil.rmtree(self.temp, ignore_errors=True)
_log(f"Deep Loop complete: {target}")
return target
+3
View File
@@ -0,0 +1,3 @@
from .cli import main
raise SystemExit(main())
+300
View File
@@ -0,0 +1,300 @@
from __future__ import annotations
import json
import math
from dataclasses import asdict
from pathlib import Path
from typing import Any, Iterable
from .models import Patch, RolloutTrace
from .scoring.pre_score import PRE_SCORE_VERSION, PreScoreResult
from .storage import (
atomic_write_json,
load_json,
package_hash,
read_jsonl,
sha256_file,
sha256_text,
)
CACHE_SCHEMA_VERSION = 1
CACHE_PRODUCER = "dynamic-compile-fast"
SCORE_ARTIFACTS = (
"pre_scores.jsonl",
"agentrm_scores.jsonl",
"effective_scores.jsonl",
)
def fingerprint(value: Any) -> str:
payload = json.dumps(
value, ensure_ascii=False, sort_keys=True, separators=(",", ":")
)
return sha256_text(payload)
def trace_fingerprint(traces: Iterable[RolloutTrace]) -> str:
return fingerprint([asdict(trace) for trace in traces])
def _artifact_hash(path: Path) -> str:
if path.is_file():
return sha256_file(path)
if path.is_dir():
return package_hash(path)
raise OSError(f"cache artifact does not exist: {path}")
def write_manifest(
root: Path,
name: str,
*,
stage: str,
input_hash: str,
config: dict[str, Any],
artifacts: Iterable[str],
) -> None:
artifact_names = tuple(artifacts)
hashes = {item: _artifact_hash(root / item) for item in artifact_names}
atomic_write_json(
root / name,
{
"schema_version": CACHE_SCHEMA_VERSION,
"producer": CACHE_PRODUCER,
"stage": stage,
"status": "complete",
"input_hash": input_hash,
"config_hash": fingerprint(config),
"artifacts": hashes,
},
)
def valid_manifest(
root: Path,
name: str,
*,
stage: str,
input_hash: str,
config: dict[str, Any],
artifacts: Iterable[str],
) -> bool:
try:
value = load_json(root / name)
expected = tuple(artifacts)
if not isinstance(value, dict):
return False
if value.get("schema_version") != CACHE_SCHEMA_VERSION:
return False
if value.get("producer") != CACHE_PRODUCER:
return False
if value.get("stage") != stage or value.get("status") != "complete":
return False
if value.get("input_hash") != input_hash:
return False
if value.get("config_hash") != fingerprint(config):
return False
hashes = value.get("artifacts")
if not isinstance(hashes, dict) or set(hashes) != set(expected):
return False
return all(
isinstance(hashes[item], str)
and hashes[item] == _artifact_hash(root / item)
for item in expected
)
except (OSError, TypeError, ValueError):
return False
def load_score_cache(
traces: list[RolloutTrace],
score_dir: Path,
*,
input_hash: str,
config: dict[str, Any],
) -> tuple[dict[str, float], list[dict[str, Any]]] | None:
if not valid_manifest(
score_dir,
".score-cache.json",
stage="score",
input_hash=input_hash,
config=config,
artifacts=SCORE_ARTIFACTS,
):
return None
try:
pre_rows = read_jsonl(score_dir / "pre_scores.jsonl")
agentrm_rows = read_jsonl(score_dir / "agentrm_scores.jsonl")
score_rows = read_jsonl(score_dir / "effective_scores.jsonl")
trace_by_id = {trace.trace_id: trace for trace in traces}
if len(trace_by_id) != len(traces):
return None
pre_by_id: dict[str, PreScoreResult] = {}
for row in pre_rows:
result = PreScoreResult.from_dict(row)
if result.trace_id in pre_by_id:
return None
if result.scoring_version != PRE_SCORE_VERSION:
return None
if result.route not in {"agentrm", "fixed_score"}:
return None
if result.route == "fixed_score":
if result.score is None or not math.isfinite(result.score):
return None
pre_by_id[result.trace_id] = result
if set(pre_by_id) != set(trace_by_id):
return None
expected_agentrm = {
trace.key.as_tuple()
for trace in traces
if pre_by_id[trace.trace_id].route == "agentrm"
}
actual_agentrm: set[tuple[str, str, str]] = set()
for row in agentrm_rows:
key = (
str(row["task_name"]),
str(row["compile_type"]),
str(row["test_name"]),
)
score = float(row["score"])
if key in actual_agentrm or not math.isfinite(score):
return None
if int(row["n_tokens"]) < 0:
return None
actual_agentrm.add(key)
if actual_agentrm != expected_agentrm:
return None
by_id: dict[str, dict[str, Any]] = {}
scores: dict[str, float] = {}
for row in score_rows:
trace_id = str(row["trace_id"])
score = float(row["effective_score"])
if trace_id in by_id or not math.isfinite(score):
return None
trace = trace_by_id.get(trace_id)
if trace is None or (
str(row["task_name"]),
str(row["compile_type"]),
str(row["test_name"]),
) != trace.key.as_tuple():
return None
expected_source = (
"agentrm"
if pre_by_id[trace_id].route == "agentrm"
else "pre_score"
)
if row.get("score_source") != expected_source:
return None
by_id[trace_id] = row
scores[trace_id] = score
if set(by_id) != set(trace_by_id):
return None
return scores, [by_id[trace.trace_id] for trace in traces]
except (KeyError, OSError, TypeError, ValueError):
return None
def load_maps_cache(
output_dir: Path,
selected: list[str],
*,
input_hash: str,
config: dict[str, Any],
) -> list[dict[str, Any]] | None:
if not valid_manifest(
output_dir,
".maps-cache.json",
stage="maps",
input_hash=input_hash,
config=config,
artifacts=("maps.json",),
):
return None
try:
value = load_json(output_dir / "maps.json")
if not isinstance(value, list) or not all(isinstance(item, dict) for item in value):
return None
by_id = {str(item.get("trace_id", "")): item for item in value}
if len(by_id) != len(value) or set(by_id) != set(selected):
return None
if any(
not isinstance(item.get("patterns"), list)
or not all(isinstance(pattern, dict) for pattern in item["patterns"])
for item in value
):
return None
return [by_id[trace_id] for trace_id in selected]
except (OSError, TypeError, ValueError):
return None
def load_reduction_cache(
output_dir: Path,
*,
input_hash: str,
config: dict[str, Any],
) -> dict[str, Any] | None:
if not valid_manifest(
output_dir,
".reduction-cache.json",
stage="reduction",
input_hash=input_hash,
config=config,
artifacts=("reduction.json",),
):
return None
try:
value = load_json(output_dir / "reduction.json")
if not isinstance(value, dict):
return None
for key in ("successful_pattern", "failure_pattern", "selected_gap"):
item = value.get(key)
if not isinstance(item, dict) or not str(item.get("description", "")).strip():
return None
return value
except (OSError, TypeError, ValueError):
return None
def load_candidate_cache(
output_dir: Path,
candidate_name: str,
skill_hash: str,
*,
input_hash: str,
config: dict[str, Any],
) -> Path | None:
candidate_relative = f"candidate-skill/{candidate_name}"
if not valid_manifest(
output_dir,
".candidate-cache.json",
stage="candidate",
input_hash=input_hash,
config=config,
artifacts=("patch.json", candidate_relative),
):
return None
candidate = output_dir / candidate_relative
try:
value = load_json(output_dir / "patch.json")
if not isinstance(value, dict) or value.get("skill_hash") != skill_hash:
return None
patches = value.get("patches")
if not isinstance(patches, list) or not 1 <= len(patches) <= 2:
return None
parsed = [Patch.from_dict(item) for item in patches if isinstance(item, dict)]
if len(parsed) != len(patches) or any(patch.skill_hash != skill_hash for patch in parsed):
return None
if [patch.role for patch in parsed] != ["promote_success", "mitigate_failure"][:len(parsed)]:
return None
candidate_skill = candidate / "SKILL.md"
if not candidate_skill.is_file():
return None
if value.get("candidate_skill_hash") != sha256_file(candidate_skill):
return None
return candidate
except (OSError, TypeError, ValueError):
return None
+97
View File
@@ -0,0 +1,97 @@
"""动态编译唯一命令行入口。"""
from __future__ import annotations
import argparse
import os
import sys
from pathlib import Path
from scripts.provider_router import parse_model_reference
from .pipeline import run_pipeline
from .scoring.agentrm import (
DEFAULT_BATCH_SIZE,
DEFAULT_CONCURRENCY,
DEFAULT_MAX_LENGTH,
DEFAULT_RM_API_URL,
DEFAULT_TIMEOUT,
)
def _provider_model(value: str) -> str:
try:
return parse_model_reference(value).value
except ValueError as exc:
raise argparse.ArgumentTypeError(str(exc)) from exc
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="python -m scripts.dynamic_compile.fast",
description="从一组 BenchFlow 历史轨迹生成一个动态编译候选 Skill。",
)
parser.add_argument("--traces", required=True, type=Path, help="BenchFlow 轨迹目录")
parser.add_argument("--skill", required=True, type=Path, help="含根 SKILL.md 的 Skill 包")
parser.add_argument("--score-output", type=Path, help="评分阶段产物目录")
parser.add_argument("--output", type=Path, help="分析与候选 Skill 产物目录")
parser.add_argument(
"--model",
required=True,
type=_provider_model,
help="所有外部模型调用使用的 provider/model",
)
parser.add_argument("--max-parallel", type=int, default=3)
parser.add_argument(
"--rm-api-url", default=os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
)
parser.add_argument(
"--rm-max-length",
type=int,
default=os.environ.get("RM_MAX_LENGTH", str(DEFAULT_MAX_LENGTH)),
)
parser.add_argument(
"--rm-timeout",
type=float,
default=os.environ.get("RM_TIMEOUT", str(DEFAULT_TIMEOUT)),
)
parser.add_argument(
"--rm-concurrency",
type=int,
default=os.environ.get("RM_CONCURRENCY", str(DEFAULT_CONCURRENCY)),
)
parser.add_argument(
"--rm-batch-size",
type=int,
default=os.environ.get("RM_BATCH_SIZE", str(DEFAULT_BATCH_SIZE)),
)
parser.add_argument(
"--force",
action="store_true",
help="复用有效评分/Map 缓存,强制重建 Reduce、Patch 和候选 Skill",
)
return parser
def main(argv: list[str] | None = None) -> int:
args = build_parser().parse_args(argv)
try:
result = run_pipeline(
args.traces,
args.skill,
score_output=args.score_output,
output=args.output,
model=args.model,
max_parallel=args.max_parallel,
rm_api_url=args.rm_api_url,
rm_max_length=args.rm_max_length,
rm_timeout=args.rm_timeout,
rm_concurrency=args.rm_concurrency,
rm_batch_size=args.rm_batch_size,
force=args.force,
)
except (OSError, RuntimeError, ValueError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
print(result)
return 0
+89
View File
@@ -0,0 +1,89 @@
from __future__ import annotations
from dataclasses import asdict, dataclass, field
from typing import Any, Mapping
@dataclass(frozen=True, order=True)
class TraceKey:
task_name: str
compile_type: str
test_name: str
@classmethod
def from_record(cls, value: Mapping[str, Any]) -> "TraceKey":
try:
return cls(
str(value["task_name"]),
str(value["compile_type"]),
str(value["test_name"]),
)
except KeyError as exc:
raise ValueError(f"missing trace identity field: {exc.args[0]}") from exc
def as_tuple(self) -> tuple[str, str, str]:
return self.task_name, self.compile_type, self.test_name
def __str__(self) -> str:
return "/".join(self.as_tuple())
@dataclass
class Patch:
edit_type: str
target_heading: str
old_text: str
new_text: str
evidence_ids: list[str]
evidence_type: str
confidence: str
reason: str
role: str = ""
skill_hash: str = ""
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "Patch":
required = {
"edit_type", "target_heading", "old_text", "new_text", "evidence_ids",
"evidence_type", "confidence", "reason",
}
missing = sorted(required - value.keys())
if missing:
raise ValueError(f"patch missing fields: {', '.join(missing)}")
if not isinstance(value["evidence_ids"], list):
raise ValueError("patch evidence_ids must be a list")
if "role" in value and not isinstance(value["role"], str):
raise ValueError("patch role must be a string")
for key in required - {"evidence_ids"}:
if not isinstance(value[key], str):
raise ValueError(f"patch {key} must be a string")
return cls(**{key: value.get(key, "") for key in cls.__dataclass_fields__})
def to_dict(self) -> dict[str, Any]:
return asdict(self)
@dataclass
class RolloutTrace:
trace_id: str
task_name: str
compile_type: str
test_name: str
state: list[dict[str, Any]]
skill_invoked: bool
exit_code: int | None = None
timed_out: bool | None = None
metadata: dict[str, Any] = field(default_factory=dict)
@property
def key(self) -> TraceKey:
return TraceKey(self.task_name, self.compile_type, self.test_name)
def agentrm_request(self) -> dict[str, Any]:
"""Project a rich runtime trace onto AgentRM's stable input schema."""
return {
"state": self.state,
"task_name": self.task_name,
"compile_type": self.compile_type,
"test_name": self.test_name,
}
@@ -0,0 +1,239 @@
from __future__ import annotations
import json
import sys
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Any
from ..models import Patch, RolloutTrace
from scripts.provider_router import resolve_model_route
from ..storage import package_manifest, sha256_file
from .trace_format import compact_trace
class SemanticClient:
def __init__(
self,
model: str = "opencode/deepseek-v4-pro",
timeout: int = 900,
):
route = resolve_model_route(model)
assert route is not None
base_url = route.url.removesuffix("/chat/completions").rstrip("/")
try:
from openai import OpenAI
except ImportError as exc:
raise RuntimeError("the openai package is required for semantic calls") from exc
self.client = OpenAI(base_url=base_url, api_key=route.api_key, timeout=timeout, max_retries=0)
self.model = route.reference.model_id
self.model_reference = route.reference.value
def json(self, system: str, user: str, attempts: int = 1) -> dict[str, Any]:
error: Exception | None = None
for attempt in range(attempts):
try:
response = self.client.chat.completions.create(
model=self.model,
temperature=0,
response_format={"type": "json_object"},
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
stream=True,
)
content = "".join(
choice.delta.content or ""
for chunk in response
for choice in chunk.choices
)
value = json.loads(content or "{}")
if not isinstance(value, dict):
raise ValueError("semantic response must be a JSON object")
return value
except Exception as exc:
error = exc
if attempt + 1 < attempts:
print(
f"[semantic] request attempt {attempt + 1}/{attempts} failed: "
f"{type(exc).__name__}: {exc}; retrying",
file=sys.stderr,
flush=True,
)
time.sleep(1 + attempt)
raise RuntimeError(f"semantic call failed after {attempts} attempts: {error}")
class SemanticAnalyzer:
def __init__(self, client: SemanticClient, max_parallel: int = 3):
self.client = client
self.max_parallel = max_parallel
def generate_probe(self, skill_package: Path) -> dict[str, str]:
skill = (skill_package / "SKILL.md").read_text(encoding="utf-8")
files = [item["path"] for item in package_manifest(skill_package)]
result = self.client.json(
"You create realistic probe tasks for testing an agent skill. Return JSON only.",
f"""Create one task prompt for the skill below. The task must explicitly tell the agent to invoke this skill, exercise its core workflow, and remain solvable in an empty workspace with network access. Do not copy the skill's operation steps, create a verifier, or create a test environment. Return {{"prompt": string, "rationale": string}}.
Package files: {json.dumps(files, ensure_ascii=False)}
SKILL.md:
{skill}""",
)
prompt = result.get("prompt")
if not isinstance(prompt, str) or not prompt.strip():
raise ValueError("probe generator returned no prompt")
return {"prompt": prompt.strip(), "rationale": str(result.get("rationale", ""))}
def map_trace(self, trace: RolloutTrace, score: float, bucket: str) -> dict[str, Any]:
result = self.client.json(
"Analyze agent behavior from a scored trace. Return concise JSON only.",
f"""Analyze this {bucket} trace (AgentRM score {score}) as an action-level workflow. Identify what the agent did, action ordering, stopping behavior, error recovery, and whether required outputs were persisted promptly. Runtime facts report observable execution only; do not infer external verification outcomes or correctness that is not visible in the trace. Use long payload details only when they are necessary to explain a behavioral effect. Return:
{{"trace_id":"{trace.trace_id}","patterns":[{{"description":string,"condition":string,"effect":string,"recovered":boolean,"final_quality_impact":string,"evidence_ids":[string]}}]}}.
Use E### or E###.T## event IDs as evidence. Skill invocation is evidence, not a quality gate.
{compact_trace(trace)}""",
)
result["runtime_facts"] = {
"termination": trace.metadata.get("termination"),
"timed_out": trace.timed_out,
"exit_code": trace.exit_code,
"duration_seconds": trace.metadata.get("agent_execution_seconds"),
"tool_calls": trace.metadata.get("tool_calls"),
"skill_invoked": trace.skill_invoked,
"timeout_reason": trace.metadata.get("timeout_reason"),
"error_category": trace.metadata.get("error_category"),
"partial_trajectory": trace.metadata.get("partial_trajectory"),
}
result.setdefault("trace_id", trace.trace_id)
result.setdefault("patterns", [])
return result
def map_all(
self,
traces: list[RolloutTrace],
scores: dict[str, float],
high_ids: set[str],
low_ids: set[str] | None = None,
progress: Any | None = None,
result_callback: Any | None = None,
) -> list[dict[str, Any]]:
def work(trace: RolloutTrace) -> dict[str, Any]:
bucket = "High" if trace.trace_id in high_ids else "Low" if low_ids is None or trace.trace_id in low_ids else "Neutral"
return self.map_trace(trace, scores[trace.trace_id], bucket)
results: list[dict[str, Any] | None] = [None] * len(traces)
errors: list[tuple[str, Exception]] = []
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
futures = {pool.submit(work, trace): index for index, trace in enumerate(traces)}
completed = 0
for future in as_completed(futures):
index = futures[future]
try:
result = future.result()
except Exception as exc:
errors.append((traces[index].trace_id, exc))
continue
results[index] = result
if result_callback is not None:
result_callback(result)
completed += 1
if progress is not None:
progress(completed, len(traces), traces[index].trace_id)
if errors:
details = "; ".join(
f"{trace_id}: {type(error).__name__}: {error}"
for trace_id, error in errors
)
raise RuntimeError(f"{len(errors)} Map trace(s) failed; successful results were preserved: {details}")
return [result for result in results if result is not None]
def reduce(
self,
skill_text: str,
maps: list[dict[str, Any]],
score_rows: list[dict[str, Any]],
high_ids: list[str],
low_ids: list[str],
history: list[dict[str, Any]],
) -> dict[str, Any]:
result = self.client.json(
"Contrast exactly one successful pattern with exactly one failure pattern to identify one bounded skill-improvement gap. Return JSON only with every requested field populated; never return an empty object.",
f"""Compare the fixed relative High and Low groups. Select exactly one successful behavioral pattern from the High traces that the skill should preserve or promote, and exactly one contrasting failure pattern from the Low traces that the skill should mitigate. Then select exactly one local, generalizable gap in the current skill that connects those two patterns and has not already been addressed in Patch history. Do not select two unrelated improvements.
The selected gap must be one behavioral clarification or stopping decision, expressed in at most two sentences, not a multi-step policy. It must explicitly preserve the selected successful behavior and mitigate the selected failure behavior, remain general to the skill, and avoid turning the failure into an exhaustive or universal obligation. Do not mention benchmark-specific files or labels, invent exact counts, thresholds, quotas, or mandatory tool sequences, add significant tool work, broaden external search, or delay a required deliverable.
Infer the contrast before proposing the remedy. Compare runtime_facts for completion, duration, tool calls, and timeouts; use Map patterns and their evidence to explain the behavior. Effective score indicates relative outcome, not a root cause. Describe group tendencies only to the extent supported, cite the supporting trace and event IDs, and reflect exceptions in confidence.
If High traces show several successful behaviors, choose the one with the clearest evidence and strongest direct contrast with the selected Low failure. If Low traces show opposing failure modes, choose only the strongest failure that can be addressed by the same qualitative decision boundary as the selected success. For incomplete work and overwork, prefer prioritization, evidence-based stopping, and timely persistence over additional checking.
You must still select one success, one failure, and one gap, each with a non-empty description. Each pattern must cite evidence from its corresponding group. If contrast is weak, use evidence_type=weak_contrast_fallback and confidence=low, and choose the most conservative supported pair and clarification.
Return every field in this exact shape: {{"successful_pattern":{{"description":"one High-group behavior to preserve or promote","evidence_ids":["trace_id:E###"]}},"failure_pattern":{{"description":"one contrasting Low-group behavior to mitigate","evidence_ids":["trace_id:E###"]}},"contrast":"direct relationship between the selected success and failure","root_cause":"skill-level cause","source":"compared evidence","skill_mitigatable":true,"confidence":"low|medium|high","selected_gap":{{"description":"one supported behavioral clarification","success_behavior_to_preserve":"the selected successful behavior","failure_behavior_to_mitigate":"the selected failure behavior","target_heading":"existing skill heading","evidence_type":"contrast type","confidence":"low|medium|high","evidence_ids":["trace_id:E###"],"reason":"why this one gap preserves the success while mitigating the failure"}}}}.
High IDs: {json.dumps(high_ids)}
Low IDs: {json.dumps(low_ids)}
Scores: {json.dumps(score_rows, ensure_ascii=False)}
Map results: {json.dumps(maps, ensure_ascii=False)}
Patch history: {json.dumps(history, ensure_ascii=False)}
Current SKILL.md:
{skill_text}""",
)
for field, label in (
("successful_pattern", "successful pattern"),
("failure_pattern", "failure pattern"),
):
pattern = result.get(field)
if not isinstance(pattern, dict) or not str(pattern.get("description", "")).strip():
raise ValueError(f"reducer returned no usable {label}")
evidence_ids = pattern.get("evidence_ids")
if not isinstance(evidence_ids, list) or not evidence_ids:
raise ValueError(f"reducer returned no evidence for {label}")
gap = result.get("selected_gap")
if not isinstance(gap, dict) or not str(gap.get("description", "")).strip():
raise ValueError("reducer returned no usable contrastive gap")
for field in ("success_behavior_to_preserve", "failure_behavior_to_mitigate"):
if not str(gap.get(field, "")).strip():
raise ValueError(f"reducer selected_gap missing {field}")
return result
def generate_patches(
self, skill_path: Path, reduction: dict[str, Any], history: list[dict[str, Any]], error: str = ""
) -> list[Patch]:
text = skill_path.read_text(encoding="utf-8")
result = self.client.json(
"Generate a small ordered patch bundle for a skill document. Return JSON only.",
f"""Generate one required success-oriented patch and, only when it adds distinct value, one optional failure-oriented patch. Both patches must address the same selected_gap; do not introduce unrelated improvements.
The required promote_success patch must express the selected successful behavior as a clear, actionable recommended workflow or stopping condition in the most appropriate existing section.
The optional mitigate_failure patch is allowed only when it adds non-duplicative detection, recovery, or exception-handling guidance. Omit it when it would merely negate, restate, or cross-reference the promote_success patch. If included, it must remain useful independently rather than existing only to repeat the preferred path.
Return patches in application order: promote_success first, then optional mitigate_failure. Each patch is one contiguous text replacement. For every patch, old_text must be a non-empty, uniquely occurring verbatim substring of the original SKILL.md and patches must target non-overlapping substrings so they can be applied sequentially. new_text must replace old_text locally and preserve general applicability. Do not rewrite the whole document; keep textual growth and behavioral scope minimal.
The patch bundle must preserve efficient successful behavior. It must not add significant tool cost, introduce mandatory tool or API calls, broaden the existing external search scope, require exhaustive checking when targeted checking is sufficient, or delay creation of a required deliverable. Prefer prioritization, bounded stopping criteria, and writing or updating required outputs as soon as the core result is supported. Do not turn a trace-specific failure into an unconditional every/all/always/never/only-after rule unless the task itself inherently requires that rule.
Return {{"patches":[{{"role":"promote_success|mitigate_failure","edit_type":string,"target_heading":string,"old_text":string,"new_text":string,"evidence_ids":[string],"evidence_type":string,"confidence":string,"reason":string}}]}}. The patches array must contain one or two items and must always begin with promote_success.
Selected analysis: {json.dumps(reduction, ensure_ascii=False)}
History: {json.dumps(history, ensure_ascii=False)}
Previous application error: {error}
SKILL.md:
{text}""",
)
values = result.get("patches")
if not isinstance(values, list) or not 1 <= len(values) <= 2:
raise ValueError("patch generator must return one or two patches")
if not all(isinstance(value, dict) for value in values):
raise ValueError("every generated patch must be an object")
expected_roles = ["promote_success", "mitigate_failure"]
roles = [value.get("role") for value in values]
if roles != expected_roles[: len(values)]:
raise ValueError(
"patch roles must be promote_success followed by optional mitigate_failure"
)
patches = [Patch.from_dict(value) for value in values]
if len(patches) == 2 and patches[0].new_text.strip() == patches[1].new_text.strip():
raise ValueError("failure patch duplicates the success patch")
skill_hash = sha256_file(skill_path)
for patch in patches:
patch.skill_hash = skill_hash
return patches
@@ -0,0 +1,77 @@
from __future__ import annotations
import shutil
import uuid
from pathlib import Path
from ..models import Patch
from ..storage import atomic_write_text, sha256_file
def _validate_skill_text(text: str) -> None:
if not text.strip():
raise ValueError("SKILL.md is empty")
if text.startswith("---"):
end = text.find("\n---", 3)
if end < 0:
raise ValueError("SKILL.md has an unterminated YAML frontmatter")
frontmatter = text[3:end].strip()
try:
import yaml
parsed = yaml.safe_load(frontmatter) if frontmatter else {}
except Exception as exc:
raise ValueError(f"invalid SKILL.md frontmatter: {exc}") from exc
if parsed is not None and not isinstance(parsed, dict):
raise ValueError("SKILL.md frontmatter must be a mapping")
def apply_patches(
current_package: Path, candidate_package: Path, patches: list[Patch]
) -> str:
skill = current_package / "SKILL.md"
if not skill.is_file():
raise ValueError(f"missing {skill}")
if not patches:
raise ValueError("at least one patch is required")
actual_hash = sha256_file(skill)
text = skill.read_text(encoding="utf-8")
for index, patch in enumerate(patches, start=1):
if not patch.skill_hash:
raise ValueError(f"patch {index} missing generation-time skill_hash")
if actual_hash != patch.skill_hash:
raise ValueError(
f"patch {index} was not generated from the current SKILL.md"
)
if not patch.old_text:
raise ValueError(f"patch {index} old_text must not be empty")
matches = text.count(patch.old_text)
if matches != 1:
raise ValueError(
f"patch {index} old_text must match exactly once; found {matches}"
)
changed = text.replace(patch.old_text, patch.new_text, 1)
if changed == text:
raise ValueError(f"patch {index} does not change SKILL.md")
_validate_skill_text(changed)
text = changed
token = uuid.uuid4().hex
temporary = candidate_package.with_name(f".{candidate_package.name}.{token}.tmp")
backup = candidate_package.with_name(f".{candidate_package.name}.{token}.bak")
shutil.copytree(current_package, temporary)
try:
atomic_write_text(temporary / "SKILL.md", text)
if candidate_package.exists():
candidate_package.rename(backup)
try:
temporary.rename(candidate_package)
except Exception:
if backup.exists() and not candidate_package.exists():
backup.rename(candidate_package)
raise
finally:
if temporary.exists():
shutil.rmtree(temporary)
if backup.exists() and candidate_package.exists():
shutil.rmtree(backup)
return sha256_file(candidate_package / "SKILL.md")
@@ -0,0 +1,21 @@
from __future__ import annotations
import random
def relative_high_low(
score_by_id: dict[str, float], count: int = 3, seed: str | int = 0
) -> tuple[list[str], list[str]]:
"""稳定选择互不重叠的相对高分组和低分组。"""
if len(score_by_id) < count * 2:
raise ValueError(f"need at least {count * 2} traces for disjoint High/Low groups")
trace_ids = list(score_by_id)
random.Random(str(seed)).shuffle(trace_ids)
high = sorted(trace_ids, key=score_by_id.__getitem__, reverse=True)[:count]
high_set = set(high)
low = sorted(
(trace_id for trace_id in trace_ids if trace_id not in high_set),
key=score_by_id.__getitem__,
)[:count]
return high, low
@@ -0,0 +1,234 @@
from __future__ import annotations
import json
import re
from pathlib import Path
from typing import Any, Iterable
from ..models import RolloutTrace
_TOOL_CALL_RE = re.compile(r"^Tool call (?P<name>[^:\n]+):[ \t]*", re.MULTILINE)
_TOOL_RESULT_RE = re.compile(
r"^Tool result(?: \((?P<name>[^;\n)]+)(?:;[ \t]*(?P<status>[^)\n]+))?\))?:?[ \t]*",
re.MULTILINE,
)
def _content(value: Any) -> str:
if value is None:
return ""
if isinstance(value, str):
return value
return json.dumps(value, ensure_ascii=False)
def opencode_events_to_state(
lines: Iterable[str], probe: str
) -> tuple[list[dict[str, str]], bool, int]:
state: list[dict[str, str]] = [{"role": "user", "content": probe}]
invoked = False
parsed = 0
for line in lines:
line = line.strip()
if not line:
continue
try:
event = json.loads(line)
except json.JSONDecodeError:
continue
if not isinstance(event, dict):
continue
parsed += 1
kind = str(event.get("type", event.get("event", ""))).lower()
part = event.get("part") if isinstance(event.get("part"), dict) else event
tool = part.get("tool") or part.get("name") or event.get("tool") or event.get("name")
title = part.get("title") or event.get("title") or ""
state_data = part.get("state") if isinstance(part.get("state"), dict) else {}
haystack = " ".join([str(kind), str(tool or ""), str(title), _content(part)])
if str(tool or "").lower() == "skill" or "<skill_content" in haystack.lower():
invoked = True
if tool or "tool" in kind:
arguments = state_data.get("input", part.get("input", part.get("arguments", {})))
output = state_data.get("output", part.get("output", part.get("result", "")))
state.append({"role": "assistant", "content": f"Tool call {tool or title}: {_content(arguments)}"})
if output not in (None, ""):
state.append({"role": "user", "content": f"Tool result: {_content(output)}"})
continue
text = part.get("text", part.get("content", event.get("message", "")))
if text not in (None, ""):
role = str(event.get("role", part.get("role", "assistant")))
if role not in {"assistant", "user", "system"}:
role = "assistant"
state.append({"role": role, "content": _content(text)})
return state, invoked, parsed
def read_event_file(path: Path, probe: str) -> tuple[list[dict[str, str]], bool, int]:
with path.open(encoding="utf-8", errors="replace") as handle:
return opencode_events_to_state(handle, probe)
def _split_tool_results(content: str) -> tuple[str, list[tuple[str, str, str]]]:
matches = list(_TOOL_RESULT_RE.finditer(content))
if not matches:
return content, []
prefix = content[:matches[0].start()].strip()
results = []
for index, match in enumerate(matches):
end = matches[index + 1].start() if index + 1 < len(matches) else len(content)
results.append((
(match.group("name") or "unknown").strip(),
(match.group("status") or "unknown").strip(),
content[match.end():end].strip(),
))
return prefix, results
def _excerpt(content: str, limit: int) -> str:
content = content.strip()
if len(content) <= limit:
return content
marker = "\n[... content omitted ...]\n"
if limit <= len(marker) + 2:
return content[:limit]
omitted = len(content) - (limit - len(marker))
while True:
marker = f"\n[... {omitted} chars omitted ...]\n"
available = limit - len(marker)
updated = len(content) - available
if updated == omitted:
break
omitted = updated
head = (available + 1) // 2
tail = available // 2
return content[:head] + marker + content[-tail:]
def _termination(trace: RolloutTrace) -> str:
value = trace.metadata.get("termination")
if value not in (None, ""):
return str(value)
if trace.timed_out is True:
return "timeout"
if trace.exit_code == 0:
return "completed"
if trace.exit_code is not None:
return "error"
return "unknown"
def _runtime_facts(trace: RolloutTrace) -> str:
metadata = trace.metadata
facts = {
"trace_id": trace.trace_id,
"termination": _termination(trace),
"timed_out": trace.timed_out,
"exit_code": trace.exit_code,
"agent_execution_seconds": metadata.get("agent_execution_seconds"),
"tool_calls": metadata.get("tool_calls"),
"skill_invoked": trace.skill_invoked,
"timeout_reason": metadata.get("timeout_reason"),
"error_category": metadata.get("error_category"),
"partial_trajectory": metadata.get("partial_trajectory"),
}
return "RUNTIME_FACTS " + json.dumps(facts, ensure_ascii=False, separators=(",", ":"))
def _allocate_excerpt_budget(caps: list[int], available: int) -> list[int]:
allocations = [0] * len(caps)
active = [index for index, cap in enumerate(caps) if cap > 0]
while active and available > 0:
share = max(1, available // len(active))
progressed = False
for index in active.copy():
amount = min(share, caps[index] - allocations[index], available)
allocations[index] += amount
available -= amount
progressed = progressed or amount > 0
if allocations[index] >= caps[index]:
active.remove(index)
if available == 0:
break
if not progressed:
break
return allocations
def compact_trace(trace: RolloutTrace, total: int = 30000) -> str:
"""Render runtime facts and a complete action ledger for one Map call."""
entries: list[dict[str, Any]] = []
for index, message in enumerate(trace.state):
event_id = f"E{index:03d}"
role = str(message.get("role", "unknown"))
content = _content(message.get("content", ""))
tool_call = _TOOL_CALL_RE.match(content)
if tool_call:
entries.append({
"id": f"{event_id}.T01",
"kind": "tool_call",
"role": role,
"name": tool_call.group("name").strip(),
"status": "unknown",
"content": content[tool_call.end():].strip(),
})
continue
prefix, tool_results = _split_tool_results(content) if index > 0 else (content, [])
if prefix:
entries.append({
"id": event_id,
"kind": "message",
"role": role,
"content": prefix,
})
for tool_index, (name, status, result) in enumerate(tool_results, 1):
entries.append({
"id": f"{event_id}.T{tool_index:02d}",
"kind": "tool_result",
"role": role,
"name": name,
"status": status,
"content": result,
})
message_entries = [entry for entry in entries if entry["kind"] == "message"]
first_user = next((entry for entry in message_entries if entry["role"] == "user"), None)
final_assistant = next(
(entry for entry in reversed(message_entries) if entry["role"] == "assistant"),
None,
)
skeletons = []
caps = []
for entry in entries:
if entry["kind"] == "message":
labels = ["message", f"role={entry['role']}"]
if entry is first_user:
labels.append("task")
if entry is final_assistant:
labels.append("final")
skeleton = f"{entry['id']} [" + " ".join(labels) + "]"
cap = 2500 if entry is first_user else 3000 if entry is final_assistant else 600
else:
name = str(entry["name"])[:80]
status = str(entry["status"])[:40]
skeleton = f"{entry['id']} [{entry['kind']} name={name} status={status}]"
cap = 600
skeletons.append(skeleton)
caps.append(min(cap, len(str(entry["content"]))))
facts = _runtime_facts(trace)
fixed_size = (
len(facts)
+ sum(len(skeleton) + 1 for skeleton in skeletons)
+ sum(1 for cap in caps if cap > 0)
)
allocations = _allocate_excerpt_budget(caps, max(0, total - fixed_size))
rendered = [facts]
for entry, skeleton, allocation in zip(entries, skeletons, allocations):
rendered.append(skeleton)
if allocation:
rendered.append(_excerpt(str(entry["content"]), allocation))
return "\n".join(rendered)
+50
View File
@@ -0,0 +1,50 @@
"""动态编译流水线的全部路径规则。"""
from __future__ import annotations
from pathlib import Path
def _find_project_root(module_dir: Path) -> Path:
"""通过项目标志定位根目录,不编码包目录深度。"""
for candidate in (module_dir, *module_dir.parents):
if (
(candidate / "provider_routes.json").is_file()
and (candidate / "scripts").is_dir()
and (candidate / "data").is_dir()
):
return candidate
raise RuntimeError(f"cannot locate project root from {module_dir}")
PROJECT_ROOT = _find_project_root(Path(__file__).resolve().parent)
DATA_ROOT = PROJECT_ROOT / "data"
RESULTS_ROOT = PROJECT_ROOT / "results"
ENV_FILE = PROJECT_ROOT / ".env"
DYNAMIC_RESULTS_ROOT = RESULTS_ROOT / "dynamic-optimization"
TRACE_ROOT = DYNAMIC_RESULTS_ROOT / "traces"
RAW_TRACE_ROOT = TRACE_ROOT / "raw_agent_trace"
FINAL_SCORE_ROOT = TRACE_ROOT / "final-score"
COMPILED_SKILL_ROOT = DYNAMIC_RESULTS_ROOT / "compiled-skills"
def project_path(value: Path) -> Path:
"""将命令行相对路径稳定地解释为项目根目录下的路径。"""
expanded = value.expanduser()
return (expanded if expanded.is_absolute() else PROJECT_ROOT / expanded).resolve()
def default_outputs(trace_input: Path, compile_type: str) -> tuple[Path, Path]:
"""按默认原始轨迹树中的身份生成评分和候选产物目录。"""
try:
relative = trace_input.resolve().relative_to(RAW_TRACE_ROOT)
except ValueError as exc:
raise ValueError(
"轨迹不在默认数据树中,请同时提供 --score-output 和 --output"
) from exc
identity = relative.parent / compile_type
return FINAL_SCORE_ROOT / identity, COMPILED_SKILL_ROOT / identity
+319
View File
@@ -0,0 +1,319 @@
"""动态编译主流水线;本模块只负责阶段编排。"""
from __future__ import annotations
import os
import sys
from pathlib import Path
from typing import Any
from .cache import (
SCORE_ARTIFACTS,
fingerprint,
load_candidate_cache,
load_maps_cache,
load_reduction_cache,
load_score_cache,
trace_fingerprint,
write_manifest,
)
from .optimization.analyzer import SemanticAnalyzer, SemanticClient
from .optimization.patch import apply_patches
from .optimization.selection import relative_high_low
from .paths import default_outputs, project_path
from .scoring.agentrm import (
DEFAULT_BATCH_SIZE,
DEFAULT_CONCURRENCY,
DEFAULT_MAX_LENGTH,
DEFAULT_RM_API_URL,
DEFAULT_TIMEOUT,
AgentRM,
)
from .scoring.service import TraceScorer
from .scoring.pre_score import PRE_SCORE_VERSION, PreScorer, RelevanceJudge
from .storage import (
atomic_write_json,
package_hash,
read_jsonl,
sha256_file,
)
from .traces.benchflow import load_benchflow_traces
GROUP_SIZE = 3
SCORE_VERSION = 1
MAP_VERSION = 1
REDUCTION_VERSION = 1
PATCH_VERSION = 1
def _log(message: str) -> None:
print(f"[dynamic_compile.fast] {message}", file=sys.stderr, flush=True)
def _implementation_name(value: object | None, default: str) -> str:
if value is None:
return default
return type(value).__module__ + "." + type(value).__qualname__
def run_pipeline(
trace_input: Path,
skill_package: Path,
*,
score_output: Path | None = None,
output: Path | None = None,
model: str = "opencode/deepseek-v4-pro",
max_parallel: int = 3,
rm_api_url: str | None = None,
rm_max_length: int = DEFAULT_MAX_LENGTH,
rm_timeout: float = DEFAULT_TIMEOUT,
rm_concurrency: int = DEFAULT_CONCURRENCY,
rm_batch_size: int = DEFAULT_BATCH_SIZE,
force: bool = False,
analyzer: SemanticAnalyzer | None = None,
pre_scorer: PreScorer | None = None,
agentrm: AgentRM | None = None,
) -> Path:
"""从历史 BenchFlow 轨迹生成一个候选 Skill 包。"""
trace_input = project_path(trace_input)
skill_package = project_path(skill_package)
if not (skill_package / "SKILL.md").is_file():
raise ValueError("skill package must contain a root SKILL.md")
if max_parallel < 1:
raise ValueError("max_parallel must be at least 1")
if agentrm is None and min(
rm_max_length, rm_timeout, rm_concurrency, rm_batch_size
) <= 0:
raise ValueError("AgentRM numeric options must be positive")
traces = load_benchflow_traces(trace_input)
identities = {(trace.task_name, trace.compile_type) for trace in traces}
if len(identities) != 1:
raise ValueError(
f"trace input must contain one task and compile type: {sorted(identities)}"
)
if len(traces) < GROUP_SIZE * 2:
raise ValueError(f"at least {GROUP_SIZE * 2} traces are required")
if len({trace.test_name for trace in traces}) != len(traces):
raise ValueError("trace input contains duplicate test names")
_, compile_type = next(iter(identities))
if score_output is None or output is None:
default_score, default_output = default_outputs(trace_input, compile_type)
score_dir = project_path(score_output) if score_output else default_score
output_dir = project_path(output) if output else default_output
if output_dir == skill_package or output_dir.is_relative_to(skill_package):
raise ValueError("output directory must not be inside the input skill package")
score_dir.mkdir(parents=True, exist_ok=True)
output_dir.mkdir(parents=True, exist_ok=True)
traces_hash = trace_fingerprint(traces)
resolved_rm_url = rm_api_url or os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
score_config: dict[str, Any] = {
"score_version": SCORE_VERSION,
"pre_score_version": PRE_SCORE_VERSION,
"model": model,
"pre_scorer": _implementation_name(pre_scorer, "PreScorer/RelevanceJudge"),
"agentrm": _implementation_name(agentrm, "AgentRM/HttpAgentRMBackend"),
"rm_api_url": resolved_rm_url,
"rm_max_length": rm_max_length,
"rm_timeout": rm_timeout,
"rm_concurrency": rm_concurrency,
"rm_batch_size": rm_batch_size,
}
score_input_hash = fingerprint({"traces": traces_hash, "config": score_config})
_log(f"loaded {len(traces)} traces from {trace_input}")
cached_score = load_score_cache(
traces, score_dir, input_hash=score_input_hash, config=score_config
)
if cached_score is not None:
scores, score_rows = cached_score
_log("reusing complete, input-matched scoring cache")
else:
scoring = TraceScorer(
pre_scorer or PreScorer(RelevanceJudge(model=model), max_parallel),
agentrm
or AgentRM(
api_url=resolved_rm_url,
max_length=rm_max_length,
timeout=rm_timeout,
concurrency=rm_concurrency,
batch_size=rm_batch_size,
),
)
_log("scoring traces")
scores = scoring.score_all(traces, score_dir)
score_rows = read_jsonl(score_dir / "effective_scores.jsonl")
write_manifest(
score_dir,
".score-cache.json",
stage="score",
input_hash=score_input_hash,
config=score_config,
artifacts=SCORE_ARTIFACTS,
)
high, low = relative_high_low(scores, count=GROUP_SIZE)
selected = high + low
semantic: SemanticAnalyzer | None = analyzer
def get_semantic() -> SemanticAnalyzer:
nonlocal semantic
if semantic is None:
semantic = SemanticAnalyzer(SemanticClient(model), max_parallel)
return semantic
semantic_implementation = _implementation_name(analyzer, "SemanticAnalyzer/SemanticClient")
map_config = {
"map_version": MAP_VERSION,
"model": model,
"semantic_implementation": semantic_implementation,
"max_parallel": max_parallel,
}
map_input_hash = fingerprint(
{
"traces": traces_hash,
"selected": selected,
"scores": {trace_id: scores[trace_id] for trace_id in selected},
}
)
maps = load_maps_cache(
output_dir,
selected,
input_hash=map_input_hash,
config=map_config,
)
if maps is not None:
_log(f"reusing complete Top {GROUP_SIZE} / Bottom {GROUP_SIZE} Map cache")
else:
_log(f"mapping Top {GROUP_SIZE} / Bottom {GROUP_SIZE} traces")
selected_set = set(selected)
maps = get_semantic().map_all(
[trace for trace in traces if trace.trace_id in selected_set],
scores,
set(high),
set(low),
progress=lambda done, total, trace_id: _log(
f"Map {done}/{total}: {trace_id}"
),
)
atomic_write_json(output_dir / "maps.json", maps)
write_manifest(
output_dir,
".maps-cache.json",
stage="maps",
input_hash=map_input_hash,
config=map_config,
artifacts=("maps.json",),
)
maps_by_id = {str(item["trace_id"]): item for item in maps}
scores_by_id = {str(item["trace_id"]): item for item in score_rows}
skill_path = skill_package / "SKILL.md"
skill_text = skill_path.read_text(encoding="utf-8")
skill_hash = sha256_file(skill_path)
skill_package_hash = package_hash(skill_package)
reduction_config = {
"reduction_version": REDUCTION_VERSION,
"model": model,
"semantic_implementation": semantic_implementation,
}
reduction_input_hash = fingerprint(
{
"traces": traces_hash,
"package": skill_package_hash,
"maps": [maps_by_id[trace_id] for trace_id in selected],
"score_rows": [scores_by_id[trace_id] for trace_id in selected],
"high": high,
"low": low,
}
)
reduction = None if force else load_reduction_cache(
output_dir,
input_hash=reduction_input_hash,
config=reduction_config,
)
if reduction is not None:
_log("reusing complete Top/Bottom reduction cache")
else:
_log("reducing Top/Bottom contrast")
reduction = get_semantic().reduce(
skill_text,
[maps_by_id[trace_id] for trace_id in selected],
[scores_by_id[trace_id] for trace_id in selected],
high,
low,
[],
)
atomic_write_json(output_dir / "reduction.json", reduction)
write_manifest(
output_dir,
".reduction-cache.json",
stage="reduction",
input_hash=reduction_input_hash,
config=reduction_config,
artifacts=("reduction.json",),
)
candidate = output_dir / "candidate-skill" / skill_package.name
candidate_config = {
"patch_version": PATCH_VERSION,
"model": model,
"semantic_implementation": semantic_implementation,
"attempts": 3,
}
candidate_input_hash = fingerprint(
{
"traces": traces_hash,
"package": skill_package_hash,
"reduction": reduction,
}
)
cached_candidate = None if force else load_candidate_cache(
output_dir,
skill_package.name,
skill_hash,
input_hash=candidate_input_hash,
config=candidate_config,
)
if cached_candidate is not None:
_log(f"reusing complete candidate skill: {cached_candidate}")
return cached_candidate
error = ""
for attempt in range(3):
try:
_log(f"generating patch bundle ({attempt + 1}/3)")
patches = get_semantic().generate_patches(skill_path, reduction, [], error)
candidate_hash = apply_patches(skill_package, candidate, patches)
atomic_write_json(
output_dir / "patch.json",
{
"patches": [patch.to_dict() for patch in patches],
"skill_hash": skill_hash,
"candidate_skill_hash": candidate_hash,
},
)
write_manifest(
output_dir,
".candidate-cache.json",
stage="candidate",
input_hash=candidate_input_hash,
config=candidate_config,
artifacts=(
"patch.json",
f"candidate-skill/{skill_package.name}",
),
)
_log(f"candidate skill ready: {candidate}")
return candidate
except (OSError, ValueError, RuntimeError) as exc:
error = str(exc)
if attempt == 2:
raise RuntimeError(
f"could not generate an applicable patch: {error}"
) from exc
raise AssertionError("unreachable")
@@ -0,0 +1,191 @@
from __future__ import annotations
from concurrent.futures import ThreadPoolExecutor, as_completed
import math
import os
import time
from typing import Any, Protocol
import requests
from .identity import composite_id
DEFAULT_MAX_LENGTH = 8192
DEFAULT_RM_API_URL = "http://127.0.0.1:28080"
DEFAULT_TIMEOUT = 300.0
DEFAULT_CONCURRENCY = 8
DEFAULT_BATCH_SIZE = 32
RETRY_ATTEMPTS = 3
class AgentRMBackend(Protocol):
def score(self, requests: list[dict[str, Any]]) -> list[dict[str, Any]]: ...
class HttpAgentRMBackend:
"""AgentRM backend backed by the remote ``/score_batch`` API."""
def __init__(
self,
api_url: str | None = None,
*,
max_length: int = DEFAULT_MAX_LENGTH,
timeout: float = DEFAULT_TIMEOUT,
concurrency: int = DEFAULT_CONCURRENCY,
batch_size: int = DEFAULT_BATCH_SIZE,
) -> None:
self.api_url = (
api_url or os.environ.get("RM_API_URL", DEFAULT_RM_API_URL)
).rstrip("/")
self.max_length = max_length
self.timeout = timeout
self.concurrency = concurrency
self.batch_size = batch_size
if not self.api_url:
raise ValueError("AgentRM API URL cannot be empty")
if max_length <= 0 or timeout <= 0 or concurrency <= 0 or batch_size <= 0:
raise ValueError(
"AgentRM max_length, timeout, concurrency, and batch_size must be positive"
)
def _post_batch(self, batch: list[dict[str, Any]]) -> list[dict[str, Any]]:
payload = {
"states": [request["state"] for request in batch],
"max_length": self.max_length,
}
last_error: BaseException | None = None
for attempt in range(RETRY_ATTEMPTS):
try:
response = requests.post(
f"{self.api_url}/score_batch",
json=payload,
timeout=self.timeout,
)
response.raise_for_status()
body = response.json()
scores = body.get("scores") if isinstance(body, dict) else None
if not isinstance(scores, list) or len(scores) != len(batch):
count = len(scores) if isinstance(scores, list) else "invalid"
raise ValueError(
f"AgentRM returned {count} scores for {len(batch)} states"
)
if any(not isinstance(score, dict) for score in scores):
raise ValueError("AgentRM returned a non-object score item")
if any(
"score" not in score or "n_tokens" not in score
for score in scores
):
raise ValueError("AgentRM returned a score item with missing fields")
return [
{
"task_name": request["task_name"],
"compile_type": request["compile_type"],
"test_name": request["test_name"],
**score,
}
for request, score in zip(batch, scores)
]
except (requests.RequestException, ValueError) as exc:
last_error = exc
if attempt + 1 < RETRY_ATTEMPTS:
time.sleep(2**attempt)
assert last_error is not None
raise RuntimeError(
f"AgentRM request failed after {RETRY_ATTEMPTS} attempts: {last_error}"
)
def score(self, requests_to_score: list[dict[str, Any]]) -> list[dict[str, Any]]:
if not requests_to_score:
return []
batches = [
requests_to_score[index : index + self.batch_size]
for index in range(0, len(requests_to_score), self.batch_size)
]
ordered: list[list[dict[str, Any]] | None] = [None] * len(batches)
with ThreadPoolExecutor(max_workers=self.concurrency) as pool:
futures = {
pool.submit(self._post_batch, batch): index
for index, batch in enumerate(batches)
}
for future in as_completed(futures):
ordered[futures[future]] = future.result()
return [row for batch in ordered if batch is not None for row in batch]
class AgentRM:
"""Score AgentRM requests through a validated, replaceable backend."""
def __init__(
self,
backend: AgentRMBackend | None = None,
*,
api_url: str | None = None,
max_length: int = DEFAULT_MAX_LENGTH,
timeout: float = DEFAULT_TIMEOUT,
concurrency: int = DEFAULT_CONCURRENCY,
batch_size: int = DEFAULT_BATCH_SIZE,
) -> None:
self.backend = (
backend
if backend is not None
else HttpAgentRMBackend(
api_url,
max_length=max_length,
timeout=timeout,
concurrency=concurrency,
batch_size=batch_size,
)
)
def score_requests(
self, requests_to_score: list[dict[str, Any]]
) -> list[dict[str, Any]]:
request_keys = [composite_id(request) for request in requests_to_score]
if len(set(request_keys)) != len(request_keys):
raise ValueError("AgentRM requests contain duplicate identities")
responses = self.backend.score(requests_to_score)
response_by_key: dict[tuple[str, str, str], dict[str, Any]] = {}
for response in responses:
key = composite_id(response)
if key in response_by_key:
raise ValueError(f"AgentRM returned duplicate score identity: {key}")
response_by_key[key] = response
requested = set(request_keys)
unexpected = set(response_by_key) - requested
missing = requested - set(response_by_key)
if unexpected:
raise ValueError(
f"AgentRM returned unexpected score identities: {sorted(unexpected)}"
)
if missing:
raise ValueError(f"AgentRM returned incomplete scores: {sorted(missing)}")
result = []
for request, key in zip(requests_to_score, request_keys):
response = response_by_key[key]
try:
score = float(response["score"])
except (KeyError, TypeError, ValueError) as exc:
raise ValueError(f"AgentRM returned an invalid score for {key}") from exc
if not math.isfinite(score):
raise ValueError(f"AgentRM returned a non-finite score for {key}")
try:
n_tokens = int(response["n_tokens"])
except (KeyError, TypeError, ValueError) as exc:
raise ValueError(
f"AgentRM returned an invalid n_tokens for {key}"
) from exc
if n_tokens < 0:
raise ValueError(f"AgentRM returned a negative n_tokens for {key}")
row = {
"task_name": str(request["task_name"]),
"compile_type": str(request["compile_type"]),
"test_name": str(request["test_name"]),
"score": score,
"n_tokens": n_tokens,
}
result.append(row)
return result
@@ -0,0 +1,13 @@
from __future__ import annotations
from typing import Any
from ..models import TraceKey
ScoreKey = tuple[str, str, str]
def composite_id(row: dict[str, Any]) -> ScoreKey:
return TraceKey.from_record(row).as_tuple()
return result
@@ -0,0 +1,254 @@
from __future__ import annotations
import json
import math
from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import asdict, dataclass, replace
from typing import Any, Protocol
from ..models import RolloutTrace
from ..optimization.analyzer import SemanticClient
RELEVANCE_MODEL = "opencode/deepseek-v4-pro"
RELEVANCE_CONFIDENCE_THRESHOLD = 0.80
TIMEOUT_SCORE = 0.0
IRRELEVANT_SCORE = 0.1
MAX_EFFICIENCY_PENALTY = 0.08
TIME_COST_WEIGHT = 0.80
TOOL_COST_WEIGHT = 0.20
EFFICIENCY_WINSOR_QUANTILE = 0.90
PRE_SCORE_VERSION = 3
class RelevanceClient(Protocol):
def json(self, system: str, user: str, attempts: int = 3) -> dict[str, Any]: ...
@dataclass(frozen=True)
class PreScoreResult:
trace_id: str
route: str
score: float | None
reason: str
relevance_label: str | None = None
relevance_confidence: float | None = None
efficiency_cost: float = 0.0
efficiency_penalty: float = 0.0
scoring_version: int = PRE_SCORE_VERSION
def to_dict(self) -> dict[str, Any]:
return asdict(self)
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "PreScoreResult":
return cls(
trace_id=str(value["trace_id"]),
route=str(value["route"]),
score=float(value["score"]) if value.get("score") is not None else None,
reason=str(value["reason"]),
relevance_label=(
str(value["relevance_label"])
if value.get("relevance_label") is not None
else None
),
relevance_confidence=(
float(value["relevance_confidence"])
if value.get("relevance_confidence") is not None
else None
),
efficiency_cost=float(value.get("efficiency_cost", 0.0)),
efficiency_penalty=float(value.get("efficiency_penalty", 0.0)),
scoring_version=int(value.get("scoring_version", 1)),
)
def adjust_agent_score(score: float, pre_score: PreScoreResult) -> float:
"""Apply the batch-relative efficiency penalty to an AgentRM quality score."""
if pre_score.route != "agentrm":
return score
return max(IRRELEVANT_SCORE, score - pre_score.efficiency_penalty)
def _quantile(values: list[float], quantile: float) -> float:
ordered = sorted(values)
if len(ordered) == 1:
return ordered[0]
position = (len(ordered) - 1) * quantile
lower = math.floor(position)
upper = math.ceil(position)
if lower == upper:
return ordered[lower]
fraction = position - lower
return ordered[lower] + fraction * (ordered[upper] - ordered[lower])
def _magnitude_costs(
values: dict[str, float],
*,
transform=lambda value: value,
power: float = 1.0,
) -> dict[str, float]:
if len(values) < 2:
return {trace_id: 0.0 for trace_id in values}
transformed = {trace_id: float(transform(value)) for trace_id, value in values.items()}
floor = min(transformed.values())
ceiling = _quantile(list(transformed.values()), EFFICIENCY_WINSOR_QUANTILE)
if ceiling <= floor:
return {trace_id: 0.0 for trace_id in values}
scale = ceiling - floor
return {
trace_id: min(1.0, max(0.0, (value - floor) / scale)) ** power
for trace_id, value in transformed.items()
}
class RelevanceJudge:
def __init__(
self,
client: RelevanceClient | None = None,
model: str = RELEVANCE_MODEL,
):
self.model = model
self.client = client or SemanticClient(model)
def judge(self, task_prompt: str, final_output: str) -> tuple[str, float]:
result = self.client.json(
"Judge only whether an agent's last output is relevant to its task. Return JSON only.",
f"""Classify the last agent output as relevant or irrelevant to the task.
An output is relevant if it attempts, plans, discusses, or reports work on the requested task, even when it is wrong, incomplete, brief, malformed, or lacks a final answer. Mark it irrelevant only when it clearly addresses a materially different task or topic. Do not judge correctness, completeness, or answer quality.
Return exactly {{"label":"relevant"|"irrelevant","confidence":number}} where confidence is between 0 and 1.
Input:
{json.dumps({"task_prompt": task_prompt, "last_agent_output": final_output}, ensure_ascii=False)}""",
)
if not isinstance(result, dict) or set(result) != {"label", "confidence"}:
raise ValueError("relevance response must contain exactly label and confidence")
label = result.get("label")
confidence = result.get("confidence")
if label not in {"relevant", "irrelevant"}:
raise ValueError("relevance label must be relevant or irrelevant")
if isinstance(confidence, bool) or not isinstance(confidence, (int, float)):
raise ValueError("relevance confidence must be a number")
confidence = float(confidence)
if not math.isfinite(confidence) or not 0.0 <= confidence <= 1.0:
raise ValueError("relevance confidence must be between 0 and 1")
return label, confidence
class PreScorer:
def __init__(
self,
judge: RelevanceJudge,
max_parallel: int = 3,
confidence_threshold: float = RELEVANCE_CONFIDENCE_THRESHOLD,
):
self.judge = judge
self.max_parallel = max(1, max_parallel)
self.confidence_threshold = confidence_threshold
@staticmethod
def _first_user_message(trace: RolloutTrace) -> str:
return next(
(
str(message.get("content", "")).strip()
for message in trace.state
if isinstance(message, dict)
and message.get("role") == "user"
and str(message.get("content", "")).strip()
),
"",
)
@staticmethod
def _last_assistant_message(trace: RolloutTrace) -> str:
return next(
(
str(message.get("content", "")).strip()
for message in reversed(trace.state)
if isinstance(message, dict)
and message.get("role") == "assistant"
and str(message.get("content", "")).strip()
),
"",
)
def score_one(self, trace: RolloutTrace) -> PreScoreResult:
if trace.timed_out:
return PreScoreResult(trace.trace_id, "fixed_score", TIMEOUT_SCORE, "hard_timeout")
final_output = self._last_assistant_message(trace)
if not final_output:
return PreScoreResult(trace.trace_id, "agentrm", None, "no_assistant_output")
task_prompt = self._first_user_message(trace)
try:
label, confidence = self.judge.judge(task_prompt, final_output)
except Exception:
return PreScoreResult(trace.trace_id, "agentrm", None, "judge_failed")
if label == "irrelevant" and confidence >= self.confidence_threshold:
return PreScoreResult(
trace.trace_id,
"fixed_score",
IRRELEVANT_SCORE,
"strongly_irrelevant",
label,
confidence,
)
reason = "low_confidence" if label == "irrelevant" else "relevant"
return PreScoreResult(
trace.trace_id, "agentrm", None, reason, label, confidence
)
def score_all(self, traces: list[RolloutTrace]) -> list[PreScoreResult]:
results: list[PreScoreResult | None] = [None] * len(traces)
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
futures = {
pool.submit(self.score_one, trace): index
for index, trace in enumerate(traces)
}
for future in as_completed(futures):
results[futures[future]] = future.result()
completed = [result for result in results if result is not None]
by_id = {trace.trace_id: trace for trace in traces}
eligible = {
result.trace_id for result in completed if result.route == "agentrm"
}
durations = {
trace_id: float(by_id[trace_id].metadata["agent_execution_seconds"])
for trace_id in eligible
if isinstance(
by_id[trace_id].metadata.get("agent_execution_seconds"), (int, float)
)
}
tool_calls = {
trace_id: float(by_id[trace_id].metadata["tool_calls"])
for trace_id in eligible
if isinstance(by_id[trace_id].metadata.get("tool_calls"), (int, float))
}
duration_cost = _magnitude_costs(durations, power=2.0)
tool_cost = _magnitude_costs(tool_calls, transform=math.log1p)
adjusted = []
for result in completed:
components = []
if result.trace_id in duration_cost:
components.append((TIME_COST_WEIGHT, duration_cost[result.trace_id]))
if result.trace_id in tool_cost:
components.append((TOOL_COST_WEIGHT, tool_cost[result.trace_id]))
total_weight = sum(weight for weight, _ in components)
cost = (
sum(weight * value for weight, value in components) / total_weight
if total_weight
else 0.0
)
adjusted.append(
replace(
result,
efficiency_cost=cost,
efficiency_penalty=MAX_EFFICIENCY_PENALTY * cost,
)
)
return adjusted
@@ -0,0 +1,75 @@
from __future__ import annotations
from pathlib import Path
from ..models import RolloutTrace
from ..storage import atomic_write_jsonl
from .agentrm import AgentRM
from .pre_score import PreScorer, adjust_agent_score
from .identity import composite_id
class TraceScorer:
"""Resolve pre-scores and AgentRM scores into one effective score per trace."""
def __init__(
self,
pre_scorer: PreScorer,
agentrm: AgentRM | None = None,
):
self.pre_scorer = pre_scorer
self.agentrm = agentrm if agentrm is not None else AgentRM()
def score_all(
self,
traces: list[RolloutTrace],
final_score_dir: Path,
agentrm_dir: Path | None = None,
) -> dict[str, float]:
final_score_dir.mkdir(parents=True, exist_ok=True)
agentrm_dir = agentrm_dir or final_score_dir
agentrm_dir.mkdir(parents=True, exist_ok=True)
results = self.pre_scorer.score_all(traces)
pre_scores = {result.trace_id: result for result in results}
if set(pre_scores) != {trace.trace_id for trace in traces}:
raise ValueError("pre-scorer returned incomplete or duplicate results")
atomic_write_jsonl(
final_score_dir / "pre_scores.jsonl",
[pre_scores[trace.trace_id].to_dict() for trace in traces],
)
agentrm_traces = [
trace for trace in traces if pre_scores[trace.trace_id].route == "agentrm"
]
requests = [trace.agentrm_request() for trace in agentrm_traces]
local_scores = agentrm_dir / "agentrm_scores.jsonl"
found = self.agentrm.score_requests(requests)
atomic_write_jsonl(local_scores, found)
by_key = {composite_id(row): float(row["score"]) for row in found}
resolved = []
scores: dict[str, float] = {}
for trace in traces:
pre_score = pre_scores[trace.trace_id]
if pre_score.route == "fixed_score":
if pre_score.score is None:
raise ValueError(f"fixed pre-score is missing for {trace.trace_id}")
score = pre_score.score
source = "pre_score"
else:
score = adjust_agent_score(by_key[trace.key.as_tuple()], pre_score)
source = "agentrm"
scores[trace.trace_id] = score
resolved.append({
"task_name": trace.task_name,
"compile_type": trace.compile_type,
"test_name": trace.test_name,
"trace_id": trace.trace_id,
"effective_score": score,
"score_source": source,
"pre_score_reason": pre_score.reason,
"agentrm_score": by_key.get(trace.key.as_tuple()),
"efficiency_cost": pre_score.efficiency_cost,
"efficiency_penalty": pre_score.efficiency_penalty,
})
atomic_write_jsonl(final_score_dir / "effective_scores.jsonl", resolved)
return scores
@@ -0,0 +1,121 @@
from __future__ import annotations
import hashlib
import json
from pathlib import Path
from typing import Any, Iterable
MAX_CHARS = 2000
LAST_MSG_BUDGET = 4000
HEAD_RATIO = 0.5
def compact(content: str, budget: int) -> str:
if len(content) <= budget:
return content
head = int(budget * HEAD_RATIO)
tail = budget - head - 50
return (
content[:head]
+ f"\n...[truncated {len(content)-head-tail} chars]...\n"
+ content[-tail:]
)
def _unwrap_text_content(value: str) -> str:
prefix = "@{type=text; text="
if value.startswith(prefix) and value.endswith("}"):
return value[len(prefix) : -1]
return value
def _tool_result_text(event: dict[str, Any]) -> str:
parts: list[str] = []
content_items = event.get("content")
if isinstance(content_items, list):
for item in content_items:
if not isinstance(item, dict) or item.get("type") != "content":
continue
value = item.get("content")
if isinstance(value, str) and value:
parts.append(_unwrap_text_content(value))
elif (
isinstance(value, dict)
and value.get("type") == "text"
and isinstance(value.get("text"), str)
and value["text"]
):
parts.append(value["text"])
return "\n".join(parts) if parts else "(no output)"
def acp_events_to_state(events: Iterable[dict[str, Any]]) -> list[dict[str, str]]:
merged: list[dict[str, str]] = []
for event in events:
event_type = event.get("type")
role: str | None = None
content: str | None = None
if event_type == "user_message":
role, content = "user", event.get("text")
elif event_type == "agent_thought":
role, content = "assistant", event.get("text")
elif event_type == "agent_message":
role, content = "assistant", event.get("text")
if content == "":
continue
elif event_type == "tool_call":
role = "user"
kind = event.get("kind", "other")
status = event.get("status", "unknown")
content = f"Tool result ({kind}; {status}):\n{_tool_result_text(event)}"
elif event_type == "agent_timeout":
continue
else:
continue
if not isinstance(content, str):
raise ValueError(f"{event_type} event has a non-string text/content value")
if merged and merged[-1]["role"] == role:
merged[-1]["content"] += "\n" + content
else:
merged.append({"role": role, "content": content})
if not merged:
raise ValueError("trajectory produced an empty state")
if merged[0]["role"] != "user":
raise ValueError("state must start with a user message")
seen: set[str] = set()
last_index = len(merged) - 1
for index, message in enumerate(merged):
budget = LAST_MSG_BUDGET if index == last_index else MAX_CHARS
content = compact(message["content"], budget)
if len(content) > 200:
digest = hashlib.md5(content.encode("utf-8")).hexdigest()
if digest in seen:
content = "[same as previous tool result]"
else:
seen.add(digest)
message["content"] = content
return merged
def read_acp_events(path: Path) -> list[dict[str, Any]]:
events: list[dict[str, Any]] = []
with path.open("r", encoding="utf-8") as handle:
for line_number, line in enumerate(handle, 1):
if not line.strip():
continue
try:
event = json.loads(line)
except json.JSONDecodeError as exc:
raise ValueError(f"invalid JSON on line {line_number}: {exc.msg}") from exc
if not isinstance(event, dict):
raise ValueError(f"line {line_number} is not a JSON object")
events.append(event)
return events
def read_acp_state(path: Path) -> list[dict[str, str]]:
return acp_events_to_state(read_acp_events(path))
+88
View File
@@ -0,0 +1,88 @@
from __future__ import annotations
import hashlib
import json
import os
import tempfile
from pathlib import Path
from typing import Any, Iterable
from .paths import ENV_FILE
def load_project_env() -> None:
try:
from dotenv import load_dotenv
except ImportError:
return
if ENV_FILE.is_file():
load_dotenv(ENV_FILE, override=False)
def atomic_write_text(path: Path, text: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
fd, tmp = tempfile.mkstemp(prefix=f".{path.name}.", dir=path.parent)
try:
with os.fdopen(fd, "w", encoding="utf-8", newline="") as handle:
handle.write(text)
handle.flush()
os.fsync(handle.fileno())
os.replace(tmp, path)
finally:
if os.path.exists(tmp):
os.unlink(tmp)
def atomic_write_json(path: Path, value: Any) -> None:
atomic_write_text(path, json.dumps(value, ensure_ascii=False, indent=2) + "\n")
def atomic_write_jsonl(path: Path, rows: Iterable[dict[str, Any]]) -> None:
atomic_write_text(path, "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows))
def load_json(path: Path, default: Any = None) -> Any:
if not path.exists():
return default
with path.open(encoding="utf-8") as handle:
return json.load(handle)
def read_jsonl(path: Path) -> list[dict[str, Any]]:
if not path.exists():
return []
rows: list[dict[str, Any]] = []
with path.open(encoding="utf-8") as handle:
for line_no, line in enumerate(handle, 1):
if not line.strip():
continue
value = json.loads(line)
if not isinstance(value, dict):
raise ValueError(f"{path}:{line_no}: expected a JSON object")
rows.append(value)
return rows
def sha256_bytes(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def sha256_file(path: Path) -> str:
return sha256_bytes(path.read_bytes())
def sha256_text(text: str) -> str:
return sha256_bytes(text.encode("utf-8"))
def package_manifest(root: Path) -> list[dict[str, Any]]:
return [
{"path": str(p.relative_to(root)), "sha256": sha256_file(p), "bytes": p.stat().st_size}
for p in sorted(root.rglob("*"))
if p.is_file()
]
def package_hash(root: Path) -> str:
payload = json.dumps(package_manifest(root), sort_keys=True, separators=(",", ":"))
return sha256_text(payload)
@@ -0,0 +1,133 @@
from __future__ import annotations
import json
from pathlib import Path
from typing import Any
from ..models import RolloutTrace
from ..scoring.state import acp_events_to_state, read_acp_events
VARIANT_TO_COMPILE_TYPE = {
"model_skill": "model_compile",
"ori_skill": "ori",
}
def event_to_state(trajectory_path: Path) -> tuple[list[dict[str, str]], bool]:
events = read_acp_events(trajectory_path)
skill_invoked = any(
event.get("type") == "tool_call"
and any(
str(event.get(field, "")).strip().lower() == "skill"
for field in ("title", "kind")
)
for event in events
)
return acp_events_to_state(events), skill_invoked
def trajectory_for_test(test_dir: Path) -> Path | None:
candidates = sorted(test_dir.rglob("acp_trajectory.jsonl"))
if not candidates:
return None
canonical = [path for path in candidates if "trajectory" in path.parts]
return canonical[0] if canonical else candidates[0]
def _result_for_trajectory(trajectory_path: Path) -> dict[str, Any]:
result_path = trajectory_path.parent.parent / "result.json"
if not result_path.is_file():
raise ValueError(f"missing structured BenchFlow result: {result_path}")
value = json.loads(result_path.read_text(encoding="utf-8"))
if not isinstance(value, dict):
raise ValueError(f"BenchFlow result must be a JSON object: {result_path}")
return value
def load_benchflow_trace(
test_dir: Path,
task_name: str,
compile_type: str,
) -> RolloutTrace:
trajectory = trajectory_for_test(test_dir)
if trajectory is None:
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
result = _result_for_trajectory(trajectory)
agent_timeout = result.get("agent_timeout_info")
idle_timeout = result.get("idle_timeout_info")
timeout_info = (
agent_timeout
if isinstance(agent_timeout, dict)
else idle_timeout if isinstance(idle_timeout, dict) else None
)
timed_out = timeout_info is not None
metadata: dict[str, Any] = {
"source": str(test_dir.resolve()),
"termination": "timeout" if timed_out else "completed",
}
timing = result.get("timing") if isinstance(result.get("timing"), dict) else {}
execution_seconds = timing.get("agent_execution")
if execution_seconds is None and isinstance(timeout_info, dict):
execution_seconds = timeout_info.get(
"wall_clock_elapsed_sec", timeout_info.get("timeout_sec")
)
if isinstance(execution_seconds, (int, float)):
metadata["agent_execution_seconds"] = float(execution_seconds)
if isinstance(result.get("n_tool_calls"), int):
metadata["tool_calls"] = result["n_tool_calls"]
if timed_out:
metadata["timeout_reason"] = timeout_info.get("reason")
metadata["timeout_seconds"] = timeout_info.get(
"timeout_sec", timeout_info.get("idle_timeout_sec")
)
metadata["partial_trajectory"] = bool(result.get("partial_trajectory", False))
metadata["error_category"] = result.get("error_category")
state, skill_invoked = event_to_state(trajectory)
return RolloutTrace(
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
task_name=task_name,
compile_type=compile_type,
test_name=test_dir.name,
state=state,
skill_invoked=skill_invoked,
timed_out=timed_out,
metadata=metadata,
)
def _variant_dirs(input_path: Path) -> list[tuple[Path, str, str]]:
if input_path.name in VARIANT_TO_COMPILE_TYPE:
return [(
input_path,
input_path.parent.name,
VARIANT_TO_COMPILE_TYPE[input_path.name],
)]
direct_variants = [
(input_path / variant_name, input_path.name, compile_type)
for variant_name, compile_type in VARIANT_TO_COMPILE_TYPE.items()
if (input_path / variant_name).is_dir()
]
if direct_variants:
return direct_variants
variants = []
for task_dir in sorted(path for path in input_path.iterdir() if path.is_dir()):
for variant_name, compile_type in VARIANT_TO_COMPILE_TYPE.items():
variant_dir = task_dir / variant_name
if variant_dir.is_dir():
variants.append((variant_dir, task_dir.name, compile_type))
return variants
def load_benchflow_traces(input_path: Path) -> list[RolloutTrace]:
input_path = input_path.resolve()
if not input_path.is_dir():
raise ValueError(f"BenchFlow input does not exist: {input_path}")
variants = _variant_dirs(input_path)
if not variants:
raise ValueError(f"no model_skill or ori_skill directories under {input_path}")
traces = []
for variant_dir, task_name, compile_type in variants:
for test_dir in sorted(path for path in variant_dir.glob("test-*") if path.is_dir()):
traces.append(load_benchflow_trace(test_dir, task_name, compile_type))
return traces