Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+53
View File
@@ -0,0 +1,53 @@
"""Deep 编译流水线的唯一命令行入口。"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from scripts.provider_router import parse_model_reference
from .pipeline import DeepLoop
def _provider_model(value: str) -> str:
try:
return parse_model_reference(value).value
except ValueError as exc:
raise argparse.ArgumentTypeError(str(exc)) from exc
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="python -m scripts.dynamic_compile.deep")
commands = parser.add_subparsers(dest="command", required=True)
run = commands.add_parser("run", help="run Deep Loop from a skill and its BenchFlow traces")
run.add_argument("--skill", type=Path, required=True)
run.add_argument("--traces", type=Path, required=True)
run.add_argument("--output", type=Path)
run.add_argument(
"--model",
required=True,
type=_provider_model,
help="all external model calls use this provider/model",
)
resume = commands.add_parser("resume", help="resume a Deep Loop run")
resume.add_argument("--run", type=Path, required=True)
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
loop = (
DeepLoop.create(args.skill, args.traces, args.output, model=args.model)
if args.command == "run"
else DeepLoop(args.run)
)
print(loop.drive())
return 0
except (OSError, ValueError, RuntimeError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
raise SystemExit(main())
@@ -0,0 +1,216 @@
"""BenchFlow 输入与运行时轨迹适配。"""
from __future__ import annotations
import json
import re
import shutil
import subprocess
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.traces.benchflow import trajectory_for_test
from scripts.dynamic_compile.fast.scoring.state import acp_events_to_state, read_acp_events
@dataclass
class BenchFlowInput:
task_name: str
task_dir: Path
agent: str
model: str
prompt: str
traces: list[RolloutTrace]
def _project_root() -> Path:
return Path(__file__).resolve().parents[4]
def _run_config(test_dir: Path) -> dict[str, Any]:
paths = sorted(test_dir.rglob("config.json"))
if not paths:
raise ValueError(f"missing BenchFlow run config under {test_dir}")
value = json.loads(paths[0].read_text(encoding="utf-8"))
if not isinstance(value, dict):
raise ValueError(f"invalid BenchFlow run config: {paths[0]}")
return value
def _prompt(test_dir: Path) -> str:
paths = sorted(test_dir.rglob("prompts.json"))
if not paths:
raise ValueError(f"missing BenchFlow prompts.json under {test_dir}")
value = json.loads(paths[0].read_text(encoding="utf-8"))
if not isinstance(value, list) or not value or not isinstance(value[0], str):
raise ValueError(f"invalid BenchFlow prompts: {paths[0]}")
return value[0].strip()
def load_runtime_trace(test_dir: Path, task_name: str, compile_type: str) -> RolloutTrace:
"""Load agent runtime evidence without opening verifier/result artifacts."""
trajectory = trajectory_for_test(test_dir)
if trajectory is None:
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
events = read_acp_events(trajectory)
state = acp_events_to_state(events)
skill_invoked = any(
event.get("type") == "tool_call"
and event.get("status") == "completed"
and any(
str(event.get(field, "")).strip().lower() == "skill"
for field in ("title", "kind")
)
for event in events
)
timed_out = any(event.get("type") == "agent_timeout" for event in events)
return RolloutTrace(
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
task_name=task_name,
compile_type=compile_type,
test_name=test_dir.name,
state=state,
skill_invoked=skill_invoked,
timed_out=timed_out,
metadata={
"termination": "timeout" if timed_out else "completed",
"tool_calls": sum(event.get("type") == "tool_call" for event in events),
},
)
def load_benchflow_input(source: Path) -> BenchFlowInput:
source = source.resolve()
tests = sorted(path for path in source.glob("test-*") if path.is_dir())[-5:]
if not tests:
raise ValueError(f"no test-* BenchFlow traces under {source}")
task_name = source.parent.name
traces = [load_runtime_trace(test, task_name, "custom") for test in tests]
configs = [_run_config(test) for test in tests]
agents = {str(item.get("agent", "")).strip() for item in configs}
models = {str(item.get("model", "")).strip() for item in configs}
if "" in agents or len(agents) != 1:
raise ValueError(f"BenchFlow traces do not identify one agent: {sorted(agents)}")
if "" in models or len(models) != 1:
raise ValueError(f"BenchFlow traces do not identify one model: {sorted(models)}")
prompts = {_prompt(test) for test in tests}
if len(prompts) != 1:
raise ValueError(f"BenchFlow traces contain {len(prompts)} different task prompts")
task_dir = _project_root() / "data" / "skills-bench" / "tasks" / task_name
if not (task_dir / "task.md").is_file():
raise ValueError(f"cannot resolve SkillsBench task directory: {task_dir}")
return BenchFlowInput(
task_name, task_dir, next(iter(agents)), next(iter(models)),
next(iter(prompts)), traces,
)
def _slug(value: str) -> str:
return re.sub(r"^-+|-+$", "", re.sub(r"[^a-z0-9]+", "-", value.lower()))
class SkillsBenchDevelopmentRolloutRunner:
"""Development-only runner; verifier output is never projected into traces."""
def __init__(
self,
context: BenchFlowInput,
work_root: Path,
max_parallel: int = 3,
archive_root: Path | None = None,
):
self.context = context
self.work_root = work_root
self.max_parallel = max_parallel
self.archive_root = archive_root
def _variant_dir(self, jobs_root: Path) -> Path:
return (
jobs_root
/ _slug(f"{self.context.agent}-{self.context.model}")
/ self.context.task_name
/ "custom_skill"
)
def _completed_traces(self, variant: Path, batch_id: str) -> list[RolloutTrace]:
completed: list[RolloutTrace] = []
for test in sorted(path for path in variant.glob("test-*") if path.is_dir()):
try:
requirement = json.loads(
(test / "required-skill.json").read_text(encoding="utf-8")
)
if requirement.get("invoked") is not True:
continue
trace = load_runtime_trace(test, self.context.task_name, batch_id)
except (AttributeError, json.JSONDecodeError, OSError, ValueError):
continue
completed.append(trace)
return completed
def artifacts_dir(self, batch_id: str) -> Path | None:
if self.archive_root is None:
return None
return self.archive_root / batch_id / "custom_skill"
def run_batch(
self,
skill_package: Path,
_prompt: str,
batch_id: str,
_task_name: str,
count: int,
progress: Any | None = None,
) -> list[RolloutTrace]:
jobs_root = self.work_root / batch_id
log_path = jobs_root / "runner.log"
variant = self._variant_dir(jobs_root)
archive = self.artifacts_dir(batch_id)
if archive is not None and archive.is_dir():
archived_traces = self._completed_traces(archive, batch_id)
if len(archived_traces) >= count:
traces = archived_traces
missing = 0
else:
variant.parent.mkdir(parents=True, exist_ok=True)
shutil.copytree(archive, variant, dirs_exist_ok=True)
traces = self._completed_traces(variant, batch_id)
missing = max(0, count - len(traces))
else:
traces = self._completed_traces(variant, batch_id)
missing = max(0, count - len(traces))
if missing:
jobs_root.mkdir(parents=True, exist_ok=True)
command = [
"bash", str(_project_root() / "scripts" / "evaluate" / "run-raw-task.sh"),
"--harness", self.context.agent,
"--model", self.context.model,
"--task", str(self.context.task_dir),
"--skill-source", str(skill_package.resolve()),
"--require-skill",
"--repeat", str(missing),
"--max-parallel", str(self.max_parallel),
"--output", str(variant),
]
with log_path.open("a", encoding="utf-8") as handle:
subprocess.run(command, stdout=handle, stderr=subprocess.STDOUT, text=True)
traces = self._completed_traces(variant, batch_id)
if archive is not None and variant.is_dir():
archive.parent.mkdir(parents=True, exist_ok=True)
shutil.copytree(variant, archive, dirs_exist_ok=True)
runner_log = jobs_root / "runner.log"
if runner_log.is_file():
shutil.copy2(runner_log, archive.parent / "runner.log")
traces = self._completed_traces(archive, batch_id)
if len(traces) < count:
raise RuntimeError(
f"BenchFlow produced {len(traces)}/{count} candidate traces; see {log_path}"
)
traces = traces[:count]
for index, trace in enumerate(traces, 1):
trace.trace_id = f"{batch_id}-{index:03d}"
trace.test_name = trace.trace_id
if progress is not None:
progress(index, len(traces), trace.trace_id)
return traces
@@ -0,0 +1,197 @@
"""语义模型评分与局部编辑适配。"""
from __future__ import annotations
import json
import re
from concurrent.futures import ThreadPoolExecutor, as_completed
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
RUBRICS = {
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
}
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
if not isinstance(values, list):
return []
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
return [
value for value in values
if isinstance(value, str)
and _EVIDENCE.fullmatch(value)
and value.startswith(prefixes)
]
class DeepAnalyzer:
def __init__(self, client: SemanticClient, max_parallel: int = 3):
self.client = client
self.max_parallel = max_parallel
@staticmethod
def _trace_payload(traces: list[RolloutTrace]) -> str:
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
def score_column(
self,
skill_text: str,
task: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
dimension: str,
) -> dict[str, CellScore]:
unit_payload = [
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
for unit in units
]
result = self.client.json(
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
Rubric: {RUBRICS[dimension]}
Task prompt:
{task}
Current units:
{json.dumps(unit_payload, ensure_ascii=False)}
Agent traces:
{self._trace_payload(traces)}
Current SKILL.md:
{skill_text}""",
)
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
raise ValueError(f"judge returned an invalid {dimension} column")
expected = {unit.unit_id for unit in units}
column: dict[str, CellScore] = {}
for item in result["scores"]:
if not isinstance(item, dict):
raise ValueError("judge score entries must be objects")
unit_id = str(item.get("unit_id", ""))
score = float(item.get("score"))
evidence = item.get("evidence", [])
if unit_id not in expected or unit_id in column:
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
evidence = valid_evidence(evidence, traces)
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
if set(column) != expected:
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
return column
def compare_cell(
self,
task: str,
incumbent_unit: SkillUnit,
candidate_unit: SkillUnit,
incumbent_traces: list[RolloutTrace],
candidate_traces: list[RolloutTrace],
dimension: str,
) -> dict[str, Any]:
result = self.client.json(
"Compare one incumbent and candidate skill unit. Return JSON only.",
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
Rubric: {RUBRICS[dimension]}
Task prompt:
{task}
Incumbent unit:
{incumbent_unit.text}
Candidate unit:
{candidate_unit.text}
Incumbent runtime facts:
{self._trace_payload(incumbent_traces)}
Candidate runtime facts:
{self._trace_payload(candidate_traces)}""",
)
if not isinstance(result.get("task_relevant"), bool):
raise ValueError("judge returned invalid task relevance")
incumbent_score = float(result.get("incumbent_score"))
candidate_score = float(result.get("candidate_score"))
for score in (incumbent_score, candidate_score):
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
raise ValueError(f"judge returned an invalid paired score: {score}")
regressions = result.get("runtime_regressions")
allowed = {
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
}
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
raise ValueError("judge returned invalid runtime regressions")
return {
"task_relevant": result["task_relevant"],
"incumbent_score": incumbent_score,
"candidate_score": candidate_score,
"candidate_evidence": valid_evidence(
result.get("candidate_evidence"), candidate_traces
),
"runtime_regressions": list(dict.fromkeys(regressions)),
"reason": str(result.get("reason", "")),
}
def score_matrix(
self,
skill_text: str,
task: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
level: str,
existing_columns: dict[str, dict[str, CellScore]] | None = None,
result_callback: Any | None = None,
) -> ScoreMatrix:
columns = dict(existing_columns or {})
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
futures = {
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
for dimension in missing
}
for future in as_completed(futures):
dimension = futures[future]
columns[dimension] = future.result()
if result_callback is not None:
result_callback(dimension, columns[dimension])
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
def generate_edit(
self,
coordinate: Coordinate,
unit: SkillUnit,
cell: CellScore,
rejected: list[dict[str, Any]],
task_prompt: str,
) -> LocalEdit:
result = self.client.json(
"Generate one bounded local edit for an agent skill. Return JSON only.",
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
Target dimension rubric: {RUBRICS[coordinate.dimension]}
Task prompt:
{task_prompt}
Target unit:
{unit.text}
Score: {cell.score}
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
Reason: {cell.reason}
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
)
edit = LocalEdit.from_dict(result)
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
raise ValueError("edit generator changed the target coordinate")
return edit
@@ -0,0 +1,195 @@
"""Markdown 单元解析与编辑边界校验。"""
from __future__ import annotations
from collections import Counter
import re
from .models import SkillUnit
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
rows: list[tuple[int, int, str]] = []
offset = 0
for line in text.splitlines(keepends=True):
rows.append((offset, offset + len(line), line))
offset += len(line)
return rows
def _headings(text: str) -> list[tuple[int, int, int, str]]:
found = []
fence_char = ""
fence_size = 0
for start, end, line in _line_offsets(text):
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if fence_char:
continue
match = _HEADING.match(line)
if match:
found.append((start, end, len(match.group(1)), match.group(2).strip()))
return found
def _frontmatter_end(text: str) -> int:
if not text.startswith("---"):
return 0
lines = text.splitlines(keepends=True)
offset = len(lines[0]) if lines else 0
for line in lines[1:]:
offset += len(line)
if line.strip() == "---":
return offset
return 0
def parse_sections(text: str) -> list[SkillUnit]:
headings = _headings(text)
body_start = _frontmatter_end(text)
body_headings = [item for item in headings if item[0] >= body_start]
title = body_headings[0] if body_headings else None
after_title = title[1] if title else body_start
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
if not candidates:
body = text[body_start:]
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
section_depth = min(item[2] for item in candidates)
peers = [item for item in candidates if item[2] == section_depth]
spans: list[tuple[int, int, str, int | None]] = []
preamble = text[after_title:peers[0][0]]
if preamble.strip():
spans.append((after_title, peers[0][0], "Preamble", section_depth))
for index, heading in enumerate(peers):
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
spans.append((heading[0], end, heading[3], heading[2]))
return [
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
for index, (start, end, heading, depth) in enumerate(spans)
]
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
replacement = preserve_unit_boundary(unit, new_text)
return document[:unit.start] + replacement + document[unit.end:]
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
text = section.text
base = section.start
rows = _line_offsets(text)
blocks: list[tuple[int, int]] = []
start: int | None = None
fence_char = ""
fence_size = 0
for row_start, row_end, line in rows:
fence = _FENCE.match(line)
if fence:
marker = fence.group(1)
if start is None:
start = row_start
if not fence_char:
fence_char, fence_size = marker[0], len(marker)
elif marker[0] == fence_char and len(marker) >= fence_size:
fence_char, fence_size = "", 0
continue
if not fence_char and not line.strip():
if start is not None:
blocks.append((start, row_start))
start = None
continue
if start is None:
start = row_start
if start is not None:
blocks.append((start, len(text)))
merged: list[tuple[int, int]] = []
index = 0
while index < len(blocks):
start, end = blocks[index]
block = text[start:end]
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
merged.append((start, blocks[index + 1][1]))
index += 2
else:
merged.append((start, end))
index += 1
blocks = merged
units = []
for index, (start, end) in enumerate(blocks):
block = text[start:end]
heading_match = next((item for item in _headings(block)), None)
units.append(SkillUnit(
f"{section.unit_id}.P{index + 1:03d}",
"paragraph",
section.unit_id,
heading_match[3] if heading_match else "",
block,
index,
base + start,
base + end,
heading_match[2] if heading_match else None,
))
return units
def fenced_blocks(text: str) -> Counter[str]:
blocks: list[str] = []
current: list[str] | None = None
fence_char = ""
fence_size = 0
for line in text.splitlines(keepends=True):
fence = _FENCE.match(line)
if current is None:
if fence:
marker = fence.group(1)
fence_char, fence_size = marker[0], len(marker)
current = [line]
continue
current.append(line)
if fence:
marker = fence.group(1)
if marker[0] == fence_char and len(marker) >= fence_size:
blocks.append("".join(current))
current = None
fence_char, fence_size = "", 0
return Counter(blocks)
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
if not new_text.strip() or new_text == unit.text:
raise ValueError("local edit must produce non-empty changed text")
if unit.level == "section":
old_headings = _headings(unit.text)
new_headings = _headings(new_text)
depth = unit.heading_depth
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
if old_peers != new_peers:
raise ValueError("section edit must preserve its peer heading")
if fenced_blocks(unit.text) != fenced_blocks(new_text):
raise ValueError("section edit must preserve fenced code contents")
elif section_depth is not None:
old_peers = [
(item[2], item[3]) for item in _headings(unit.text)
if item[2] <= section_depth
]
new_peers = [
(item[2], item[3]) for item in _headings(new_text)
if item[2] <= section_depth
]
if old_peers != new_peers:
raise ValueError("paragraph edit must not add or change a section heading")
+134
View File
@@ -0,0 +1,134 @@
"""领域模型。"""
from __future__ import annotations
from dataclasses import asdict, dataclass
from typing import Any
DIMENSIONS = (
"Clarity",
"Structure",
"Executability",
"Completeness",
"Constraint Salience",
)
@dataclass
class SkillUnit:
unit_id: str
level: str
parent_id: str | None
heading: str
text: str
order: int
start: int
end: int
heading_depth: int | None = None
@dataclass
class CellScore:
score: float
evidence: list[str]
reason: str
@dataclass
class Coordinate:
unit_id: str
dimension: str
normalized_gap: float
@dataclass
class LocalEdit:
unit_id: str
dimension: str
new_text: str
edit_summary: str
reason: str
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "LocalEdit":
required = {"unit_id", "dimension", "new_text", "edit_summary", "reason"}
missing = sorted(required - value.keys())
if missing:
raise ValueError(f"local edit missing fields: {', '.join(missing)}")
if not all(isinstance(value[key], str) for key in required):
raise ValueError("local edit fields must be strings")
return cls(**{key: value[key] for key in cls.__dataclass_fields__})
@dataclass
class ScoreMatrix:
level: str
units: list[SkillUnit]
columns: dict[str, dict[str, CellScore]]
def to_dict(self) -> dict[str, Any]:
return {
"level": self.level,
"units": [asdict(unit) for unit in self.units],
"columns": {
dimension: {unit_id: asdict(cell) for unit_id, cell in column.items()}
for dimension, column in self.columns.items()
},
}
@classmethod
def from_dict(cls, value: dict[str, Any]) -> "ScoreMatrix":
return cls(
level=str(value["level"]),
units=[SkillUnit(**item) for item in value["units"]],
columns={
dimension: {
unit_id: CellScore(float(cell["score"]), list(cell["evidence"]), str(cell["reason"]))
for unit_id, cell in column.items()
}
for dimension, column in value["columns"].items()
},
)
def unit(self, unit_id: str) -> SkillUnit:
return next(unit for unit in self.units if unit.unit_id == unit_id)
def normalized_gaps(self) -> dict[str, float]:
gaps: dict[str, float] = {}
for dimension in DIMENSIONS:
values = [self.columns[dimension][unit.unit_id].score for unit in self.units]
gaps[dimension] = (max(values) - min(values)) / 4.0 if values else 0.0
return gaps
def select_coordinate(
self,
threshold: float,
dimension: str | None = None,
excluded: set[tuple[str, str]] | None = None,
) -> Coordinate | None:
gaps = self.normalized_gaps()
excluded = excluded or set()
def weak_units(item: str) -> list[SkillUnit]:
column = self.columns[item]
maximum = max((cell.score for cell in column.values()), default=0.0)
return [
unit for unit in self.units
if (unit.unit_id, item) not in excluded
and (maximum - column[unit.unit_id].score) / 4.0 > threshold
]
available = [
item for item in ([dimension] if dimension else DIMENSIONS)
if item is not None and gaps[item] > threshold and weak_units(item)
]
if not available:
return None
dimension = max(available, key=lambda item: gaps[item])
target = min(
weak_units(dimension),
key=lambda unit: (self.columns[dimension][unit.unit_id].score, unit.order),
)
return Coordinate(
target.unit_id,
dimension,
gaps[dimension],
)
+826
View File
@@ -0,0 +1,826 @@
"""Deep 编译流水线编排。"""
from __future__ import annotations
import shutil
import sys
import tempfile
import uuid
from dataclasses import asdict, replace
from pathlib import Path
from typing import Any
from scripts.dynamic_compile.fast.models import RolloutTrace
from scripts.dynamic_compile.fast.storage import (
atomic_write_json,
atomic_write_jsonl,
atomic_write_text,
load_json,
package_hash,
read_jsonl,
sha256_file,
sha256_text,
)
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
from .adapters.benchflow import (
BenchFlowInput,
SkillsBenchDevelopmentRolloutRunner,
load_benchflow_input,
)
from .adapters.semantic import DeepAnalyzer, valid_evidence
from .core.models import (
CellScore,
Coordinate,
DIMENSIONS,
LocalEdit,
ScoreMatrix,
SkillUnit,
)
from .core.markdown import (
parse_paragraphs,
parse_sections,
preserve_unit_boundary,
replace_unit_text,
validate_edit,
)
ROLLOUTS = 3
MAX_SECTION_ITERATIONS = 6
MAX_PARAGRAPH_ITERATIONS = 3
MAX_EDIT_GENERATION_ATTEMPTS = 3
GAP_THRESHOLD = 0.375
REJECTION_LIMIT = 2
MAX_PARALLEL = 3
DEFAULT_MODEL = "ali/deepseek-v4-pro-0813"
def _log(message: str) -> None:
print(f"[deep] {message}", file=sys.stderr, flush=True)
def _trace_from_dict(value: dict[str, Any]) -> RolloutTrace:
return RolloutTrace(**value)
def _column_to_dict(column: dict[str, CellScore]) -> dict[str, Any]:
return {unit_id: asdict(cell) for unit_id, cell in column.items()}
def _column_from_dict(
value: dict[str, Any], traces: list[RolloutTrace]
) -> dict[str, CellScore]:
return {
unit_id: CellScore(
float(cell["score"]), valid_evidence(cell.get("evidence"), traces), str(cell["reason"])
)
for unit_id, cell in value.items()
}
class DeepLoop:
def __init__(
self,
run_dir: Path,
analyzer: DeepAnalyzer | None = None,
runner: Any | None = None,
):
self.run_dir = run_dir.resolve()
self.state_path = self.run_dir / "run.json"
state = load_json(self.state_path)
if not isinstance(state, dict):
raise ValueError(f"invalid or missing run state: {self.state_path}")
self.state = state
self.temp = self.run_dir / ".tmp"
self.current = self.temp / "current"
self.model = state.get("model", state.get("semantic_model", DEFAULT_MODEL))
self.analyzer = analyzer or DeepAnalyzer(
SemanticClient(self.model), MAX_PARALLEL,
)
context = BenchFlowInput(
state["task"]["name"],
Path(state["task"]["directory"]),
state["task"]["agent"],
state["task"]["model"],
state["task"]["prompt"],
[],
)
self.runner = runner or SkillsBenchDevelopmentRolloutRunner(
context,
Path(tempfile.gettempdir()) / "skill-compiler-deep" / state["run_id"],
MAX_PARALLEL,
archive_root=self.run_dir / "rollouts",
)
@classmethod
def create(
cls,
skill: Path,
traces: Path,
output: Path | None = None,
*,
model: str = DEFAULT_MODEL,
analyzer: DeepAnalyzer | None = None,
runner: Any | None = None,
) -> "DeepLoop":
skill = skill.resolve()
if not (skill / "SKILL.md").is_file():
raise ValueError("--skill must be a skill package containing SKILL.md")
context = load_benchflow_input(traces)
target = (output or skill.parent / f"{skill.name}-deep").resolve()
if target.exists() and any(target.iterdir()):
raise ValueError(f"deep run directory is not empty: {target}")
target.mkdir(parents=True, exist_ok=True)
shutil.copytree(skill, target / "S_fast")
(target / ".tmp").mkdir()
(target / "levels").mkdir()
shutil.copytree(target / "S_fast", target / ".tmp" / "current")
atomic_write_jsonl(target / "input-traces.jsonl", [asdict(trace) for trace in context.traces])
state = {
"run_id": uuid.uuid4().hex[:12],
"status": "created",
"model": model,
"skill_name": skill.name,
"task": {
"name": context.task_name,
"directory": str(context.task_dir),
"agent": context.agent,
"model": context.model,
"prompt": context.prompt,
},
"current_rollouts": str(traces.resolve()),
"current_rollout_skill_sha256": sha256_file(skill / "SKILL.md"),
"levels": {},
}
atomic_write_json(target / "run.json", state)
return cls(target, analyzer=analyzer, runner=runner)
def _save(self) -> None:
atomic_write_json(self.state_path, self.state)
def _prompt(self) -> str:
return str(self.state["task"]["prompt"])
def _current_text(self) -> str:
return (self.current / "SKILL.md").read_text(encoding="utf-8")
def _load_or_rollout(
self,
package: Path,
trace_path: Path,
batch_id: str,
seed: list[RolloutTrace] | None = None,
) -> list[RolloutTrace]:
if trace_path.is_file():
return [_trace_from_dict(item) for item in read_jsonl(trace_path)]
if seed is not None:
traces = seed
else:
_log(f"Starting {ROLLOUTS} rollouts for {batch_id}")
traces = self.runner.run_batch(
package,
self._prompt(),
batch_id,
str(self.state["task"]["name"]),
ROLLOUTS,
progress=lambda done, total, trace_id: _log(
f"Rollout {done}/{total} complete: {trace_id}"
),
)
atomic_write_jsonl(trace_path, [asdict(trace) for trace in traces])
return traces
def _load_or_matrix(
self,
path: Path,
level: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
) -> ScoreMatrix:
value = load_json(path)
if isinstance(value, dict):
return ScoreMatrix.from_dict(value)
columns_dir = path.parent / "matrix-columns"
cached_columns: dict[str, dict[str, CellScore]] = {}
for dimension in DIMENSIONS:
cached = load_json(columns_dir / f"{dimension.lower().replace(' ', '-')}.json")
if isinstance(cached, dict):
cached_columns[dimension] = _column_from_dict(cached, traces)
def save_column(dimension: str, column: dict[str, CellScore]) -> None:
atomic_write_json(
columns_dir / f"{dimension.lower().replace(' ', '-')}.json",
_column_to_dict(column),
)
_log(f"Scoring full {level} matrix with {self.model}")
matrix = self.analyzer.score_matrix(
self._current_text(), self._prompt(), units, traces, level,
existing_columns=cached_columns,
result_callback=save_column,
)
atomic_write_json(path, matrix.to_dict())
return matrix
def _refresh_matrix(
self,
level: str,
units: list[SkillUnit],
traces: list[RolloutTrace],
) -> ScoreMatrix:
level_state = self.state["levels"][level]
number = int(level_state.get("refreshes", 0)) + 1
path = (
self.run_dir
/ "levels"
/ level
/ "refreshes"
/ f"refresh-{number:02d}"
/ "matrix.json"
)
_log(f"Refreshing full {level} matrix")
matrix = self._load_or_matrix(path, level, units, traces)
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
level_state["refreshes"] = number
level_state.pop("active_dimension", None)
self._save()
return matrix
def _decisions(self, level: str | None = None) -> list[dict[str, Any]]:
roots = (
[self.run_dir / "levels" / level]
if level else list((self.run_dir / "levels").glob("*"))
)
decisions = []
for root in roots:
for path in sorted((root / "iterations").glob("iteration-*/decision.json")):
value = load_json(path)
if isinstance(value, dict):
decisions.append(value)
return decisions
def _rejected(self, coordinate: Coordinate) -> list[dict[str, Any]]:
return [
decision["rejected_edit"]
for decision in self._decisions()
if not decision["accepted"]
and decision["rejected_edit"]["unit_id"] == coordinate.unit_id
and decision["rejected_edit"]["dimension"] == coordinate.dimension
]
def _exhausted(self, level: str) -> set[tuple[str, str]]:
counts: dict[tuple[str, str], int] = {}
for decision in self._decisions(level):
if decision["accepted"]:
continue
rejected = decision["rejected_edit"]
key = (rejected["unit_id"], rejected["dimension"])
counts[key] = counts.get(key, 0) + 1
return {key for key, count in counts.items() if count >= REJECTION_LIMIT}
@staticmethod
def _block_dimension(level_state: dict[str, Any], dimension: str) -> None:
blocked = set(level_state.get("blocked_dimensions", []))
blocked.add(dimension)
level_state["blocked_dimensions"] = sorted(blocked)
@staticmethod
def _candidate_units(
units: list[SkillUnit], target: SkillUnit, new_text: str
) -> list[SkillUnit]:
delta = len(new_text) - len(target.text)
updated = []
for unit in units:
value = replace(unit)
if unit.unit_id == target.unit_id:
value.text = new_text
value.end = value.start + len(new_text)
elif unit.start >= target.end:
value.start += delta
value.end += delta
updated.append(value)
return updated
def _apply_edit(
self,
unit: SkillUnit,
edit: LocalEdit,
candidate: Path,
section_depth: int | None,
) -> None:
self._validate_edit_candidate(unit, edit, section_depth)
text = self._current_text()
changed = replace_unit_text(text, unit, edit.new_text)
temporary = candidate.with_name(f".{candidate.name}.{uuid.uuid4().hex}.tmp")
shutil.copytree(self.current, temporary)
atomic_write_text(temporary / "SKILL.md", changed)
if candidate.exists():
shutil.rmtree(candidate)
candidate.parent.mkdir(parents=True, exist_ok=True)
temporary.rename(candidate)
def _validate_edit_candidate(
self,
unit: SkillUnit,
edit: LocalEdit,
section_depth: int | None,
) -> None:
"""Validate a local edit against both unit and whole-document invariants."""
validate_edit(unit, edit.new_text, section_depth)
text = self._current_text()
if text[unit.start:unit.end] != unit.text:
raise ValueError("target unit no longer matches current SKILL.md")
changed = replace_unit_text(text, unit, edit.new_text)
before = [(item.heading, item.heading_depth) for item in parse_sections(text)]
after = [(item.heading, item.heading_depth) for item in parse_sections(changed)]
if before != after:
raise ValueError("local edit changed section boundaries")
@staticmethod
def _validation_feedback(
attempts: list[dict[str, Any]],
) -> list[dict[str, Any]]:
return [
{
"edit_summary": str(item.get("edit_summary", "invalid generated edit")),
"reject_reason": f"structural_validation_failed: {item['error']}",
"new_text_hash": str(item.get("new_text_hash", "")),
}
for item in attempts
]
def _valid_edit_or_rejection(
self,
*,
level: str,
number: int,
iteration_dir: Path,
unit: SkillUnit,
coordinate: Coordinate,
cell: CellScore,
section_depth: int | None,
) -> tuple[LocalEdit | None, bool, list[dict[str, Any]]]:
"""Load or generate a valid edit, feeding structural failures back to the model."""
edit_path = iteration_dir / "edit.json"
attempts_path = iteration_dir / "edit-attempts.json"
attempts_value = load_json(attempts_path, [])
attempts = attempts_value if isinstance(attempts_value, list) else []
cached_value = load_json(edit_path)
if isinstance(cached_value, dict):
try:
cached = LocalEdit.from_dict(cached_value)
cached.new_text = preserve_unit_boundary(unit, cached.new_text)
self._validate_edit_candidate(unit, cached, section_depth)
return cached, False, attempts
except ValueError as exc:
text = str(cached_value.get("new_text", ""))
attempts.append({
"source": "cached",
"error": str(exc),
"edit_summary": str(cached_value.get("edit_summary", "")),
"new_text_hash": sha256_text(text) if text else "",
})
atomic_write_json(attempts_path, attempts)
_log(
f"{level} iteration {number}: cached local edit is invalid: {exc}; "
"regenerating"
)
for attempt in range(1, MAX_EDIT_GENERATION_ATTEMPTS + 1):
_log(
f"{level} iteration {number}: generating local edit for "
f"{coordinate.unit_id}/{coordinate.dimension} "
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS})"
)
feedback = self._rejected(coordinate) + self._validation_feedback(attempts)
edit: LocalEdit | None = None
try:
edit = self.analyzer.generate_edit(
coordinate,
unit,
cell,
feedback,
self._prompt(),
)
edit.new_text = preserve_unit_boundary(unit, edit.new_text)
self._validate_edit_candidate(unit, edit, section_depth)
except ValueError as exc:
text = edit.new_text if edit is not None else ""
attempts.append({
"source": "generated",
"generation_attempt": attempt,
"error": str(exc),
"edit_summary": edit.edit_summary if edit is not None else "",
"new_text_hash": sha256_text(text) if text else "",
})
atomic_write_json(attempts_path, attempts)
_log(
f"{level} iteration {number}: local edit validation failed "
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS}): {exc}"
)
continue
assert edit is not None
atomic_write_json(edit_path, asdict(edit))
return edit, True, attempts
return None, True, attempts
def _commit_iteration(
self,
level: str,
number: int,
iteration_dir: Path,
matrix: ScoreMatrix,
current_trace_path: Path,
) -> ScoreMatrix:
decision = load_json(iteration_dir / "decision.json")
if not isinstance(decision, dict):
raise ValueError("missing iteration decision")
if decision["accepted"]:
candidate = iteration_dir / "candidate" / self.state["skill_name"]
replacement = self.temp / "next-current"
shutil.rmtree(replacement, ignore_errors=True)
shutil.copytree(candidate, replacement)
shutil.rmtree(self.current)
replacement.rename(self.current)
matrix = ScoreMatrix.from_dict(decision["matrix_after"])
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
candidate_traces = read_jsonl(iteration_dir / "candidate-traces.jsonl")
atomic_write_jsonl(current_trace_path, candidate_traces)
rollout_output = decision.get("rollout_output")
rollout_skill_sha256 = decision.get("rollout_skill_sha256")
if isinstance(rollout_output, str) and isinstance(rollout_skill_sha256, str):
self.state["current_rollouts"] = rollout_output
self.state["current_rollout_skill_sha256"] = rollout_skill_sha256
else:
self.state.pop("current_rollouts", None)
self.state.pop("current_rollout_skill_sha256", None)
level_state = self.state["levels"][level]
if int(level_state.get("iterations", 0)) < number:
level_state["iterations"] = number
self._save()
return matrix
def _run_iteration(
self,
level: str,
number: int,
level_dir: Path,
matrix: ScoreMatrix,
coordinate: Coordinate,
current_trace_path: Path,
section_depth: int | None,
) -> ScoreMatrix:
iteration_dir = level_dir / "iterations" / f"iteration-{number:02d}"
iteration_dir.mkdir(parents=True, exist_ok=True)
unit = matrix.unit(coordinate.unit_id)
edit, regenerated, validation_attempts = self._valid_edit_or_rejection(
level=level,
number=number,
iteration_dir=iteration_dir,
unit=unit,
coordinate=coordinate,
cell=matrix.columns[coordinate.dimension][coordinate.unit_id],
section_depth=section_depth,
)
if edit is None:
decision = {
"coordinate": asdict(coordinate),
"accepted": False,
"reason": "edit_validation_exhausted",
"target_delta": 0.0,
"validation_attempts": validation_attempts,
"rejected_edit": {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": (
"Could not generate a structurally valid local edit after "
f"{MAX_EDIT_GENERATION_ATTEMPTS} attempts."
),
"score_change": 0.0,
"reject_reason": "edit_validation_exhausted",
"new_text_hash": str(
validation_attempts[-1].get("new_text_hash", "")
) if validation_attempts else "",
},
}
atomic_write_json(iteration_dir / "decision.json", decision)
_log(
f"{level} iteration {number}: local edit validation exhausted; "
"recording rejection and continuing"
)
return self._commit_iteration(
level, number, iteration_dir, matrix, current_trace_path
)
edit_hash = sha256_text(edit.new_text)
if any(item["new_text_hash"] == edit_hash for item in self._rejected(coordinate)):
decision = {
"coordinate": asdict(coordinate),
"accepted": False,
"reason": "exact_duplicate_rejected_edit",
"target_delta": 0.0,
"rejected_edit": {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": edit.edit_summary,
"score_change": 0.0,
"reject_reason": "exact_duplicate_rejected_edit",
"new_text_hash": edit_hash,
},
}
atomic_write_json(iteration_dir / "decision.json", decision)
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
candidate = iteration_dir / "candidate" / self.state["skill_name"]
if regenerated:
shutil.rmtree(candidate, ignore_errors=True)
for stale in (
iteration_dir / "candidate-traces.jsonl",
iteration_dir / "comparison.json",
):
if stale.exists():
stale.unlink()
if not (candidate / "SKILL.md").is_file():
self._apply_edit(unit, edit, candidate, section_depth)
candidate_units = self._candidate_units(matrix.units, unit, edit.new_text)
candidate_skill_sha256 = sha256_file(candidate / "SKILL.md")
batch_id = (
f"{self.state['run_id']}-deep-{level}-i{number:02d}-"
f"{candidate_skill_sha256[:12]}"
)
traces = self._load_or_rollout(
candidate,
iteration_dir / "candidate-traces.jsonl",
batch_id,
)
comparison_path = iteration_dir / "comparison.json"
comparison = load_json(comparison_path)
if not isinstance(comparison, dict):
_log(
f"{level} iteration {number}: comparing "
f"{coordinate.unit_id}/{coordinate.dimension}"
)
incumbent_traces = [
_trace_from_dict(item) for item in read_jsonl(current_trace_path)
]
comparison = self.analyzer.compare_cell(
self._prompt(),
unit,
next(item for item in candidate_units if item.unit_id == unit.unit_id),
incumbent_traces,
traces,
coordinate.dimension,
)
incumbent_timeout_rate = (
sum(trace.timed_out is True for trace in incumbent_traces)
/ len(incumbent_traces)
)
candidate_timeout_rate = (
sum(trace.timed_out is True for trace in traces) / len(traces)
)
if (
candidate_timeout_rate > incumbent_timeout_rate
and "timeout" not in comparison["runtime_regressions"]
):
comparison["runtime_regressions"].append("timeout")
atomic_write_json(comparison_path, comparison)
delta = float(comparison["candidate_score"]) - float(
comparison["incumbent_score"]
)
if not comparison["task_relevant"]:
accepted, reason = False, "target_unit_not_task_relevant"
elif comparison["runtime_regressions"]:
accepted = False
reason = "runtime_regressed:" + ",".join(comparison["runtime_regressions"])
elif delta < 0.5:
accepted, reason = False, "target_cell_did_not_improve"
else:
accepted, reason = True, "target_improved_without_runtime_regression"
decision: dict[str, Any] = {
"coordinate": asdict(coordinate),
"accepted": accepted,
"reason": reason,
"target_delta": delta,
"rollout_skill_sha256": candidate_skill_sha256,
}
artifacts_dir = getattr(self.runner, "artifacts_dir", None)
rollout_output = artifacts_dir(batch_id) if callable(artifacts_dir) else None
if isinstance(rollout_output, Path):
decision["rollout_output"] = str(rollout_output.resolve())
if accepted:
updated = ScoreMatrix(matrix.level, candidate_units, dict(matrix.columns))
updated.columns[coordinate.dimension] = dict(
matrix.columns[coordinate.dimension]
)
updated.columns[coordinate.dimension][coordinate.unit_id] = CellScore(
float(comparison["candidate_score"]),
list(comparison["candidate_evidence"]),
str(comparison["reason"]),
)
decision["matrix_after"] = updated.to_dict()
else:
decision["rejected_edit"] = {
"unit_id": coordinate.unit_id,
"dimension": coordinate.dimension,
"edit_summary": edit.edit_summary,
"score_change": delta,
"reject_reason": reason,
"new_text_hash": edit_hash,
}
atomic_write_json(iteration_dir / "decision.json", decision)
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
def _run_level(
self,
level: str,
units: list[SkillUnit],
seed_traces: list[RolloutTrace] | None = None,
section_depth: int | None = None,
) -> tuple[ScoreMatrix, list[RolloutTrace], Coordinate | None]:
level_dir = self.run_dir / "levels" / level
level_dir.mkdir(parents=True, exist_ok=True)
level_state = self.state["levels"].setdefault(level, {
"iterations": 0, "completed": False,
})
current_trace_path = level_dir / "current-traces.jsonl"
traces = self._load_or_rollout(
self.current,
current_trace_path,
f"{self.state['run_id']}-deep-{level}-initial",
seed=seed_traces,
)
matrix = self._load_or_matrix(level_dir / "matrix.json", level, units, traces)
if level_state.get("completed"):
exhausted_value = level_state.get("exhausted_coordinate")
exhausted = Coordinate(**exhausted_value) if isinstance(exhausted_value, dict) else None
return matrix, traces, exhausted
exhausted_coordinate: Coordinate | None = None
max_iterations = (
MAX_SECTION_ITERATIONS if level == "section" else MAX_PARAGRAPH_ITERATIONS
)
while int(level_state["iterations"]) < max_iterations:
excluded = self._exhausted(level)
excluded.update(
(unit.unit_id, dimension)
for dimension in level_state.get("blocked_dimensions", [])
for unit in matrix.units
)
pending_number = int(level_state["iterations"]) + 1
pending_dir = level_dir / "iterations" / f"iteration-{pending_number:02d}"
pending_decision = load_json(pending_dir / "decision.json")
if isinstance(pending_decision, dict):
pending_coordinate = Coordinate(**pending_decision["coordinate"])
level_state.setdefault("active_dimension", pending_coordinate.dimension)
matrix = self._commit_iteration(
level, pending_number, pending_dir, matrix, current_trace_path
)
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
if len(self._rejected(pending_coordinate)) >= REJECTION_LIMIT:
exhausted_coordinate = pending_coordinate
level_state["stop_reason"] = "coordinate_exhausted"
level_state["exhausted_coordinate"] = asdict(pending_coordinate)
break
if matrix.select_coordinate(
GAP_THRESHOLD, level_state.get("active_dimension"), excluded
) is None:
matrix = self._refresh_matrix(level, matrix.units, traces)
continue
active_dimension = level_state.get("active_dimension")
coordinate = matrix.select_coordinate(
GAP_THRESHOLD, active_dimension, excluded
)
if coordinate is None:
if active_dimension is None:
level_state["stop_reason"] = (
"normalized_gap_converged"
if max(matrix.normalized_gaps().values(), default=0.0) <= GAP_THRESHOLD
else "available_coordinates_exhausted"
)
break
matrix = self._refresh_matrix(level, matrix.units, traces)
continue
if active_dimension is None:
level_state["active_dimension"] = coordinate.dimension
self._save()
number = int(level_state["iterations"]) + 1
matrix = self._run_iteration(
level, number, level_dir, matrix, coordinate,
current_trace_path, section_depth,
)
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
if len(self._rejected(coordinate)) >= REJECTION_LIMIT:
exhausted_coordinate = coordinate
level_state["stop_reason"] = "coordinate_exhausted"
level_state["exhausted_coordinate"] = asdict(coordinate)
break
else:
decisions = self._decisions(level)
last = decisions[-1] if decisions else {}
if matrix.select_coordinate(GAP_THRESHOLD) is None:
level_state["stop_reason"] = "normalized_gap_converged"
else:
level_state["stop_reason"] = (
"max_iterations_after_accept" if last.get("accepted") else "max_iterations"
)
level_state["completed"] = True
self._save()
return matrix, traces, exhausted_coordinate
def drive(self) -> Path:
if self.state.get("status") == "complete" and (self.run_dir / "S_final").is_dir():
return self.run_dir / "S_final"
_log(f"Deep Loop start/resume: {self.run_dir}")
input_traces = [
_trace_from_dict(item) for item in read_jsonl(self.run_dir / "input-traces.jsonl")
]
while True:
section_units = parse_sections(self._current_text())
section_matrix, traces, exhausted = self._run_level(
"section", section_units, seed_traces=input_traces
)
if exhausted is None:
break
target_score = section_matrix.columns[exhausted.dimension][exhausted.unit_id].score
current_sections = parse_sections(self._current_text())
section = next((item for item in current_sections if item.unit_id == exhausted.unit_id), None)
paragraphs = parse_paragraphs(section) if section is not None else []
section_state = self.state["levels"]["section"]
if (
target_score > 3.5
or len(paragraphs) < 2
or section_state.get("paragraph_returned")
):
self._block_dimension(section_state, exhausted.dimension)
section_state["completed"] = False
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
section_state.pop(key, None)
self._save()
continue
_log(f"Descending into paragraphs of {exhausted.unit_id}")
self.state["levels"].setdefault(
"paragraph", {"iterations": 0, "completed": False}
)["active_dimension"] = exhausted.dimension
self._save()
_, paragraph_traces, _ = self._run_level(
"paragraph", paragraphs, seed_traces=traces,
section_depth=section.heading_depth if section else None,
)
atomic_write_jsonl(
self.run_dir / "levels" / "section" / "current-traces.jsonl",
[asdict(trace) for trace in paragraph_traces],
)
section_state["completed"] = False
section_state["paragraph_returned"] = True
self._block_dimension(section_state, exhausted.dimension)
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
section_state.pop(key, None)
self._refresh_matrix(
"section", parse_sections(self._current_text()), paragraph_traces
)
return self._complete()
def _complete(self) -> Path:
target = self.run_dir / "S_final"
if target.exists():
shutil.rmtree(target)
shutil.copytree(self.current, target)
final_skill_hash = sha256_file(target / "SKILL.md")
final_rollouts = self.state.get("current_rollouts")
final_rollout_hash = self.state.get("current_rollout_skill_sha256")
if (
not isinstance(final_rollouts, str)
or not Path(final_rollouts).is_dir()
or final_rollout_hash != final_skill_hash
):
final_rollouts = None
report = {
"status": "complete",
"input_package_hash": package_hash(self.run_dir / "S_fast"),
"final_package_hash": package_hash(target),
"final_rollouts": final_rollouts,
"final_rollout_skill_sha256": final_skill_hash if final_rollouts else None,
"task": self.state["task"]["name"],
"agent": self.state["task"]["agent"],
"target_model": self.state["task"]["model"],
"model": self.model,
"rollout_backend": "skillsbench_development",
"production_rollout_backend": "blank_container_required",
"verifier_signal_used": False,
"levels": self.state["levels"],
"decisions": {
name: self._decisions(name)
for name in ("section", "paragraph")
if name in self.state["levels"]
},
}
atomic_write_json(self.run_dir / "report.json", report)
self.state["status"] = "complete"
self._save()
shutil.rmtree(self.temp, ignore_errors=True)
_log(f"Deep Loop complete: {target}")
return target