Initial commit
This commit is contained in:
@@ -0,0 +1,53 @@
|
||||
"""Deep 编译流水线的唯一命令行入口。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.provider_router import parse_model_reference
|
||||
|
||||
from .pipeline import DeepLoop
|
||||
|
||||
|
||||
def _provider_model(value: str) -> str:
|
||||
try:
|
||||
return parse_model_reference(value).value
|
||||
except ValueError as exc:
|
||||
raise argparse.ArgumentTypeError(str(exc)) from exc
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(prog="python -m scripts.dynamic_compile.deep")
|
||||
commands = parser.add_subparsers(dest="command", required=True)
|
||||
run = commands.add_parser("run", help="run Deep Loop from a skill and its BenchFlow traces")
|
||||
run.add_argument("--skill", type=Path, required=True)
|
||||
run.add_argument("--traces", type=Path, required=True)
|
||||
run.add_argument("--output", type=Path)
|
||||
run.add_argument(
|
||||
"--model",
|
||||
required=True,
|
||||
type=_provider_model,
|
||||
help="all external model calls use this provider/model",
|
||||
)
|
||||
resume = commands.add_parser("resume", help="resume a Deep Loop run")
|
||||
resume.add_argument("--run", type=Path, required=True)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _parser().parse_args(argv)
|
||||
try:
|
||||
loop = (
|
||||
DeepLoop.create(args.skill, args.traces, args.output, model=args.model)
|
||||
if args.command == "run"
|
||||
else DeepLoop(args.run)
|
||||
)
|
||||
print(loop.drive())
|
||||
return 0
|
||||
except (OSError, ValueError, RuntimeError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,216 @@
|
||||
"""BenchFlow 输入与运行时轨迹适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.traces.benchflow import trajectory_for_test
|
||||
from scripts.dynamic_compile.fast.scoring.state import acp_events_to_state, read_acp_events
|
||||
|
||||
|
||||
@dataclass
|
||||
class BenchFlowInput:
|
||||
task_name: str
|
||||
task_dir: Path
|
||||
agent: str
|
||||
model: str
|
||||
prompt: str
|
||||
traces: list[RolloutTrace]
|
||||
|
||||
|
||||
def _project_root() -> Path:
|
||||
return Path(__file__).resolve().parents[4]
|
||||
|
||||
|
||||
def _run_config(test_dir: Path) -> dict[str, Any]:
|
||||
paths = sorted(test_dir.rglob("config.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow run config under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError(f"invalid BenchFlow run config: {paths[0]}")
|
||||
return value
|
||||
|
||||
|
||||
def _prompt(test_dir: Path) -> str:
|
||||
paths = sorted(test_dir.rglob("prompts.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow prompts.json under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, list) or not value or not isinstance(value[0], str):
|
||||
raise ValueError(f"invalid BenchFlow prompts: {paths[0]}")
|
||||
return value[0].strip()
|
||||
|
||||
|
||||
def load_runtime_trace(test_dir: Path, task_name: str, compile_type: str) -> RolloutTrace:
|
||||
"""Load agent runtime evidence without opening verifier/result artifacts."""
|
||||
trajectory = trajectory_for_test(test_dir)
|
||||
if trajectory is None:
|
||||
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
|
||||
events = read_acp_events(trajectory)
|
||||
state = acp_events_to_state(events)
|
||||
skill_invoked = any(
|
||||
event.get("type") == "tool_call"
|
||||
and event.get("status") == "completed"
|
||||
and any(
|
||||
str(event.get(field, "")).strip().lower() == "skill"
|
||||
for field in ("title", "kind")
|
||||
)
|
||||
for event in events
|
||||
)
|
||||
timed_out = any(event.get("type") == "agent_timeout" for event in events)
|
||||
return RolloutTrace(
|
||||
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
|
||||
task_name=task_name,
|
||||
compile_type=compile_type,
|
||||
test_name=test_dir.name,
|
||||
state=state,
|
||||
skill_invoked=skill_invoked,
|
||||
timed_out=timed_out,
|
||||
metadata={
|
||||
"termination": "timeout" if timed_out else "completed",
|
||||
"tool_calls": sum(event.get("type") == "tool_call" for event in events),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def load_benchflow_input(source: Path) -> BenchFlowInput:
|
||||
source = source.resolve()
|
||||
tests = sorted(path for path in source.glob("test-*") if path.is_dir())[-5:]
|
||||
if not tests:
|
||||
raise ValueError(f"no test-* BenchFlow traces under {source}")
|
||||
task_name = source.parent.name
|
||||
traces = [load_runtime_trace(test, task_name, "custom") for test in tests]
|
||||
configs = [_run_config(test) for test in tests]
|
||||
agents = {str(item.get("agent", "")).strip() for item in configs}
|
||||
models = {str(item.get("model", "")).strip() for item in configs}
|
||||
if "" in agents or len(agents) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one agent: {sorted(agents)}")
|
||||
if "" in models or len(models) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one model: {sorted(models)}")
|
||||
prompts = {_prompt(test) for test in tests}
|
||||
if len(prompts) != 1:
|
||||
raise ValueError(f"BenchFlow traces contain {len(prompts)} different task prompts")
|
||||
task_dir = _project_root() / "data" / "skills-bench" / "tasks" / task_name
|
||||
if not (task_dir / "task.md").is_file():
|
||||
raise ValueError(f"cannot resolve SkillsBench task directory: {task_dir}")
|
||||
return BenchFlowInput(
|
||||
task_name, task_dir, next(iter(agents)), next(iter(models)),
|
||||
next(iter(prompts)), traces,
|
||||
)
|
||||
|
||||
|
||||
def _slug(value: str) -> str:
|
||||
return re.sub(r"^-+|-+$", "", re.sub(r"[^a-z0-9]+", "-", value.lower()))
|
||||
|
||||
|
||||
class SkillsBenchDevelopmentRolloutRunner:
|
||||
"""Development-only runner; verifier output is never projected into traces."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
context: BenchFlowInput,
|
||||
work_root: Path,
|
||||
max_parallel: int = 3,
|
||||
archive_root: Path | None = None,
|
||||
):
|
||||
self.context = context
|
||||
self.work_root = work_root
|
||||
self.max_parallel = max_parallel
|
||||
self.archive_root = archive_root
|
||||
|
||||
def _variant_dir(self, jobs_root: Path) -> Path:
|
||||
return (
|
||||
jobs_root
|
||||
/ _slug(f"{self.context.agent}-{self.context.model}")
|
||||
/ self.context.task_name
|
||||
/ "custom_skill"
|
||||
)
|
||||
|
||||
def _completed_traces(self, variant: Path, batch_id: str) -> list[RolloutTrace]:
|
||||
completed: list[RolloutTrace] = []
|
||||
for test in sorted(path for path in variant.glob("test-*") if path.is_dir()):
|
||||
try:
|
||||
requirement = json.loads(
|
||||
(test / "required-skill.json").read_text(encoding="utf-8")
|
||||
)
|
||||
if requirement.get("invoked") is not True:
|
||||
continue
|
||||
trace = load_runtime_trace(test, self.context.task_name, batch_id)
|
||||
except (AttributeError, json.JSONDecodeError, OSError, ValueError):
|
||||
continue
|
||||
completed.append(trace)
|
||||
return completed
|
||||
|
||||
def artifacts_dir(self, batch_id: str) -> Path | None:
|
||||
if self.archive_root is None:
|
||||
return None
|
||||
return self.archive_root / batch_id / "custom_skill"
|
||||
|
||||
def run_batch(
|
||||
self,
|
||||
skill_package: Path,
|
||||
_prompt: str,
|
||||
batch_id: str,
|
||||
_task_name: str,
|
||||
count: int,
|
||||
progress: Any | None = None,
|
||||
) -> list[RolloutTrace]:
|
||||
jobs_root = self.work_root / batch_id
|
||||
log_path = jobs_root / "runner.log"
|
||||
variant = self._variant_dir(jobs_root)
|
||||
archive = self.artifacts_dir(batch_id)
|
||||
if archive is not None and archive.is_dir():
|
||||
archived_traces = self._completed_traces(archive, batch_id)
|
||||
if len(archived_traces) >= count:
|
||||
traces = archived_traces
|
||||
missing = 0
|
||||
else:
|
||||
variant.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(archive, variant, dirs_exist_ok=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
else:
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
if missing:
|
||||
jobs_root.mkdir(parents=True, exist_ok=True)
|
||||
command = [
|
||||
"bash", str(_project_root() / "scripts" / "evaluate" / "run-raw-task.sh"),
|
||||
"--harness", self.context.agent,
|
||||
"--model", self.context.model,
|
||||
"--task", str(self.context.task_dir),
|
||||
"--skill-source", str(skill_package.resolve()),
|
||||
"--require-skill",
|
||||
"--repeat", str(missing),
|
||||
"--max-parallel", str(self.max_parallel),
|
||||
"--output", str(variant),
|
||||
]
|
||||
with log_path.open("a", encoding="utf-8") as handle:
|
||||
subprocess.run(command, stdout=handle, stderr=subprocess.STDOUT, text=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
if archive is not None and variant.is_dir():
|
||||
archive.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(variant, archive, dirs_exist_ok=True)
|
||||
runner_log = jobs_root / "runner.log"
|
||||
if runner_log.is_file():
|
||||
shutil.copy2(runner_log, archive.parent / "runner.log")
|
||||
traces = self._completed_traces(archive, batch_id)
|
||||
if len(traces) < count:
|
||||
raise RuntimeError(
|
||||
f"BenchFlow produced {len(traces)}/{count} candidate traces; see {log_path}"
|
||||
)
|
||||
traces = traces[:count]
|
||||
for index, trace in enumerate(traces, 1):
|
||||
trace.trace_id = f"{batch_id}-{index:03d}"
|
||||
trace.test_name = trace.trace_id
|
||||
if progress is not None:
|
||||
progress(index, len(traces), trace.trace_id)
|
||||
return traces
|
||||
@@ -0,0 +1,197 @@
|
||||
"""语义模型评分与局部编辑适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
|
||||
|
||||
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
|
||||
|
||||
|
||||
RUBRICS = {
|
||||
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
|
||||
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
|
||||
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
|
||||
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
|
||||
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
|
||||
}
|
||||
|
||||
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
|
||||
|
||||
|
||||
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
|
||||
return [
|
||||
value for value in values
|
||||
if isinstance(value, str)
|
||||
and _EVIDENCE.fullmatch(value)
|
||||
and value.startswith(prefixes)
|
||||
]
|
||||
|
||||
|
||||
class DeepAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
@staticmethod
|
||||
def _trace_payload(traces: list[RolloutTrace]) -> str:
|
||||
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
|
||||
|
||||
def score_column(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, CellScore]:
|
||||
unit_payload = [
|
||||
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
|
||||
for unit in units
|
||||
]
|
||||
result = self.client.json(
|
||||
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
|
||||
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Current units:
|
||||
{json.dumps(unit_payload, ensure_ascii=False)}
|
||||
Agent traces:
|
||||
{self._trace_payload(traces)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
|
||||
raise ValueError(f"judge returned an invalid {dimension} column")
|
||||
expected = {unit.unit_id for unit in units}
|
||||
column: dict[str, CellScore] = {}
|
||||
for item in result["scores"]:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("judge score entries must be objects")
|
||||
unit_id = str(item.get("unit_id", ""))
|
||||
score = float(item.get("score"))
|
||||
evidence = item.get("evidence", [])
|
||||
if unit_id not in expected or unit_id in column:
|
||||
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
|
||||
evidence = valid_evidence(evidence, traces)
|
||||
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
|
||||
if set(column) != expected:
|
||||
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
|
||||
return column
|
||||
|
||||
def compare_cell(
|
||||
self,
|
||||
task: str,
|
||||
incumbent_unit: SkillUnit,
|
||||
candidate_unit: SkillUnit,
|
||||
incumbent_traces: list[RolloutTrace],
|
||||
candidate_traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Compare one incumbent and candidate skill unit. Return JSON only.",
|
||||
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Incumbent unit:
|
||||
{incumbent_unit.text}
|
||||
Candidate unit:
|
||||
{candidate_unit.text}
|
||||
Incumbent runtime facts:
|
||||
{self._trace_payload(incumbent_traces)}
|
||||
Candidate runtime facts:
|
||||
{self._trace_payload(candidate_traces)}""",
|
||||
)
|
||||
if not isinstance(result.get("task_relevant"), bool):
|
||||
raise ValueError("judge returned invalid task relevance")
|
||||
incumbent_score = float(result.get("incumbent_score"))
|
||||
candidate_score = float(result.get("candidate_score"))
|
||||
for score in (incumbent_score, candidate_score):
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid paired score: {score}")
|
||||
regressions = result.get("runtime_regressions")
|
||||
allowed = {
|
||||
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
|
||||
}
|
||||
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
|
||||
raise ValueError("judge returned invalid runtime regressions")
|
||||
return {
|
||||
"task_relevant": result["task_relevant"],
|
||||
"incumbent_score": incumbent_score,
|
||||
"candidate_score": candidate_score,
|
||||
"candidate_evidence": valid_evidence(
|
||||
result.get("candidate_evidence"), candidate_traces
|
||||
),
|
||||
"runtime_regressions": list(dict.fromkeys(regressions)),
|
||||
"reason": str(result.get("reason", "")),
|
||||
}
|
||||
|
||||
def score_matrix(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
level: str,
|
||||
existing_columns: dict[str, dict[str, CellScore]] | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> ScoreMatrix:
|
||||
columns = dict(existing_columns or {})
|
||||
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {
|
||||
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
|
||||
for dimension in missing
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
dimension = futures[future]
|
||||
columns[dimension] = future.result()
|
||||
if result_callback is not None:
|
||||
result_callback(dimension, columns[dimension])
|
||||
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
|
||||
|
||||
def generate_edit(
|
||||
self,
|
||||
coordinate: Coordinate,
|
||||
unit: SkillUnit,
|
||||
cell: CellScore,
|
||||
rejected: list[dict[str, Any]],
|
||||
task_prompt: str,
|
||||
) -> LocalEdit:
|
||||
result = self.client.json(
|
||||
"Generate one bounded local edit for an agent skill. Return JSON only.",
|
||||
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
|
||||
|
||||
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
|
||||
|
||||
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
|
||||
|
||||
Target dimension rubric: {RUBRICS[coordinate.dimension]}
|
||||
Task prompt:
|
||||
{task_prompt}
|
||||
Target unit:
|
||||
{unit.text}
|
||||
Score: {cell.score}
|
||||
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
|
||||
Reason: {cell.reason}
|
||||
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
|
||||
)
|
||||
edit = LocalEdit.from_dict(result)
|
||||
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
|
||||
raise ValueError("edit generator changed the target coordinate")
|
||||
return edit
|
||||
@@ -0,0 +1,195 @@
|
||||
"""Markdown 单元解析与编辑边界校验。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
import re
|
||||
|
||||
from .models import SkillUnit
|
||||
|
||||
|
||||
_HEADING = re.compile(r"^(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*(?:\n|$)")
|
||||
_FENCE = re.compile(r"^[ \t]*(`{3,}|~{3,})")
|
||||
|
||||
|
||||
def _line_offsets(text: str) -> list[tuple[int, int, str]]:
|
||||
rows: list[tuple[int, int, str]] = []
|
||||
offset = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
rows.append((offset, offset + len(line), line))
|
||||
offset += len(line)
|
||||
return rows
|
||||
|
||||
|
||||
def _headings(text: str) -> list[tuple[int, int, int, str]]:
|
||||
found = []
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for start, end, line in _line_offsets(text):
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if fence_char:
|
||||
continue
|
||||
match = _HEADING.match(line)
|
||||
if match:
|
||||
found.append((start, end, len(match.group(1)), match.group(2).strip()))
|
||||
return found
|
||||
|
||||
|
||||
def _frontmatter_end(text: str) -> int:
|
||||
if not text.startswith("---"):
|
||||
return 0
|
||||
lines = text.splitlines(keepends=True)
|
||||
offset = len(lines[0]) if lines else 0
|
||||
for line in lines[1:]:
|
||||
offset += len(line)
|
||||
if line.strip() == "---":
|
||||
return offset
|
||||
return 0
|
||||
|
||||
|
||||
def parse_sections(text: str) -> list[SkillUnit]:
|
||||
headings = _headings(text)
|
||||
body_start = _frontmatter_end(text)
|
||||
body_headings = [item for item in headings if item[0] >= body_start]
|
||||
title = body_headings[0] if body_headings else None
|
||||
after_title = title[1] if title else body_start
|
||||
candidates = [item for item in body_headings[1:] if not title or item[2] > title[2]]
|
||||
if not candidates:
|
||||
body = text[body_start:]
|
||||
return [SkillUnit("S001", "section", None, title[3] if title else "Document", body, 0, body_start, len(text), title[2] if title else None)]
|
||||
section_depth = min(item[2] for item in candidates)
|
||||
peers = [item for item in candidates if item[2] == section_depth]
|
||||
spans: list[tuple[int, int, str, int | None]] = []
|
||||
preamble = text[after_title:peers[0][0]]
|
||||
if preamble.strip():
|
||||
spans.append((after_title, peers[0][0], "Preamble", section_depth))
|
||||
for index, heading in enumerate(peers):
|
||||
end = peers[index + 1][0] if index + 1 < len(peers) else len(text)
|
||||
spans.append((heading[0], end, heading[3], heading[2]))
|
||||
return [
|
||||
SkillUnit(f"S{index + 1:03d}", "section", None, heading, text[start:end], index, start, end, depth)
|
||||
for index, (start, end, heading, depth) in enumerate(spans)
|
||||
]
|
||||
|
||||
|
||||
def preserve_unit_boundary(unit: SkillUnit, new_text: str) -> str:
|
||||
return new_text.rstrip() + unit.text[len(unit.text.rstrip()):]
|
||||
|
||||
|
||||
def replace_unit_text(document: str, unit: SkillUnit, new_text: str) -> str:
|
||||
replacement = preserve_unit_boundary(unit, new_text)
|
||||
return document[:unit.start] + replacement + document[unit.end:]
|
||||
|
||||
|
||||
def parse_paragraphs(section: SkillUnit) -> list[SkillUnit]:
|
||||
text = section.text
|
||||
base = section.start
|
||||
rows = _line_offsets(text)
|
||||
blocks: list[tuple[int, int]] = []
|
||||
start: int | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for row_start, row_end, line in rows:
|
||||
fence = _FENCE.match(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if start is None:
|
||||
start = row_start
|
||||
if not fence_char:
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_size:
|
||||
fence_char, fence_size = "", 0
|
||||
continue
|
||||
if not fence_char and not line.strip():
|
||||
if start is not None:
|
||||
blocks.append((start, row_start))
|
||||
start = None
|
||||
continue
|
||||
if start is None:
|
||||
start = row_start
|
||||
if start is not None:
|
||||
blocks.append((start, len(text)))
|
||||
merged: list[tuple[int, int]] = []
|
||||
index = 0
|
||||
while index < len(blocks):
|
||||
start, end = blocks[index]
|
||||
block = text[start:end]
|
||||
if index + 1 < len(blocks) and _HEADING.fullmatch(block.strip() + "\n"):
|
||||
merged.append((start, blocks[index + 1][1]))
|
||||
index += 2
|
||||
else:
|
||||
merged.append((start, end))
|
||||
index += 1
|
||||
blocks = merged
|
||||
units = []
|
||||
for index, (start, end) in enumerate(blocks):
|
||||
block = text[start:end]
|
||||
heading_match = next((item for item in _headings(block)), None)
|
||||
units.append(SkillUnit(
|
||||
f"{section.unit_id}.P{index + 1:03d}",
|
||||
"paragraph",
|
||||
section.unit_id,
|
||||
heading_match[3] if heading_match else "",
|
||||
block,
|
||||
index,
|
||||
base + start,
|
||||
base + end,
|
||||
heading_match[2] if heading_match else None,
|
||||
))
|
||||
return units
|
||||
|
||||
|
||||
def fenced_blocks(text: str) -> Counter[str]:
|
||||
blocks: list[str] = []
|
||||
current: list[str] | None = None
|
||||
fence_char = ""
|
||||
fence_size = 0
|
||||
for line in text.splitlines(keepends=True):
|
||||
fence = _FENCE.match(line)
|
||||
if current is None:
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
fence_char, fence_size = marker[0], len(marker)
|
||||
current = [line]
|
||||
continue
|
||||
current.append(line)
|
||||
if fence:
|
||||
marker = fence.group(1)
|
||||
if marker[0] == fence_char and len(marker) >= fence_size:
|
||||
blocks.append("".join(current))
|
||||
current = None
|
||||
fence_char, fence_size = "", 0
|
||||
return Counter(blocks)
|
||||
|
||||
|
||||
def validate_edit(unit: SkillUnit, new_text: str, section_depth: int | None = None) -> None:
|
||||
if not new_text.strip() or new_text == unit.text:
|
||||
raise ValueError("local edit must produce non-empty changed text")
|
||||
if unit.level == "section":
|
||||
old_headings = _headings(unit.text)
|
||||
new_headings = _headings(new_text)
|
||||
depth = unit.heading_depth
|
||||
old_peers = [(item[2], item[3]) for item in old_headings if item[2] == depth]
|
||||
new_peers = [(item[2], item[3]) for item in new_headings if item[2] == depth]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("section edit must preserve its peer heading")
|
||||
if fenced_blocks(unit.text) != fenced_blocks(new_text):
|
||||
raise ValueError("section edit must preserve fenced code contents")
|
||||
elif section_depth is not None:
|
||||
old_peers = [
|
||||
(item[2], item[3]) for item in _headings(unit.text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
new_peers = [
|
||||
(item[2], item[3]) for item in _headings(new_text)
|
||||
if item[2] <= section_depth
|
||||
]
|
||||
if old_peers != new_peers:
|
||||
raise ValueError("paragraph edit must not add or change a section heading")
|
||||
@@ -0,0 +1,134 @@
|
||||
"""领域模型。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
DIMENSIONS = (
|
||||
"Clarity",
|
||||
"Structure",
|
||||
"Executability",
|
||||
"Completeness",
|
||||
"Constraint Salience",
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkillUnit:
|
||||
unit_id: str
|
||||
level: str
|
||||
parent_id: str | None
|
||||
heading: str
|
||||
text: str
|
||||
order: int
|
||||
start: int
|
||||
end: int
|
||||
heading_depth: int | None = None
|
||||
|
||||
@dataclass
|
||||
class CellScore:
|
||||
score: float
|
||||
evidence: list[str]
|
||||
reason: str
|
||||
|
||||
@dataclass
|
||||
class Coordinate:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
normalized_gap: float
|
||||
|
||||
@dataclass
|
||||
class LocalEdit:
|
||||
unit_id: str
|
||||
dimension: str
|
||||
new_text: str
|
||||
edit_summary: str
|
||||
reason: str
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "LocalEdit":
|
||||
required = {"unit_id", "dimension", "new_text", "edit_summary", "reason"}
|
||||
missing = sorted(required - value.keys())
|
||||
if missing:
|
||||
raise ValueError(f"local edit missing fields: {', '.join(missing)}")
|
||||
if not all(isinstance(value[key], str) for key in required):
|
||||
raise ValueError("local edit fields must be strings")
|
||||
return cls(**{key: value[key] for key in cls.__dataclass_fields__})
|
||||
|
||||
|
||||
@dataclass
|
||||
class ScoreMatrix:
|
||||
level: str
|
||||
units: list[SkillUnit]
|
||||
columns: dict[str, dict[str, CellScore]]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"level": self.level,
|
||||
"units": [asdict(unit) for unit in self.units],
|
||||
"columns": {
|
||||
dimension: {unit_id: asdict(cell) for unit_id, cell in column.items()}
|
||||
for dimension, column in self.columns.items()
|
||||
},
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: dict[str, Any]) -> "ScoreMatrix":
|
||||
return cls(
|
||||
level=str(value["level"]),
|
||||
units=[SkillUnit(**item) for item in value["units"]],
|
||||
columns={
|
||||
dimension: {
|
||||
unit_id: CellScore(float(cell["score"]), list(cell["evidence"]), str(cell["reason"]))
|
||||
for unit_id, cell in column.items()
|
||||
}
|
||||
for dimension, column in value["columns"].items()
|
||||
},
|
||||
)
|
||||
|
||||
def unit(self, unit_id: str) -> SkillUnit:
|
||||
return next(unit for unit in self.units if unit.unit_id == unit_id)
|
||||
|
||||
def normalized_gaps(self) -> dict[str, float]:
|
||||
gaps: dict[str, float] = {}
|
||||
for dimension in DIMENSIONS:
|
||||
values = [self.columns[dimension][unit.unit_id].score for unit in self.units]
|
||||
gaps[dimension] = (max(values) - min(values)) / 4.0 if values else 0.0
|
||||
return gaps
|
||||
|
||||
def select_coordinate(
|
||||
self,
|
||||
threshold: float,
|
||||
dimension: str | None = None,
|
||||
excluded: set[tuple[str, str]] | None = None,
|
||||
) -> Coordinate | None:
|
||||
gaps = self.normalized_gaps()
|
||||
excluded = excluded or set()
|
||||
|
||||
def weak_units(item: str) -> list[SkillUnit]:
|
||||
column = self.columns[item]
|
||||
maximum = max((cell.score for cell in column.values()), default=0.0)
|
||||
return [
|
||||
unit for unit in self.units
|
||||
if (unit.unit_id, item) not in excluded
|
||||
and (maximum - column[unit.unit_id].score) / 4.0 > threshold
|
||||
]
|
||||
|
||||
available = [
|
||||
item for item in ([dimension] if dimension else DIMENSIONS)
|
||||
if item is not None and gaps[item] > threshold and weak_units(item)
|
||||
]
|
||||
if not available:
|
||||
return None
|
||||
dimension = max(available, key=lambda item: gaps[item])
|
||||
target = min(
|
||||
weak_units(dimension),
|
||||
key=lambda unit: (self.columns[dimension][unit.unit_id].score, unit.order),
|
||||
)
|
||||
return Coordinate(
|
||||
target.unit_id,
|
||||
dimension,
|
||||
gaps[dimension],
|
||||
)
|
||||
@@ -0,0 +1,826 @@
|
||||
"""Deep 编译流水线编排。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import uuid
|
||||
from dataclasses import asdict, replace
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.storage import (
|
||||
atomic_write_json,
|
||||
atomic_write_jsonl,
|
||||
atomic_write_text,
|
||||
load_json,
|
||||
package_hash,
|
||||
read_jsonl,
|
||||
sha256_file,
|
||||
sha256_text,
|
||||
)
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
|
||||
from .adapters.benchflow import (
|
||||
BenchFlowInput,
|
||||
SkillsBenchDevelopmentRolloutRunner,
|
||||
load_benchflow_input,
|
||||
)
|
||||
from .adapters.semantic import DeepAnalyzer, valid_evidence
|
||||
from .core.models import (
|
||||
CellScore,
|
||||
Coordinate,
|
||||
DIMENSIONS,
|
||||
LocalEdit,
|
||||
ScoreMatrix,
|
||||
SkillUnit,
|
||||
)
|
||||
from .core.markdown import (
|
||||
parse_paragraphs,
|
||||
parse_sections,
|
||||
preserve_unit_boundary,
|
||||
replace_unit_text,
|
||||
validate_edit,
|
||||
)
|
||||
|
||||
|
||||
ROLLOUTS = 3
|
||||
MAX_SECTION_ITERATIONS = 6
|
||||
MAX_PARAGRAPH_ITERATIONS = 3
|
||||
MAX_EDIT_GENERATION_ATTEMPTS = 3
|
||||
GAP_THRESHOLD = 0.375
|
||||
REJECTION_LIMIT = 2
|
||||
MAX_PARALLEL = 3
|
||||
DEFAULT_MODEL = "ali/deepseek-v4-pro-0813"
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"[deep] {message}", file=sys.stderr, flush=True)
|
||||
|
||||
|
||||
def _trace_from_dict(value: dict[str, Any]) -> RolloutTrace:
|
||||
return RolloutTrace(**value)
|
||||
|
||||
|
||||
def _column_to_dict(column: dict[str, CellScore]) -> dict[str, Any]:
|
||||
return {unit_id: asdict(cell) for unit_id, cell in column.items()}
|
||||
|
||||
|
||||
def _column_from_dict(
|
||||
value: dict[str, Any], traces: list[RolloutTrace]
|
||||
) -> dict[str, CellScore]:
|
||||
return {
|
||||
unit_id: CellScore(
|
||||
float(cell["score"]), valid_evidence(cell.get("evidence"), traces), str(cell["reason"])
|
||||
)
|
||||
for unit_id, cell in value.items()
|
||||
}
|
||||
|
||||
|
||||
class DeepLoop:
|
||||
def __init__(
|
||||
self,
|
||||
run_dir: Path,
|
||||
analyzer: DeepAnalyzer | None = None,
|
||||
runner: Any | None = None,
|
||||
):
|
||||
self.run_dir = run_dir.resolve()
|
||||
self.state_path = self.run_dir / "run.json"
|
||||
state = load_json(self.state_path)
|
||||
if not isinstance(state, dict):
|
||||
raise ValueError(f"invalid or missing run state: {self.state_path}")
|
||||
self.state = state
|
||||
self.temp = self.run_dir / ".tmp"
|
||||
self.current = self.temp / "current"
|
||||
self.model = state.get("model", state.get("semantic_model", DEFAULT_MODEL))
|
||||
self.analyzer = analyzer or DeepAnalyzer(
|
||||
SemanticClient(self.model), MAX_PARALLEL,
|
||||
)
|
||||
context = BenchFlowInput(
|
||||
state["task"]["name"],
|
||||
Path(state["task"]["directory"]),
|
||||
state["task"]["agent"],
|
||||
state["task"]["model"],
|
||||
state["task"]["prompt"],
|
||||
[],
|
||||
)
|
||||
self.runner = runner or SkillsBenchDevelopmentRolloutRunner(
|
||||
context,
|
||||
Path(tempfile.gettempdir()) / "skill-compiler-deep" / state["run_id"],
|
||||
MAX_PARALLEL,
|
||||
archive_root=self.run_dir / "rollouts",
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
skill: Path,
|
||||
traces: Path,
|
||||
output: Path | None = None,
|
||||
*,
|
||||
model: str = DEFAULT_MODEL,
|
||||
analyzer: DeepAnalyzer | None = None,
|
||||
runner: Any | None = None,
|
||||
) -> "DeepLoop":
|
||||
skill = skill.resolve()
|
||||
if not (skill / "SKILL.md").is_file():
|
||||
raise ValueError("--skill must be a skill package containing SKILL.md")
|
||||
context = load_benchflow_input(traces)
|
||||
target = (output or skill.parent / f"{skill.name}-deep").resolve()
|
||||
if target.exists() and any(target.iterdir()):
|
||||
raise ValueError(f"deep run directory is not empty: {target}")
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(skill, target / "S_fast")
|
||||
(target / ".tmp").mkdir()
|
||||
(target / "levels").mkdir()
|
||||
shutil.copytree(target / "S_fast", target / ".tmp" / "current")
|
||||
atomic_write_jsonl(target / "input-traces.jsonl", [asdict(trace) for trace in context.traces])
|
||||
state = {
|
||||
"run_id": uuid.uuid4().hex[:12],
|
||||
"status": "created",
|
||||
"model": model,
|
||||
"skill_name": skill.name,
|
||||
"task": {
|
||||
"name": context.task_name,
|
||||
"directory": str(context.task_dir),
|
||||
"agent": context.agent,
|
||||
"model": context.model,
|
||||
"prompt": context.prompt,
|
||||
},
|
||||
"current_rollouts": str(traces.resolve()),
|
||||
"current_rollout_skill_sha256": sha256_file(skill / "SKILL.md"),
|
||||
"levels": {},
|
||||
}
|
||||
atomic_write_json(target / "run.json", state)
|
||||
return cls(target, analyzer=analyzer, runner=runner)
|
||||
|
||||
def _save(self) -> None:
|
||||
atomic_write_json(self.state_path, self.state)
|
||||
|
||||
def _prompt(self) -> str:
|
||||
return str(self.state["task"]["prompt"])
|
||||
|
||||
def _current_text(self) -> str:
|
||||
return (self.current / "SKILL.md").read_text(encoding="utf-8")
|
||||
|
||||
def _load_or_rollout(
|
||||
self,
|
||||
package: Path,
|
||||
trace_path: Path,
|
||||
batch_id: str,
|
||||
seed: list[RolloutTrace] | None = None,
|
||||
) -> list[RolloutTrace]:
|
||||
if trace_path.is_file():
|
||||
return [_trace_from_dict(item) for item in read_jsonl(trace_path)]
|
||||
if seed is not None:
|
||||
traces = seed
|
||||
else:
|
||||
_log(f"Starting {ROLLOUTS} rollouts for {batch_id}")
|
||||
traces = self.runner.run_batch(
|
||||
package,
|
||||
self._prompt(),
|
||||
batch_id,
|
||||
str(self.state["task"]["name"]),
|
||||
ROLLOUTS,
|
||||
progress=lambda done, total, trace_id: _log(
|
||||
f"Rollout {done}/{total} complete: {trace_id}"
|
||||
),
|
||||
)
|
||||
atomic_write_jsonl(trace_path, [asdict(trace) for trace in traces])
|
||||
return traces
|
||||
|
||||
def _load_or_matrix(
|
||||
self,
|
||||
path: Path,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
) -> ScoreMatrix:
|
||||
value = load_json(path)
|
||||
if isinstance(value, dict):
|
||||
return ScoreMatrix.from_dict(value)
|
||||
columns_dir = path.parent / "matrix-columns"
|
||||
cached_columns: dict[str, dict[str, CellScore]] = {}
|
||||
for dimension in DIMENSIONS:
|
||||
cached = load_json(columns_dir / f"{dimension.lower().replace(' ', '-')}.json")
|
||||
if isinstance(cached, dict):
|
||||
cached_columns[dimension] = _column_from_dict(cached, traces)
|
||||
|
||||
def save_column(dimension: str, column: dict[str, CellScore]) -> None:
|
||||
atomic_write_json(
|
||||
columns_dir / f"{dimension.lower().replace(' ', '-')}.json",
|
||||
_column_to_dict(column),
|
||||
)
|
||||
|
||||
_log(f"Scoring full {level} matrix with {self.model}")
|
||||
matrix = self.analyzer.score_matrix(
|
||||
self._current_text(), self._prompt(), units, traces, level,
|
||||
existing_columns=cached_columns,
|
||||
result_callback=save_column,
|
||||
)
|
||||
atomic_write_json(path, matrix.to_dict())
|
||||
return matrix
|
||||
|
||||
def _refresh_matrix(
|
||||
self,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
) -> ScoreMatrix:
|
||||
level_state = self.state["levels"][level]
|
||||
number = int(level_state.get("refreshes", 0)) + 1
|
||||
path = (
|
||||
self.run_dir
|
||||
/ "levels"
|
||||
/ level
|
||||
/ "refreshes"
|
||||
/ f"refresh-{number:02d}"
|
||||
/ "matrix.json"
|
||||
)
|
||||
_log(f"Refreshing full {level} matrix")
|
||||
matrix = self._load_or_matrix(path, level, units, traces)
|
||||
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
|
||||
level_state["refreshes"] = number
|
||||
level_state.pop("active_dimension", None)
|
||||
self._save()
|
||||
return matrix
|
||||
|
||||
def _decisions(self, level: str | None = None) -> list[dict[str, Any]]:
|
||||
roots = (
|
||||
[self.run_dir / "levels" / level]
|
||||
if level else list((self.run_dir / "levels").glob("*"))
|
||||
)
|
||||
decisions = []
|
||||
for root in roots:
|
||||
for path in sorted((root / "iterations").glob("iteration-*/decision.json")):
|
||||
value = load_json(path)
|
||||
if isinstance(value, dict):
|
||||
decisions.append(value)
|
||||
return decisions
|
||||
|
||||
def _rejected(self, coordinate: Coordinate) -> list[dict[str, Any]]:
|
||||
return [
|
||||
decision["rejected_edit"]
|
||||
for decision in self._decisions()
|
||||
if not decision["accepted"]
|
||||
and decision["rejected_edit"]["unit_id"] == coordinate.unit_id
|
||||
and decision["rejected_edit"]["dimension"] == coordinate.dimension
|
||||
]
|
||||
|
||||
def _exhausted(self, level: str) -> set[tuple[str, str]]:
|
||||
counts: dict[tuple[str, str], int] = {}
|
||||
for decision in self._decisions(level):
|
||||
if decision["accepted"]:
|
||||
continue
|
||||
rejected = decision["rejected_edit"]
|
||||
key = (rejected["unit_id"], rejected["dimension"])
|
||||
counts[key] = counts.get(key, 0) + 1
|
||||
return {key for key, count in counts.items() if count >= REJECTION_LIMIT}
|
||||
|
||||
@staticmethod
|
||||
def _block_dimension(level_state: dict[str, Any], dimension: str) -> None:
|
||||
blocked = set(level_state.get("blocked_dimensions", []))
|
||||
blocked.add(dimension)
|
||||
level_state["blocked_dimensions"] = sorted(blocked)
|
||||
|
||||
@staticmethod
|
||||
def _candidate_units(
|
||||
units: list[SkillUnit], target: SkillUnit, new_text: str
|
||||
) -> list[SkillUnit]:
|
||||
delta = len(new_text) - len(target.text)
|
||||
updated = []
|
||||
for unit in units:
|
||||
value = replace(unit)
|
||||
if unit.unit_id == target.unit_id:
|
||||
value.text = new_text
|
||||
value.end = value.start + len(new_text)
|
||||
elif unit.start >= target.end:
|
||||
value.start += delta
|
||||
value.end += delta
|
||||
updated.append(value)
|
||||
return updated
|
||||
|
||||
def _apply_edit(
|
||||
self,
|
||||
unit: SkillUnit,
|
||||
edit: LocalEdit,
|
||||
candidate: Path,
|
||||
section_depth: int | None,
|
||||
) -> None:
|
||||
self._validate_edit_candidate(unit, edit, section_depth)
|
||||
text = self._current_text()
|
||||
changed = replace_unit_text(text, unit, edit.new_text)
|
||||
temporary = candidate.with_name(f".{candidate.name}.{uuid.uuid4().hex}.tmp")
|
||||
shutil.copytree(self.current, temporary)
|
||||
atomic_write_text(temporary / "SKILL.md", changed)
|
||||
if candidate.exists():
|
||||
shutil.rmtree(candidate)
|
||||
candidate.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary.rename(candidate)
|
||||
|
||||
def _validate_edit_candidate(
|
||||
self,
|
||||
unit: SkillUnit,
|
||||
edit: LocalEdit,
|
||||
section_depth: int | None,
|
||||
) -> None:
|
||||
"""Validate a local edit against both unit and whole-document invariants."""
|
||||
|
||||
validate_edit(unit, edit.new_text, section_depth)
|
||||
text = self._current_text()
|
||||
if text[unit.start:unit.end] != unit.text:
|
||||
raise ValueError("target unit no longer matches current SKILL.md")
|
||||
changed = replace_unit_text(text, unit, edit.new_text)
|
||||
before = [(item.heading, item.heading_depth) for item in parse_sections(text)]
|
||||
after = [(item.heading, item.heading_depth) for item in parse_sections(changed)]
|
||||
if before != after:
|
||||
raise ValueError("local edit changed section boundaries")
|
||||
|
||||
@staticmethod
|
||||
def _validation_feedback(
|
||||
attempts: list[dict[str, Any]],
|
||||
) -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"edit_summary": str(item.get("edit_summary", "invalid generated edit")),
|
||||
"reject_reason": f"structural_validation_failed: {item['error']}",
|
||||
"new_text_hash": str(item.get("new_text_hash", "")),
|
||||
}
|
||||
for item in attempts
|
||||
]
|
||||
|
||||
def _valid_edit_or_rejection(
|
||||
self,
|
||||
*,
|
||||
level: str,
|
||||
number: int,
|
||||
iteration_dir: Path,
|
||||
unit: SkillUnit,
|
||||
coordinate: Coordinate,
|
||||
cell: CellScore,
|
||||
section_depth: int | None,
|
||||
) -> tuple[LocalEdit | None, bool, list[dict[str, Any]]]:
|
||||
"""Load or generate a valid edit, feeding structural failures back to the model."""
|
||||
|
||||
edit_path = iteration_dir / "edit.json"
|
||||
attempts_path = iteration_dir / "edit-attempts.json"
|
||||
attempts_value = load_json(attempts_path, [])
|
||||
attempts = attempts_value if isinstance(attempts_value, list) else []
|
||||
cached_value = load_json(edit_path)
|
||||
|
||||
if isinstance(cached_value, dict):
|
||||
try:
|
||||
cached = LocalEdit.from_dict(cached_value)
|
||||
cached.new_text = preserve_unit_boundary(unit, cached.new_text)
|
||||
self._validate_edit_candidate(unit, cached, section_depth)
|
||||
return cached, False, attempts
|
||||
except ValueError as exc:
|
||||
text = str(cached_value.get("new_text", ""))
|
||||
attempts.append({
|
||||
"source": "cached",
|
||||
"error": str(exc),
|
||||
"edit_summary": str(cached_value.get("edit_summary", "")),
|
||||
"new_text_hash": sha256_text(text) if text else "",
|
||||
})
|
||||
atomic_write_json(attempts_path, attempts)
|
||||
_log(
|
||||
f"{level} iteration {number}: cached local edit is invalid: {exc}; "
|
||||
"regenerating"
|
||||
)
|
||||
|
||||
for attempt in range(1, MAX_EDIT_GENERATION_ATTEMPTS + 1):
|
||||
_log(
|
||||
f"{level} iteration {number}: generating local edit for "
|
||||
f"{coordinate.unit_id}/{coordinate.dimension} "
|
||||
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS})"
|
||||
)
|
||||
feedback = self._rejected(coordinate) + self._validation_feedback(attempts)
|
||||
edit: LocalEdit | None = None
|
||||
try:
|
||||
edit = self.analyzer.generate_edit(
|
||||
coordinate,
|
||||
unit,
|
||||
cell,
|
||||
feedback,
|
||||
self._prompt(),
|
||||
)
|
||||
edit.new_text = preserve_unit_boundary(unit, edit.new_text)
|
||||
self._validate_edit_candidate(unit, edit, section_depth)
|
||||
except ValueError as exc:
|
||||
text = edit.new_text if edit is not None else ""
|
||||
attempts.append({
|
||||
"source": "generated",
|
||||
"generation_attempt": attempt,
|
||||
"error": str(exc),
|
||||
"edit_summary": edit.edit_summary if edit is not None else "",
|
||||
"new_text_hash": sha256_text(text) if text else "",
|
||||
})
|
||||
atomic_write_json(attempts_path, attempts)
|
||||
_log(
|
||||
f"{level} iteration {number}: local edit validation failed "
|
||||
f"(attempt {attempt}/{MAX_EDIT_GENERATION_ATTEMPTS}): {exc}"
|
||||
)
|
||||
continue
|
||||
|
||||
assert edit is not None
|
||||
atomic_write_json(edit_path, asdict(edit))
|
||||
return edit, True, attempts
|
||||
|
||||
return None, True, attempts
|
||||
|
||||
def _commit_iteration(
|
||||
self,
|
||||
level: str,
|
||||
number: int,
|
||||
iteration_dir: Path,
|
||||
matrix: ScoreMatrix,
|
||||
current_trace_path: Path,
|
||||
) -> ScoreMatrix:
|
||||
decision = load_json(iteration_dir / "decision.json")
|
||||
if not isinstance(decision, dict):
|
||||
raise ValueError("missing iteration decision")
|
||||
if decision["accepted"]:
|
||||
candidate = iteration_dir / "candidate" / self.state["skill_name"]
|
||||
replacement = self.temp / "next-current"
|
||||
shutil.rmtree(replacement, ignore_errors=True)
|
||||
shutil.copytree(candidate, replacement)
|
||||
shutil.rmtree(self.current)
|
||||
replacement.rename(self.current)
|
||||
matrix = ScoreMatrix.from_dict(decision["matrix_after"])
|
||||
atomic_write_json(self.run_dir / "levels" / level / "matrix.json", matrix.to_dict())
|
||||
candidate_traces = read_jsonl(iteration_dir / "candidate-traces.jsonl")
|
||||
atomic_write_jsonl(current_trace_path, candidate_traces)
|
||||
rollout_output = decision.get("rollout_output")
|
||||
rollout_skill_sha256 = decision.get("rollout_skill_sha256")
|
||||
if isinstance(rollout_output, str) and isinstance(rollout_skill_sha256, str):
|
||||
self.state["current_rollouts"] = rollout_output
|
||||
self.state["current_rollout_skill_sha256"] = rollout_skill_sha256
|
||||
else:
|
||||
self.state.pop("current_rollouts", None)
|
||||
self.state.pop("current_rollout_skill_sha256", None)
|
||||
level_state = self.state["levels"][level]
|
||||
if int(level_state.get("iterations", 0)) < number:
|
||||
level_state["iterations"] = number
|
||||
self._save()
|
||||
return matrix
|
||||
|
||||
def _run_iteration(
|
||||
self,
|
||||
level: str,
|
||||
number: int,
|
||||
level_dir: Path,
|
||||
matrix: ScoreMatrix,
|
||||
coordinate: Coordinate,
|
||||
current_trace_path: Path,
|
||||
section_depth: int | None,
|
||||
) -> ScoreMatrix:
|
||||
iteration_dir = level_dir / "iterations" / f"iteration-{number:02d}"
|
||||
iteration_dir.mkdir(parents=True, exist_ok=True)
|
||||
unit = matrix.unit(coordinate.unit_id)
|
||||
edit, regenerated, validation_attempts = self._valid_edit_or_rejection(
|
||||
level=level,
|
||||
number=number,
|
||||
iteration_dir=iteration_dir,
|
||||
unit=unit,
|
||||
coordinate=coordinate,
|
||||
cell=matrix.columns[coordinate.dimension][coordinate.unit_id],
|
||||
section_depth=section_depth,
|
||||
)
|
||||
if edit is None:
|
||||
decision = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": False,
|
||||
"reason": "edit_validation_exhausted",
|
||||
"target_delta": 0.0,
|
||||
"validation_attempts": validation_attempts,
|
||||
"rejected_edit": {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": (
|
||||
"Could not generate a structurally valid local edit after "
|
||||
f"{MAX_EDIT_GENERATION_ATTEMPTS} attempts."
|
||||
),
|
||||
"score_change": 0.0,
|
||||
"reject_reason": "edit_validation_exhausted",
|
||||
"new_text_hash": str(
|
||||
validation_attempts[-1].get("new_text_hash", "")
|
||||
) if validation_attempts else "",
|
||||
},
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
_log(
|
||||
f"{level} iteration {number}: local edit validation exhausted; "
|
||||
"recording rejection and continuing"
|
||||
)
|
||||
return self._commit_iteration(
|
||||
level, number, iteration_dir, matrix, current_trace_path
|
||||
)
|
||||
edit_hash = sha256_text(edit.new_text)
|
||||
if any(item["new_text_hash"] == edit_hash for item in self._rejected(coordinate)):
|
||||
decision = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": False,
|
||||
"reason": "exact_duplicate_rejected_edit",
|
||||
"target_delta": 0.0,
|
||||
"rejected_edit": {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": edit.edit_summary,
|
||||
"score_change": 0.0,
|
||||
"reject_reason": "exact_duplicate_rejected_edit",
|
||||
"new_text_hash": edit_hash,
|
||||
},
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
|
||||
candidate = iteration_dir / "candidate" / self.state["skill_name"]
|
||||
if regenerated:
|
||||
shutil.rmtree(candidate, ignore_errors=True)
|
||||
for stale in (
|
||||
iteration_dir / "candidate-traces.jsonl",
|
||||
iteration_dir / "comparison.json",
|
||||
):
|
||||
if stale.exists():
|
||||
stale.unlink()
|
||||
if not (candidate / "SKILL.md").is_file():
|
||||
self._apply_edit(unit, edit, candidate, section_depth)
|
||||
candidate_units = self._candidate_units(matrix.units, unit, edit.new_text)
|
||||
candidate_skill_sha256 = sha256_file(candidate / "SKILL.md")
|
||||
batch_id = (
|
||||
f"{self.state['run_id']}-deep-{level}-i{number:02d}-"
|
||||
f"{candidate_skill_sha256[:12]}"
|
||||
)
|
||||
traces = self._load_or_rollout(
|
||||
candidate,
|
||||
iteration_dir / "candidate-traces.jsonl",
|
||||
batch_id,
|
||||
)
|
||||
comparison_path = iteration_dir / "comparison.json"
|
||||
comparison = load_json(comparison_path)
|
||||
if not isinstance(comparison, dict):
|
||||
_log(
|
||||
f"{level} iteration {number}: comparing "
|
||||
f"{coordinate.unit_id}/{coordinate.dimension}"
|
||||
)
|
||||
incumbent_traces = [
|
||||
_trace_from_dict(item) for item in read_jsonl(current_trace_path)
|
||||
]
|
||||
comparison = self.analyzer.compare_cell(
|
||||
self._prompt(),
|
||||
unit,
|
||||
next(item for item in candidate_units if item.unit_id == unit.unit_id),
|
||||
incumbent_traces,
|
||||
traces,
|
||||
coordinate.dimension,
|
||||
)
|
||||
incumbent_timeout_rate = (
|
||||
sum(trace.timed_out is True for trace in incumbent_traces)
|
||||
/ len(incumbent_traces)
|
||||
)
|
||||
candidate_timeout_rate = (
|
||||
sum(trace.timed_out is True for trace in traces) / len(traces)
|
||||
)
|
||||
if (
|
||||
candidate_timeout_rate > incumbent_timeout_rate
|
||||
and "timeout" not in comparison["runtime_regressions"]
|
||||
):
|
||||
comparison["runtime_regressions"].append("timeout")
|
||||
atomic_write_json(comparison_path, comparison)
|
||||
|
||||
delta = float(comparison["candidate_score"]) - float(
|
||||
comparison["incumbent_score"]
|
||||
)
|
||||
if not comparison["task_relevant"]:
|
||||
accepted, reason = False, "target_unit_not_task_relevant"
|
||||
elif comparison["runtime_regressions"]:
|
||||
accepted = False
|
||||
reason = "runtime_regressed:" + ",".join(comparison["runtime_regressions"])
|
||||
elif delta < 0.5:
|
||||
accepted, reason = False, "target_cell_did_not_improve"
|
||||
else:
|
||||
accepted, reason = True, "target_improved_without_runtime_regression"
|
||||
decision: dict[str, Any] = {
|
||||
"coordinate": asdict(coordinate),
|
||||
"accepted": accepted,
|
||||
"reason": reason,
|
||||
"target_delta": delta,
|
||||
"rollout_skill_sha256": candidate_skill_sha256,
|
||||
}
|
||||
artifacts_dir = getattr(self.runner, "artifacts_dir", None)
|
||||
rollout_output = artifacts_dir(batch_id) if callable(artifacts_dir) else None
|
||||
if isinstance(rollout_output, Path):
|
||||
decision["rollout_output"] = str(rollout_output.resolve())
|
||||
if accepted:
|
||||
updated = ScoreMatrix(matrix.level, candidate_units, dict(matrix.columns))
|
||||
updated.columns[coordinate.dimension] = dict(
|
||||
matrix.columns[coordinate.dimension]
|
||||
)
|
||||
updated.columns[coordinate.dimension][coordinate.unit_id] = CellScore(
|
||||
float(comparison["candidate_score"]),
|
||||
list(comparison["candidate_evidence"]),
|
||||
str(comparison["reason"]),
|
||||
)
|
||||
decision["matrix_after"] = updated.to_dict()
|
||||
else:
|
||||
decision["rejected_edit"] = {
|
||||
"unit_id": coordinate.unit_id,
|
||||
"dimension": coordinate.dimension,
|
||||
"edit_summary": edit.edit_summary,
|
||||
"score_change": delta,
|
||||
"reject_reason": reason,
|
||||
"new_text_hash": edit_hash,
|
||||
}
|
||||
atomic_write_json(iteration_dir / "decision.json", decision)
|
||||
return self._commit_iteration(level, number, iteration_dir, matrix, current_trace_path)
|
||||
|
||||
def _run_level(
|
||||
self,
|
||||
level: str,
|
||||
units: list[SkillUnit],
|
||||
seed_traces: list[RolloutTrace] | None = None,
|
||||
section_depth: int | None = None,
|
||||
) -> tuple[ScoreMatrix, list[RolloutTrace], Coordinate | None]:
|
||||
level_dir = self.run_dir / "levels" / level
|
||||
level_dir.mkdir(parents=True, exist_ok=True)
|
||||
level_state = self.state["levels"].setdefault(level, {
|
||||
"iterations": 0, "completed": False,
|
||||
})
|
||||
current_trace_path = level_dir / "current-traces.jsonl"
|
||||
traces = self._load_or_rollout(
|
||||
self.current,
|
||||
current_trace_path,
|
||||
f"{self.state['run_id']}-deep-{level}-initial",
|
||||
seed=seed_traces,
|
||||
)
|
||||
matrix = self._load_or_matrix(level_dir / "matrix.json", level, units, traces)
|
||||
if level_state.get("completed"):
|
||||
exhausted_value = level_state.get("exhausted_coordinate")
|
||||
exhausted = Coordinate(**exhausted_value) if isinstance(exhausted_value, dict) else None
|
||||
return matrix, traces, exhausted
|
||||
exhausted_coordinate: Coordinate | None = None
|
||||
max_iterations = (
|
||||
MAX_SECTION_ITERATIONS if level == "section" else MAX_PARAGRAPH_ITERATIONS
|
||||
)
|
||||
while int(level_state["iterations"]) < max_iterations:
|
||||
excluded = self._exhausted(level)
|
||||
excluded.update(
|
||||
(unit.unit_id, dimension)
|
||||
for dimension in level_state.get("blocked_dimensions", [])
|
||||
for unit in matrix.units
|
||||
)
|
||||
pending_number = int(level_state["iterations"]) + 1
|
||||
pending_dir = level_dir / "iterations" / f"iteration-{pending_number:02d}"
|
||||
pending_decision = load_json(pending_dir / "decision.json")
|
||||
if isinstance(pending_decision, dict):
|
||||
pending_coordinate = Coordinate(**pending_decision["coordinate"])
|
||||
level_state.setdefault("active_dimension", pending_coordinate.dimension)
|
||||
matrix = self._commit_iteration(
|
||||
level, pending_number, pending_dir, matrix, current_trace_path
|
||||
)
|
||||
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
|
||||
if len(self._rejected(pending_coordinate)) >= REJECTION_LIMIT:
|
||||
exhausted_coordinate = pending_coordinate
|
||||
level_state["stop_reason"] = "coordinate_exhausted"
|
||||
level_state["exhausted_coordinate"] = asdict(pending_coordinate)
|
||||
break
|
||||
if matrix.select_coordinate(
|
||||
GAP_THRESHOLD, level_state.get("active_dimension"), excluded
|
||||
) is None:
|
||||
matrix = self._refresh_matrix(level, matrix.units, traces)
|
||||
continue
|
||||
active_dimension = level_state.get("active_dimension")
|
||||
coordinate = matrix.select_coordinate(
|
||||
GAP_THRESHOLD, active_dimension, excluded
|
||||
)
|
||||
if coordinate is None:
|
||||
if active_dimension is None:
|
||||
level_state["stop_reason"] = (
|
||||
"normalized_gap_converged"
|
||||
if max(matrix.normalized_gaps().values(), default=0.0) <= GAP_THRESHOLD
|
||||
else "available_coordinates_exhausted"
|
||||
)
|
||||
break
|
||||
matrix = self._refresh_matrix(level, matrix.units, traces)
|
||||
continue
|
||||
if active_dimension is None:
|
||||
level_state["active_dimension"] = coordinate.dimension
|
||||
self._save()
|
||||
number = int(level_state["iterations"]) + 1
|
||||
matrix = self._run_iteration(
|
||||
level, number, level_dir, matrix, coordinate,
|
||||
current_trace_path, section_depth,
|
||||
)
|
||||
traces = [_trace_from_dict(item) for item in read_jsonl(current_trace_path)]
|
||||
if len(self._rejected(coordinate)) >= REJECTION_LIMIT:
|
||||
exhausted_coordinate = coordinate
|
||||
level_state["stop_reason"] = "coordinate_exhausted"
|
||||
level_state["exhausted_coordinate"] = asdict(coordinate)
|
||||
break
|
||||
else:
|
||||
decisions = self._decisions(level)
|
||||
last = decisions[-1] if decisions else {}
|
||||
if matrix.select_coordinate(GAP_THRESHOLD) is None:
|
||||
level_state["stop_reason"] = "normalized_gap_converged"
|
||||
else:
|
||||
level_state["stop_reason"] = (
|
||||
"max_iterations_after_accept" if last.get("accepted") else "max_iterations"
|
||||
)
|
||||
level_state["completed"] = True
|
||||
self._save()
|
||||
return matrix, traces, exhausted_coordinate
|
||||
|
||||
def drive(self) -> Path:
|
||||
if self.state.get("status") == "complete" and (self.run_dir / "S_final").is_dir():
|
||||
return self.run_dir / "S_final"
|
||||
_log(f"Deep Loop start/resume: {self.run_dir}")
|
||||
input_traces = [
|
||||
_trace_from_dict(item) for item in read_jsonl(self.run_dir / "input-traces.jsonl")
|
||||
]
|
||||
while True:
|
||||
section_units = parse_sections(self._current_text())
|
||||
section_matrix, traces, exhausted = self._run_level(
|
||||
"section", section_units, seed_traces=input_traces
|
||||
)
|
||||
if exhausted is None:
|
||||
break
|
||||
target_score = section_matrix.columns[exhausted.dimension][exhausted.unit_id].score
|
||||
current_sections = parse_sections(self._current_text())
|
||||
section = next((item for item in current_sections if item.unit_id == exhausted.unit_id), None)
|
||||
paragraphs = parse_paragraphs(section) if section is not None else []
|
||||
section_state = self.state["levels"]["section"]
|
||||
if (
|
||||
target_score > 3.5
|
||||
or len(paragraphs) < 2
|
||||
or section_state.get("paragraph_returned")
|
||||
):
|
||||
self._block_dimension(section_state, exhausted.dimension)
|
||||
section_state["completed"] = False
|
||||
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
|
||||
section_state.pop(key, None)
|
||||
self._save()
|
||||
continue
|
||||
_log(f"Descending into paragraphs of {exhausted.unit_id}")
|
||||
self.state["levels"].setdefault(
|
||||
"paragraph", {"iterations": 0, "completed": False}
|
||||
)["active_dimension"] = exhausted.dimension
|
||||
self._save()
|
||||
_, paragraph_traces, _ = self._run_level(
|
||||
"paragraph", paragraphs, seed_traces=traces,
|
||||
section_depth=section.heading_depth if section else None,
|
||||
)
|
||||
atomic_write_jsonl(
|
||||
self.run_dir / "levels" / "section" / "current-traces.jsonl",
|
||||
[asdict(trace) for trace in paragraph_traces],
|
||||
)
|
||||
section_state["completed"] = False
|
||||
section_state["paragraph_returned"] = True
|
||||
self._block_dimension(section_state, exhausted.dimension)
|
||||
for key in ("stop_reason", "exhausted_coordinate", "active_dimension"):
|
||||
section_state.pop(key, None)
|
||||
self._refresh_matrix(
|
||||
"section", parse_sections(self._current_text()), paragraph_traces
|
||||
)
|
||||
return self._complete()
|
||||
|
||||
def _complete(self) -> Path:
|
||||
target = self.run_dir / "S_final"
|
||||
if target.exists():
|
||||
shutil.rmtree(target)
|
||||
shutil.copytree(self.current, target)
|
||||
final_skill_hash = sha256_file(target / "SKILL.md")
|
||||
final_rollouts = self.state.get("current_rollouts")
|
||||
final_rollout_hash = self.state.get("current_rollout_skill_sha256")
|
||||
if (
|
||||
not isinstance(final_rollouts, str)
|
||||
or not Path(final_rollouts).is_dir()
|
||||
or final_rollout_hash != final_skill_hash
|
||||
):
|
||||
final_rollouts = None
|
||||
report = {
|
||||
"status": "complete",
|
||||
"input_package_hash": package_hash(self.run_dir / "S_fast"),
|
||||
"final_package_hash": package_hash(target),
|
||||
"final_rollouts": final_rollouts,
|
||||
"final_rollout_skill_sha256": final_skill_hash if final_rollouts else None,
|
||||
"task": self.state["task"]["name"],
|
||||
"agent": self.state["task"]["agent"],
|
||||
"target_model": self.state["task"]["model"],
|
||||
"model": self.model,
|
||||
"rollout_backend": "skillsbench_development",
|
||||
"production_rollout_backend": "blank_container_required",
|
||||
"verifier_signal_used": False,
|
||||
"levels": self.state["levels"],
|
||||
"decisions": {
|
||||
name: self._decisions(name)
|
||||
for name in ("section", "paragraph")
|
||||
if name in self.state["levels"]
|
||||
},
|
||||
}
|
||||
atomic_write_json(self.run_dir / "report.json", report)
|
||||
self.state["status"] = "complete"
|
||||
self._save()
|
||||
shutil.rmtree(self.temp, ignore_errors=True)
|
||||
_log(f"Deep Loop complete: {target}")
|
||||
return target
|
||||
Reference in New Issue
Block a user