Initial commit
This commit is contained in:
@@ -0,0 +1,216 @@
|
||||
"""BenchFlow 输入与运行时轨迹适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.traces.benchflow import trajectory_for_test
|
||||
from scripts.dynamic_compile.fast.scoring.state import acp_events_to_state, read_acp_events
|
||||
|
||||
|
||||
@dataclass
|
||||
class BenchFlowInput:
|
||||
task_name: str
|
||||
task_dir: Path
|
||||
agent: str
|
||||
model: str
|
||||
prompt: str
|
||||
traces: list[RolloutTrace]
|
||||
|
||||
|
||||
def _project_root() -> Path:
|
||||
return Path(__file__).resolve().parents[4]
|
||||
|
||||
|
||||
def _run_config(test_dir: Path) -> dict[str, Any]:
|
||||
paths = sorted(test_dir.rglob("config.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow run config under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError(f"invalid BenchFlow run config: {paths[0]}")
|
||||
return value
|
||||
|
||||
|
||||
def _prompt(test_dir: Path) -> str:
|
||||
paths = sorted(test_dir.rglob("prompts.json"))
|
||||
if not paths:
|
||||
raise ValueError(f"missing BenchFlow prompts.json under {test_dir}")
|
||||
value = json.loads(paths[0].read_text(encoding="utf-8"))
|
||||
if not isinstance(value, list) or not value or not isinstance(value[0], str):
|
||||
raise ValueError(f"invalid BenchFlow prompts: {paths[0]}")
|
||||
return value[0].strip()
|
||||
|
||||
|
||||
def load_runtime_trace(test_dir: Path, task_name: str, compile_type: str) -> RolloutTrace:
|
||||
"""Load agent runtime evidence without opening verifier/result artifacts."""
|
||||
trajectory = trajectory_for_test(test_dir)
|
||||
if trajectory is None:
|
||||
raise ValueError(f"missing acp_trajectory.jsonl under {test_dir}")
|
||||
events = read_acp_events(trajectory)
|
||||
state = acp_events_to_state(events)
|
||||
skill_invoked = any(
|
||||
event.get("type") == "tool_call"
|
||||
and event.get("status") == "completed"
|
||||
and any(
|
||||
str(event.get(field, "")).strip().lower() == "skill"
|
||||
for field in ("title", "kind")
|
||||
)
|
||||
for event in events
|
||||
)
|
||||
timed_out = any(event.get("type") == "agent_timeout" for event in events)
|
||||
return RolloutTrace(
|
||||
trace_id=f"{task_name}/{compile_type}/{test_dir.name}",
|
||||
task_name=task_name,
|
||||
compile_type=compile_type,
|
||||
test_name=test_dir.name,
|
||||
state=state,
|
||||
skill_invoked=skill_invoked,
|
||||
timed_out=timed_out,
|
||||
metadata={
|
||||
"termination": "timeout" if timed_out else "completed",
|
||||
"tool_calls": sum(event.get("type") == "tool_call" for event in events),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def load_benchflow_input(source: Path) -> BenchFlowInput:
|
||||
source = source.resolve()
|
||||
tests = sorted(path for path in source.glob("test-*") if path.is_dir())[-5:]
|
||||
if not tests:
|
||||
raise ValueError(f"no test-* BenchFlow traces under {source}")
|
||||
task_name = source.parent.name
|
||||
traces = [load_runtime_trace(test, task_name, "custom") for test in tests]
|
||||
configs = [_run_config(test) for test in tests]
|
||||
agents = {str(item.get("agent", "")).strip() for item in configs}
|
||||
models = {str(item.get("model", "")).strip() for item in configs}
|
||||
if "" in agents or len(agents) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one agent: {sorted(agents)}")
|
||||
if "" in models or len(models) != 1:
|
||||
raise ValueError(f"BenchFlow traces do not identify one model: {sorted(models)}")
|
||||
prompts = {_prompt(test) for test in tests}
|
||||
if len(prompts) != 1:
|
||||
raise ValueError(f"BenchFlow traces contain {len(prompts)} different task prompts")
|
||||
task_dir = _project_root() / "data" / "skills-bench" / "tasks" / task_name
|
||||
if not (task_dir / "task.md").is_file():
|
||||
raise ValueError(f"cannot resolve SkillsBench task directory: {task_dir}")
|
||||
return BenchFlowInput(
|
||||
task_name, task_dir, next(iter(agents)), next(iter(models)),
|
||||
next(iter(prompts)), traces,
|
||||
)
|
||||
|
||||
|
||||
def _slug(value: str) -> str:
|
||||
return re.sub(r"^-+|-+$", "", re.sub(r"[^a-z0-9]+", "-", value.lower()))
|
||||
|
||||
|
||||
class SkillsBenchDevelopmentRolloutRunner:
|
||||
"""Development-only runner; verifier output is never projected into traces."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
context: BenchFlowInput,
|
||||
work_root: Path,
|
||||
max_parallel: int = 3,
|
||||
archive_root: Path | None = None,
|
||||
):
|
||||
self.context = context
|
||||
self.work_root = work_root
|
||||
self.max_parallel = max_parallel
|
||||
self.archive_root = archive_root
|
||||
|
||||
def _variant_dir(self, jobs_root: Path) -> Path:
|
||||
return (
|
||||
jobs_root
|
||||
/ _slug(f"{self.context.agent}-{self.context.model}")
|
||||
/ self.context.task_name
|
||||
/ "custom_skill"
|
||||
)
|
||||
|
||||
def _completed_traces(self, variant: Path, batch_id: str) -> list[RolloutTrace]:
|
||||
completed: list[RolloutTrace] = []
|
||||
for test in sorted(path for path in variant.glob("test-*") if path.is_dir()):
|
||||
try:
|
||||
requirement = json.loads(
|
||||
(test / "required-skill.json").read_text(encoding="utf-8")
|
||||
)
|
||||
if requirement.get("invoked") is not True:
|
||||
continue
|
||||
trace = load_runtime_trace(test, self.context.task_name, batch_id)
|
||||
except (AttributeError, json.JSONDecodeError, OSError, ValueError):
|
||||
continue
|
||||
completed.append(trace)
|
||||
return completed
|
||||
|
||||
def artifacts_dir(self, batch_id: str) -> Path | None:
|
||||
if self.archive_root is None:
|
||||
return None
|
||||
return self.archive_root / batch_id / "custom_skill"
|
||||
|
||||
def run_batch(
|
||||
self,
|
||||
skill_package: Path,
|
||||
_prompt: str,
|
||||
batch_id: str,
|
||||
_task_name: str,
|
||||
count: int,
|
||||
progress: Any | None = None,
|
||||
) -> list[RolloutTrace]:
|
||||
jobs_root = self.work_root / batch_id
|
||||
log_path = jobs_root / "runner.log"
|
||||
variant = self._variant_dir(jobs_root)
|
||||
archive = self.artifacts_dir(batch_id)
|
||||
if archive is not None and archive.is_dir():
|
||||
archived_traces = self._completed_traces(archive, batch_id)
|
||||
if len(archived_traces) >= count:
|
||||
traces = archived_traces
|
||||
missing = 0
|
||||
else:
|
||||
variant.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(archive, variant, dirs_exist_ok=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
else:
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
missing = max(0, count - len(traces))
|
||||
if missing:
|
||||
jobs_root.mkdir(parents=True, exist_ok=True)
|
||||
command = [
|
||||
"bash", str(_project_root() / "scripts" / "evaluate" / "run-raw-task.sh"),
|
||||
"--harness", self.context.agent,
|
||||
"--model", self.context.model,
|
||||
"--task", str(self.context.task_dir),
|
||||
"--skill-source", str(skill_package.resolve()),
|
||||
"--require-skill",
|
||||
"--repeat", str(missing),
|
||||
"--max-parallel", str(self.max_parallel),
|
||||
"--output", str(variant),
|
||||
]
|
||||
with log_path.open("a", encoding="utf-8") as handle:
|
||||
subprocess.run(command, stdout=handle, stderr=subprocess.STDOUT, text=True)
|
||||
traces = self._completed_traces(variant, batch_id)
|
||||
if archive is not None and variant.is_dir():
|
||||
archive.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copytree(variant, archive, dirs_exist_ok=True)
|
||||
runner_log = jobs_root / "runner.log"
|
||||
if runner_log.is_file():
|
||||
shutil.copy2(runner_log, archive.parent / "runner.log")
|
||||
traces = self._completed_traces(archive, batch_id)
|
||||
if len(traces) < count:
|
||||
raise RuntimeError(
|
||||
f"BenchFlow produced {len(traces)}/{count} candidate traces; see {log_path}"
|
||||
)
|
||||
traces = traces[:count]
|
||||
for index, trace in enumerate(traces, 1):
|
||||
trace.trace_id = f"{batch_id}-{index:03d}"
|
||||
trace.test_name = trace.trace_id
|
||||
if progress is not None:
|
||||
progress(index, len(traces), trace.trace_id)
|
||||
return traces
|
||||
@@ -0,0 +1,197 @@
|
||||
"""语义模型评分与局部编辑适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
|
||||
|
||||
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
|
||||
|
||||
|
||||
RUBRICS = {
|
||||
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
|
||||
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
|
||||
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
|
||||
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
|
||||
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
|
||||
}
|
||||
|
||||
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
|
||||
|
||||
|
||||
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
|
||||
return [
|
||||
value for value in values
|
||||
if isinstance(value, str)
|
||||
and _EVIDENCE.fullmatch(value)
|
||||
and value.startswith(prefixes)
|
||||
]
|
||||
|
||||
|
||||
class DeepAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
@staticmethod
|
||||
def _trace_payload(traces: list[RolloutTrace]) -> str:
|
||||
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
|
||||
|
||||
def score_column(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, CellScore]:
|
||||
unit_payload = [
|
||||
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
|
||||
for unit in units
|
||||
]
|
||||
result = self.client.json(
|
||||
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
|
||||
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Current units:
|
||||
{json.dumps(unit_payload, ensure_ascii=False)}
|
||||
Agent traces:
|
||||
{self._trace_payload(traces)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
|
||||
raise ValueError(f"judge returned an invalid {dimension} column")
|
||||
expected = {unit.unit_id for unit in units}
|
||||
column: dict[str, CellScore] = {}
|
||||
for item in result["scores"]:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("judge score entries must be objects")
|
||||
unit_id = str(item.get("unit_id", ""))
|
||||
score = float(item.get("score"))
|
||||
evidence = item.get("evidence", [])
|
||||
if unit_id not in expected or unit_id in column:
|
||||
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
|
||||
evidence = valid_evidence(evidence, traces)
|
||||
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
|
||||
if set(column) != expected:
|
||||
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
|
||||
return column
|
||||
|
||||
def compare_cell(
|
||||
self,
|
||||
task: str,
|
||||
incumbent_unit: SkillUnit,
|
||||
candidate_unit: SkillUnit,
|
||||
incumbent_traces: list[RolloutTrace],
|
||||
candidate_traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Compare one incumbent and candidate skill unit. Return JSON only.",
|
||||
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Incumbent unit:
|
||||
{incumbent_unit.text}
|
||||
Candidate unit:
|
||||
{candidate_unit.text}
|
||||
Incumbent runtime facts:
|
||||
{self._trace_payload(incumbent_traces)}
|
||||
Candidate runtime facts:
|
||||
{self._trace_payload(candidate_traces)}""",
|
||||
)
|
||||
if not isinstance(result.get("task_relevant"), bool):
|
||||
raise ValueError("judge returned invalid task relevance")
|
||||
incumbent_score = float(result.get("incumbent_score"))
|
||||
candidate_score = float(result.get("candidate_score"))
|
||||
for score in (incumbent_score, candidate_score):
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid paired score: {score}")
|
||||
regressions = result.get("runtime_regressions")
|
||||
allowed = {
|
||||
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
|
||||
}
|
||||
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
|
||||
raise ValueError("judge returned invalid runtime regressions")
|
||||
return {
|
||||
"task_relevant": result["task_relevant"],
|
||||
"incumbent_score": incumbent_score,
|
||||
"candidate_score": candidate_score,
|
||||
"candidate_evidence": valid_evidence(
|
||||
result.get("candidate_evidence"), candidate_traces
|
||||
),
|
||||
"runtime_regressions": list(dict.fromkeys(regressions)),
|
||||
"reason": str(result.get("reason", "")),
|
||||
}
|
||||
|
||||
def score_matrix(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
level: str,
|
||||
existing_columns: dict[str, dict[str, CellScore]] | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> ScoreMatrix:
|
||||
columns = dict(existing_columns or {})
|
||||
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {
|
||||
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
|
||||
for dimension in missing
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
dimension = futures[future]
|
||||
columns[dimension] = future.result()
|
||||
if result_callback is not None:
|
||||
result_callback(dimension, columns[dimension])
|
||||
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
|
||||
|
||||
def generate_edit(
|
||||
self,
|
||||
coordinate: Coordinate,
|
||||
unit: SkillUnit,
|
||||
cell: CellScore,
|
||||
rejected: list[dict[str, Any]],
|
||||
task_prompt: str,
|
||||
) -> LocalEdit:
|
||||
result = self.client.json(
|
||||
"Generate one bounded local edit for an agent skill. Return JSON only.",
|
||||
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
|
||||
|
||||
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
|
||||
|
||||
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
|
||||
|
||||
Target dimension rubric: {RUBRICS[coordinate.dimension]}
|
||||
Task prompt:
|
||||
{task_prompt}
|
||||
Target unit:
|
||||
{unit.text}
|
||||
Score: {cell.score}
|
||||
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
|
||||
Reason: {cell.reason}
|
||||
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
|
||||
)
|
||||
edit = LocalEdit.from_dict(result)
|
||||
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
|
||||
raise ValueError("edit generator changed the target coordinate")
|
||||
return edit
|
||||
Reference in New Issue
Block a user