Initial commit
This commit is contained in:
@@ -0,0 +1,197 @@
|
||||
"""语义模型评分与局部编辑适配。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from typing import Any
|
||||
|
||||
from scripts.dynamic_compile.fast.models import RolloutTrace
|
||||
from scripts.dynamic_compile.fast.optimization.analyzer import SemanticClient
|
||||
from scripts.dynamic_compile.fast.optimization.trace_format import compact_trace
|
||||
|
||||
from ..core.models import CellScore, Coordinate, DIMENSIONS, LocalEdit, ScoreMatrix, SkillUnit
|
||||
|
||||
|
||||
RUBRICS = {
|
||||
"Clarity": "Judge whether requirements, actions, conditions, references, and terms are unambiguous and internally consistent.",
|
||||
"Structure": "Judge whether information, rules, prerequisites, and action order form a clear execution path at the current unit level.",
|
||||
"Executability": "Judge whether the unit specifies the necessary concrete actions for its relevant responsibility without adding unrelated work, unsupported tools, task-specific literals, or unjustified fixed procedures.",
|
||||
"Completeness": "Judge whether the unit contains the information, conditions, and steps needed to fulfill its own responsibility.",
|
||||
"Constraint Salience": "Judge whether important constraints are explicit, well placed, noticeable, and consistently followed in the traces.",
|
||||
}
|
||||
|
||||
_EVIDENCE = re.compile(r"^[^:]+:E\d{3}(?:\.T\d{2})?$")
|
||||
|
||||
|
||||
def valid_evidence(values: Any, traces: list[RolloutTrace]) -> list[str]:
|
||||
if not isinstance(values, list):
|
||||
return []
|
||||
prefixes = tuple(f"{trace.trace_id}:" for trace in traces)
|
||||
return [
|
||||
value for value in values
|
||||
if isinstance(value, str)
|
||||
and _EVIDENCE.fullmatch(value)
|
||||
and value.startswith(prefixes)
|
||||
]
|
||||
|
||||
|
||||
class DeepAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
@staticmethod
|
||||
def _trace_payload(traces: list[RolloutTrace]) -> str:
|
||||
return "\n\n".join(compact_trace(trace, total=6000) for trace in traces)
|
||||
|
||||
def score_column(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, CellScore]:
|
||||
unit_payload = [
|
||||
{"unit_id": unit.unit_id, "heading": unit.heading, "text": unit.text}
|
||||
for unit in units
|
||||
]
|
||||
result = self.client.json(
|
||||
"You are a rubric-based judge for agent skill instructions. Return JSON only.",
|
||||
f"""Score every current-level unit only on {dimension}. Use the task prompt to judge relevance and the complete skill and observable traces as evidence. Do not reward task-specific literals, benchmark orchestration, unrelated mandatory work, unsupported tools, or unjustified fixed procedures. Scores must be from 1.0 to 5.0 in 0.5 increments. Every evidence entry must copy the actual trace_id from RUNTIME_FACTS followed by :E### or :E###.T##; never write the literal word trace_id. Use an empty evidence list when the judgment is textual rather than trace-supported. Return exactly {{"dimension":"{dimension}","scores":[{{"unit_id":string,"score":number,"evidence":[string],"reason":string}}]}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Current units:
|
||||
{json.dumps(unit_payload, ensure_ascii=False)}
|
||||
Agent traces:
|
||||
{self._trace_payload(traces)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
if result.get("dimension") != dimension or not isinstance(result.get("scores"), list):
|
||||
raise ValueError(f"judge returned an invalid {dimension} column")
|
||||
expected = {unit.unit_id for unit in units}
|
||||
column: dict[str, CellScore] = {}
|
||||
for item in result["scores"]:
|
||||
if not isinstance(item, dict):
|
||||
raise ValueError("judge score entries must be objects")
|
||||
unit_id = str(item.get("unit_id", ""))
|
||||
score = float(item.get("score"))
|
||||
evidence = item.get("evidence", [])
|
||||
if unit_id not in expected or unit_id in column:
|
||||
raise ValueError(f"judge returned unexpected or duplicate unit: {unit_id}")
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid score for {unit_id}: {score}")
|
||||
evidence = valid_evidence(evidence, traces)
|
||||
column[unit_id] = CellScore(score, evidence, str(item.get("reason", "")))
|
||||
if set(column) != expected:
|
||||
raise ValueError(f"judge omitted units: {sorted(expected - set(column))}")
|
||||
return column
|
||||
|
||||
def compare_cell(
|
||||
self,
|
||||
task: str,
|
||||
incumbent_unit: SkillUnit,
|
||||
candidate_unit: SkillUnit,
|
||||
incumbent_traces: list[RolloutTrace],
|
||||
candidate_traces: list[RolloutTrace],
|
||||
dimension: str,
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Compare one incumbent and candidate skill unit. Return JSON only.",
|
||||
f"""Compare only the target {incumbent_unit.level} on {dimension}. Decide whether the edit is relevant to the task prompt, including edits that remove unrelated work. Score incumbent and candidate from 1.0 to 5.0 in 0.5 increments using the same calibration. Report a runtime regression only when candidate traces newly show a higher rate of timeout, tool_not_found, invalid_parameters, or required_output_missing than incumbent traces. Use observable runtime facts only; do not infer verifier outcomes or hidden correctness. Return exactly {{"task_relevant":boolean,"incumbent_score":number,"candidate_score":number,"candidate_evidence":[string],"runtime_regressions":["timeout"|"tool_not_found"|"invalid_parameters"|"required_output_missing"],"reason":string}}.
|
||||
|
||||
Rubric: {RUBRICS[dimension]}
|
||||
Task prompt:
|
||||
{task}
|
||||
Incumbent unit:
|
||||
{incumbent_unit.text}
|
||||
Candidate unit:
|
||||
{candidate_unit.text}
|
||||
Incumbent runtime facts:
|
||||
{self._trace_payload(incumbent_traces)}
|
||||
Candidate runtime facts:
|
||||
{self._trace_payload(candidate_traces)}""",
|
||||
)
|
||||
if not isinstance(result.get("task_relevant"), bool):
|
||||
raise ValueError("judge returned invalid task relevance")
|
||||
incumbent_score = float(result.get("incumbent_score"))
|
||||
candidate_score = float(result.get("candidate_score"))
|
||||
for score in (incumbent_score, candidate_score):
|
||||
if score < 1 or score > 5 or abs(score * 2 - round(score * 2)) > 1e-9:
|
||||
raise ValueError(f"judge returned an invalid paired score: {score}")
|
||||
regressions = result.get("runtime_regressions")
|
||||
allowed = {
|
||||
"timeout", "tool_not_found", "invalid_parameters", "required_output_missing",
|
||||
}
|
||||
if not isinstance(regressions, list) or any(item not in allowed for item in regressions):
|
||||
raise ValueError("judge returned invalid runtime regressions")
|
||||
return {
|
||||
"task_relevant": result["task_relevant"],
|
||||
"incumbent_score": incumbent_score,
|
||||
"candidate_score": candidate_score,
|
||||
"candidate_evidence": valid_evidence(
|
||||
result.get("candidate_evidence"), candidate_traces
|
||||
),
|
||||
"runtime_regressions": list(dict.fromkeys(regressions)),
|
||||
"reason": str(result.get("reason", "")),
|
||||
}
|
||||
|
||||
def score_matrix(
|
||||
self,
|
||||
skill_text: str,
|
||||
task: str,
|
||||
units: list[SkillUnit],
|
||||
traces: list[RolloutTrace],
|
||||
level: str,
|
||||
existing_columns: dict[str, dict[str, CellScore]] | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> ScoreMatrix:
|
||||
columns = dict(existing_columns or {})
|
||||
missing = [dimension for dimension in DIMENSIONS if dimension not in columns]
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {
|
||||
pool.submit(self.score_column, skill_text, task, units, traces, dimension): dimension
|
||||
for dimension in missing
|
||||
}
|
||||
for future in as_completed(futures):
|
||||
dimension = futures[future]
|
||||
columns[dimension] = future.result()
|
||||
if result_callback is not None:
|
||||
result_callback(dimension, columns[dimension])
|
||||
return ScoreMatrix(level, units, {dimension: columns[dimension] for dimension in DIMENSIONS})
|
||||
|
||||
def generate_edit(
|
||||
self,
|
||||
coordinate: Coordinate,
|
||||
unit: SkillUnit,
|
||||
cell: CellScore,
|
||||
rejected: list[dict[str, Any]],
|
||||
task_prompt: str,
|
||||
) -> LocalEdit:
|
||||
result = self.client.json(
|
||||
"Generate one bounded local edit for an agent skill. Return JSON only.",
|
||||
f"""Improve exactly one {unit.level} unit on exactly one dimension. Return {{"unit_id":"{unit.unit_id}","dimension":"{coordinate.dimension}","new_text":string,"edit_summary":string,"reason":string}}.
|
||||
|
||||
Use the task prompt only to determine which capability is relevant. Make the smallest reusable edit for the skill's general domain. Do not copy task-specific paths, filenames, output schemas, fixed counts, one-off entities, or benchmark and Skill-invocation instructions into new_text.
|
||||
|
||||
new_text must be a complete replacement for the target unit. Preserve the peer heading and unrelated behavior. For a section edit, copy every fenced code block byte-for-byte, including its fence markers, language tag, contents, whitespace, and line endings; improve incorrect or obsolete examples only through surrounding prose. Do not repeat a rejected edit. Rejected memory may include structural_validation_failed feedback from an earlier generation attempt; correct that exact failure in the next edit.
|
||||
|
||||
Target dimension rubric: {RUBRICS[coordinate.dimension]}
|
||||
Task prompt:
|
||||
{task_prompt}
|
||||
Target unit:
|
||||
{unit.text}
|
||||
Score: {cell.score}
|
||||
Evidence: {json.dumps(cell.evidence, ensure_ascii=False)}
|
||||
Reason: {cell.reason}
|
||||
Rejected memory: {json.dumps(rejected, ensure_ascii=False)}""",
|
||||
)
|
||||
edit = LocalEdit.from_dict(result)
|
||||
if edit.unit_id != unit.unit_id or edit.dimension != coordinate.dimension:
|
||||
raise ValueError("edit generator changed the target coordinate")
|
||||
return edit
|
||||
Reference in New Issue
Block a user