Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
@@ -0,0 +1,239 @@
from __future__ import annotations
import json
import sys
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Any
from ..models import Patch, RolloutTrace
from scripts.provider_router import resolve_model_route
from ..storage import package_manifest, sha256_file
from .trace_format import compact_trace
class SemanticClient:
def __init__(
self,
model: str = "opencode/deepseek-v4-pro",
timeout: int = 900,
):
route = resolve_model_route(model)
assert route is not None
base_url = route.url.removesuffix("/chat/completions").rstrip("/")
try:
from openai import OpenAI
except ImportError as exc:
raise RuntimeError("the openai package is required for semantic calls") from exc
self.client = OpenAI(base_url=base_url, api_key=route.api_key, timeout=timeout, max_retries=0)
self.model = route.reference.model_id
self.model_reference = route.reference.value
def json(self, system: str, user: str, attempts: int = 1) -> dict[str, Any]:
error: Exception | None = None
for attempt in range(attempts):
try:
response = self.client.chat.completions.create(
model=self.model,
temperature=0,
response_format={"type": "json_object"},
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
stream=True,
)
content = "".join(
choice.delta.content or ""
for chunk in response
for choice in chunk.choices
)
value = json.loads(content or "{}")
if not isinstance(value, dict):
raise ValueError("semantic response must be a JSON object")
return value
except Exception as exc:
error = exc
if attempt + 1 < attempts:
print(
f"[semantic] request attempt {attempt + 1}/{attempts} failed: "
f"{type(exc).__name__}: {exc}; retrying",
file=sys.stderr,
flush=True,
)
time.sleep(1 + attempt)
raise RuntimeError(f"semantic call failed after {attempts} attempts: {error}")
class SemanticAnalyzer:
def __init__(self, client: SemanticClient, max_parallel: int = 3):
self.client = client
self.max_parallel = max_parallel
def generate_probe(self, skill_package: Path) -> dict[str, str]:
skill = (skill_package / "SKILL.md").read_text(encoding="utf-8")
files = [item["path"] for item in package_manifest(skill_package)]
result = self.client.json(
"You create realistic probe tasks for testing an agent skill. Return JSON only.",
f"""Create one task prompt for the skill below. The task must explicitly tell the agent to invoke this skill, exercise its core workflow, and remain solvable in an empty workspace with network access. Do not copy the skill's operation steps, create a verifier, or create a test environment. Return {{"prompt": string, "rationale": string}}.
Package files: {json.dumps(files, ensure_ascii=False)}
SKILL.md:
{skill}""",
)
prompt = result.get("prompt")
if not isinstance(prompt, str) or not prompt.strip():
raise ValueError("probe generator returned no prompt")
return {"prompt": prompt.strip(), "rationale": str(result.get("rationale", ""))}
def map_trace(self, trace: RolloutTrace, score: float, bucket: str) -> dict[str, Any]:
result = self.client.json(
"Analyze agent behavior from a scored trace. Return concise JSON only.",
f"""Analyze this {bucket} trace (AgentRM score {score}) as an action-level workflow. Identify what the agent did, action ordering, stopping behavior, error recovery, and whether required outputs were persisted promptly. Runtime facts report observable execution only; do not infer external verification outcomes or correctness that is not visible in the trace. Use long payload details only when they are necessary to explain a behavioral effect. Return:
{{"trace_id":"{trace.trace_id}","patterns":[{{"description":string,"condition":string,"effect":string,"recovered":boolean,"final_quality_impact":string,"evidence_ids":[string]}}]}}.
Use E### or E###.T## event IDs as evidence. Skill invocation is evidence, not a quality gate.
{compact_trace(trace)}""",
)
result["runtime_facts"] = {
"termination": trace.metadata.get("termination"),
"timed_out": trace.timed_out,
"exit_code": trace.exit_code,
"duration_seconds": trace.metadata.get("agent_execution_seconds"),
"tool_calls": trace.metadata.get("tool_calls"),
"skill_invoked": trace.skill_invoked,
"timeout_reason": trace.metadata.get("timeout_reason"),
"error_category": trace.metadata.get("error_category"),
"partial_trajectory": trace.metadata.get("partial_trajectory"),
}
result.setdefault("trace_id", trace.trace_id)
result.setdefault("patterns", [])
return result
def map_all(
self,
traces: list[RolloutTrace],
scores: dict[str, float],
high_ids: set[str],
low_ids: set[str] | None = None,
progress: Any | None = None,
result_callback: Any | None = None,
) -> list[dict[str, Any]]:
def work(trace: RolloutTrace) -> dict[str, Any]:
bucket = "High" if trace.trace_id in high_ids else "Low" if low_ids is None or trace.trace_id in low_ids else "Neutral"
return self.map_trace(trace, scores[trace.trace_id], bucket)
results: list[dict[str, Any] | None] = [None] * len(traces)
errors: list[tuple[str, Exception]] = []
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
futures = {pool.submit(work, trace): index for index, trace in enumerate(traces)}
completed = 0
for future in as_completed(futures):
index = futures[future]
try:
result = future.result()
except Exception as exc:
errors.append((traces[index].trace_id, exc))
continue
results[index] = result
if result_callback is not None:
result_callback(result)
completed += 1
if progress is not None:
progress(completed, len(traces), traces[index].trace_id)
if errors:
details = "; ".join(
f"{trace_id}: {type(error).__name__}: {error}"
for trace_id, error in errors
)
raise RuntimeError(f"{len(errors)} Map trace(s) failed; successful results were preserved: {details}")
return [result for result in results if result is not None]
def reduce(
self,
skill_text: str,
maps: list[dict[str, Any]],
score_rows: list[dict[str, Any]],
high_ids: list[str],
low_ids: list[str],
history: list[dict[str, Any]],
) -> dict[str, Any]:
result = self.client.json(
"Contrast exactly one successful pattern with exactly one failure pattern to identify one bounded skill-improvement gap. Return JSON only with every requested field populated; never return an empty object.",
f"""Compare the fixed relative High and Low groups. Select exactly one successful behavioral pattern from the High traces that the skill should preserve or promote, and exactly one contrasting failure pattern from the Low traces that the skill should mitigate. Then select exactly one local, generalizable gap in the current skill that connects those two patterns and has not already been addressed in Patch history. Do not select two unrelated improvements.
The selected gap must be one behavioral clarification or stopping decision, expressed in at most two sentences, not a multi-step policy. It must explicitly preserve the selected successful behavior and mitigate the selected failure behavior, remain general to the skill, and avoid turning the failure into an exhaustive or universal obligation. Do not mention benchmark-specific files or labels, invent exact counts, thresholds, quotas, or mandatory tool sequences, add significant tool work, broaden external search, or delay a required deliverable.
Infer the contrast before proposing the remedy. Compare runtime_facts for completion, duration, tool calls, and timeouts; use Map patterns and their evidence to explain the behavior. Effective score indicates relative outcome, not a root cause. Describe group tendencies only to the extent supported, cite the supporting trace and event IDs, and reflect exceptions in confidence.
If High traces show several successful behaviors, choose the one with the clearest evidence and strongest direct contrast with the selected Low failure. If Low traces show opposing failure modes, choose only the strongest failure that can be addressed by the same qualitative decision boundary as the selected success. For incomplete work and overwork, prefer prioritization, evidence-based stopping, and timely persistence over additional checking.
You must still select one success, one failure, and one gap, each with a non-empty description. Each pattern must cite evidence from its corresponding group. If contrast is weak, use evidence_type=weak_contrast_fallback and confidence=low, and choose the most conservative supported pair and clarification.
Return every field in this exact shape: {{"successful_pattern":{{"description":"one High-group behavior to preserve or promote","evidence_ids":["trace_id:E###"]}},"failure_pattern":{{"description":"one contrasting Low-group behavior to mitigate","evidence_ids":["trace_id:E###"]}},"contrast":"direct relationship between the selected success and failure","root_cause":"skill-level cause","source":"compared evidence","skill_mitigatable":true,"confidence":"low|medium|high","selected_gap":{{"description":"one supported behavioral clarification","success_behavior_to_preserve":"the selected successful behavior","failure_behavior_to_mitigate":"the selected failure behavior","target_heading":"existing skill heading","evidence_type":"contrast type","confidence":"low|medium|high","evidence_ids":["trace_id:E###"],"reason":"why this one gap preserves the success while mitigating the failure"}}}}.
High IDs: {json.dumps(high_ids)}
Low IDs: {json.dumps(low_ids)}
Scores: {json.dumps(score_rows, ensure_ascii=False)}
Map results: {json.dumps(maps, ensure_ascii=False)}
Patch history: {json.dumps(history, ensure_ascii=False)}
Current SKILL.md:
{skill_text}""",
)
for field, label in (
("successful_pattern", "successful pattern"),
("failure_pattern", "failure pattern"),
):
pattern = result.get(field)
if not isinstance(pattern, dict) or not str(pattern.get("description", "")).strip():
raise ValueError(f"reducer returned no usable {label}")
evidence_ids = pattern.get("evidence_ids")
if not isinstance(evidence_ids, list) or not evidence_ids:
raise ValueError(f"reducer returned no evidence for {label}")
gap = result.get("selected_gap")
if not isinstance(gap, dict) or not str(gap.get("description", "")).strip():
raise ValueError("reducer returned no usable contrastive gap")
for field in ("success_behavior_to_preserve", "failure_behavior_to_mitigate"):
if not str(gap.get(field, "")).strip():
raise ValueError(f"reducer selected_gap missing {field}")
return result
def generate_patches(
self, skill_path: Path, reduction: dict[str, Any], history: list[dict[str, Any]], error: str = ""
) -> list[Patch]:
text = skill_path.read_text(encoding="utf-8")
result = self.client.json(
"Generate a small ordered patch bundle for a skill document. Return JSON only.",
f"""Generate one required success-oriented patch and, only when it adds distinct value, one optional failure-oriented patch. Both patches must address the same selected_gap; do not introduce unrelated improvements.
The required promote_success patch must express the selected successful behavior as a clear, actionable recommended workflow or stopping condition in the most appropriate existing section.
The optional mitigate_failure patch is allowed only when it adds non-duplicative detection, recovery, or exception-handling guidance. Omit it when it would merely negate, restate, or cross-reference the promote_success patch. If included, it must remain useful independently rather than existing only to repeat the preferred path.
Return patches in application order: promote_success first, then optional mitigate_failure. Each patch is one contiguous text replacement. For every patch, old_text must be a non-empty, uniquely occurring verbatim substring of the original SKILL.md and patches must target non-overlapping substrings so they can be applied sequentially. new_text must replace old_text locally and preserve general applicability. Do not rewrite the whole document; keep textual growth and behavioral scope minimal.
The patch bundle must preserve efficient successful behavior. It must not add significant tool cost, introduce mandatory tool or API calls, broaden the existing external search scope, require exhaustive checking when targeted checking is sufficient, or delay creation of a required deliverable. Prefer prioritization, bounded stopping criteria, and writing or updating required outputs as soon as the core result is supported. Do not turn a trace-specific failure into an unconditional every/all/always/never/only-after rule unless the task itself inherently requires that rule.
Return {{"patches":[{{"role":"promote_success|mitigate_failure","edit_type":string,"target_heading":string,"old_text":string,"new_text":string,"evidence_ids":[string],"evidence_type":string,"confidence":string,"reason":string}}]}}. The patches array must contain one or two items and must always begin with promote_success.
Selected analysis: {json.dumps(reduction, ensure_ascii=False)}
History: {json.dumps(history, ensure_ascii=False)}
Previous application error: {error}
SKILL.md:
{text}""",
)
values = result.get("patches")
if not isinstance(values, list) or not 1 <= len(values) <= 2:
raise ValueError("patch generator must return one or two patches")
if not all(isinstance(value, dict) for value in values):
raise ValueError("every generated patch must be an object")
expected_roles = ["promote_success", "mitigate_failure"]
roles = [value.get("role") for value in values]
if roles != expected_roles[: len(values)]:
raise ValueError(
"patch roles must be promote_success followed by optional mitigate_failure"
)
patches = [Patch.from_dict(value) for value in values]
if len(patches) == 2 and patches[0].new_text.strip() == patches[1].new_text.strip():
raise ValueError("failure patch duplicates the success patch")
skill_hash = sha256_file(skill_path)
for patch in patches:
patch.skill_hash = skill_hash
return patches