Initial commit
This commit is contained in:
@@ -0,0 +1,239 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from ..models import Patch, RolloutTrace
|
||||
from scripts.provider_router import resolve_model_route
|
||||
|
||||
from ..storage import package_manifest, sha256_file
|
||||
from .trace_format import compact_trace
|
||||
|
||||
|
||||
class SemanticClient:
|
||||
def __init__(
|
||||
self,
|
||||
model: str = "opencode/deepseek-v4-pro",
|
||||
timeout: int = 900,
|
||||
):
|
||||
route = resolve_model_route(model)
|
||||
assert route is not None
|
||||
base_url = route.url.removesuffix("/chat/completions").rstrip("/")
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("the openai package is required for semantic calls") from exc
|
||||
self.client = OpenAI(base_url=base_url, api_key=route.api_key, timeout=timeout, max_retries=0)
|
||||
self.model = route.reference.model_id
|
||||
self.model_reference = route.reference.value
|
||||
|
||||
def json(self, system: str, user: str, attempts: int = 1) -> dict[str, Any]:
|
||||
error: Exception | None = None
|
||||
for attempt in range(attempts):
|
||||
try:
|
||||
response = self.client.chat.completions.create(
|
||||
model=self.model,
|
||||
temperature=0,
|
||||
response_format={"type": "json_object"},
|
||||
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
stream=True,
|
||||
)
|
||||
content = "".join(
|
||||
choice.delta.content or ""
|
||||
for chunk in response
|
||||
for choice in chunk.choices
|
||||
)
|
||||
value = json.loads(content or "{}")
|
||||
if not isinstance(value, dict):
|
||||
raise ValueError("semantic response must be a JSON object")
|
||||
return value
|
||||
except Exception as exc:
|
||||
error = exc
|
||||
if attempt + 1 < attempts:
|
||||
print(
|
||||
f"[semantic] request attempt {attempt + 1}/{attempts} failed: "
|
||||
f"{type(exc).__name__}: {exc}; retrying",
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
time.sleep(1 + attempt)
|
||||
raise RuntimeError(f"semantic call failed after {attempts} attempts: {error}")
|
||||
|
||||
|
||||
class SemanticAnalyzer:
|
||||
def __init__(self, client: SemanticClient, max_parallel: int = 3):
|
||||
self.client = client
|
||||
self.max_parallel = max_parallel
|
||||
|
||||
def generate_probe(self, skill_package: Path) -> dict[str, str]:
|
||||
skill = (skill_package / "SKILL.md").read_text(encoding="utf-8")
|
||||
files = [item["path"] for item in package_manifest(skill_package)]
|
||||
result = self.client.json(
|
||||
"You create realistic probe tasks for testing an agent skill. Return JSON only.",
|
||||
f"""Create one task prompt for the skill below. The task must explicitly tell the agent to invoke this skill, exercise its core workflow, and remain solvable in an empty workspace with network access. Do not copy the skill's operation steps, create a verifier, or create a test environment. Return {{"prompt": string, "rationale": string}}.
|
||||
|
||||
Package files: {json.dumps(files, ensure_ascii=False)}
|
||||
SKILL.md:
|
||||
{skill}""",
|
||||
)
|
||||
prompt = result.get("prompt")
|
||||
if not isinstance(prompt, str) or not prompt.strip():
|
||||
raise ValueError("probe generator returned no prompt")
|
||||
return {"prompt": prompt.strip(), "rationale": str(result.get("rationale", ""))}
|
||||
|
||||
def map_trace(self, trace: RolloutTrace, score: float, bucket: str) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Analyze agent behavior from a scored trace. Return concise JSON only.",
|
||||
f"""Analyze this {bucket} trace (AgentRM score {score}) as an action-level workflow. Identify what the agent did, action ordering, stopping behavior, error recovery, and whether required outputs were persisted promptly. Runtime facts report observable execution only; do not infer external verification outcomes or correctness that is not visible in the trace. Use long payload details only when they are necessary to explain a behavioral effect. Return:
|
||||
{{"trace_id":"{trace.trace_id}","patterns":[{{"description":string,"condition":string,"effect":string,"recovered":boolean,"final_quality_impact":string,"evidence_ids":[string]}}]}}.
|
||||
Use E### or E###.T## event IDs as evidence. Skill invocation is evidence, not a quality gate.
|
||||
|
||||
{compact_trace(trace)}""",
|
||||
)
|
||||
result["runtime_facts"] = {
|
||||
"termination": trace.metadata.get("termination"),
|
||||
"timed_out": trace.timed_out,
|
||||
"exit_code": trace.exit_code,
|
||||
"duration_seconds": trace.metadata.get("agent_execution_seconds"),
|
||||
"tool_calls": trace.metadata.get("tool_calls"),
|
||||
"skill_invoked": trace.skill_invoked,
|
||||
"timeout_reason": trace.metadata.get("timeout_reason"),
|
||||
"error_category": trace.metadata.get("error_category"),
|
||||
"partial_trajectory": trace.metadata.get("partial_trajectory"),
|
||||
}
|
||||
result.setdefault("trace_id", trace.trace_id)
|
||||
result.setdefault("patterns", [])
|
||||
return result
|
||||
|
||||
def map_all(
|
||||
self,
|
||||
traces: list[RolloutTrace],
|
||||
scores: dict[str, float],
|
||||
high_ids: set[str],
|
||||
low_ids: set[str] | None = None,
|
||||
progress: Any | None = None,
|
||||
result_callback: Any | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
def work(trace: RolloutTrace) -> dict[str, Any]:
|
||||
bucket = "High" if trace.trace_id in high_ids else "Low" if low_ids is None or trace.trace_id in low_ids else "Neutral"
|
||||
return self.map_trace(trace, scores[trace.trace_id], bucket)
|
||||
|
||||
results: list[dict[str, Any] | None] = [None] * len(traces)
|
||||
errors: list[tuple[str, Exception]] = []
|
||||
with ThreadPoolExecutor(max_workers=self.max_parallel) as pool:
|
||||
futures = {pool.submit(work, trace): index for index, trace in enumerate(traces)}
|
||||
completed = 0
|
||||
for future in as_completed(futures):
|
||||
index = futures[future]
|
||||
try:
|
||||
result = future.result()
|
||||
except Exception as exc:
|
||||
errors.append((traces[index].trace_id, exc))
|
||||
continue
|
||||
results[index] = result
|
||||
if result_callback is not None:
|
||||
result_callback(result)
|
||||
completed += 1
|
||||
if progress is not None:
|
||||
progress(completed, len(traces), traces[index].trace_id)
|
||||
if errors:
|
||||
details = "; ".join(
|
||||
f"{trace_id}: {type(error).__name__}: {error}"
|
||||
for trace_id, error in errors
|
||||
)
|
||||
raise RuntimeError(f"{len(errors)} Map trace(s) failed; successful results were preserved: {details}")
|
||||
return [result for result in results if result is not None]
|
||||
|
||||
def reduce(
|
||||
self,
|
||||
skill_text: str,
|
||||
maps: list[dict[str, Any]],
|
||||
score_rows: list[dict[str, Any]],
|
||||
high_ids: list[str],
|
||||
low_ids: list[str],
|
||||
history: list[dict[str, Any]],
|
||||
) -> dict[str, Any]:
|
||||
result = self.client.json(
|
||||
"Contrast exactly one successful pattern with exactly one failure pattern to identify one bounded skill-improvement gap. Return JSON only with every requested field populated; never return an empty object.",
|
||||
f"""Compare the fixed relative High and Low groups. Select exactly one successful behavioral pattern from the High traces that the skill should preserve or promote, and exactly one contrasting failure pattern from the Low traces that the skill should mitigate. Then select exactly one local, generalizable gap in the current skill that connects those two patterns and has not already been addressed in Patch history. Do not select two unrelated improvements.
|
||||
|
||||
The selected gap must be one behavioral clarification or stopping decision, expressed in at most two sentences, not a multi-step policy. It must explicitly preserve the selected successful behavior and mitigate the selected failure behavior, remain general to the skill, and avoid turning the failure into an exhaustive or universal obligation. Do not mention benchmark-specific files or labels, invent exact counts, thresholds, quotas, or mandatory tool sequences, add significant tool work, broaden external search, or delay a required deliverable.
|
||||
|
||||
Infer the contrast before proposing the remedy. Compare runtime_facts for completion, duration, tool calls, and timeouts; use Map patterns and their evidence to explain the behavior. Effective score indicates relative outcome, not a root cause. Describe group tendencies only to the extent supported, cite the supporting trace and event IDs, and reflect exceptions in confidence.
|
||||
|
||||
If High traces show several successful behaviors, choose the one with the clearest evidence and strongest direct contrast with the selected Low failure. If Low traces show opposing failure modes, choose only the strongest failure that can be addressed by the same qualitative decision boundary as the selected success. For incomplete work and overwork, prefer prioritization, evidence-based stopping, and timely persistence over additional checking.
|
||||
|
||||
You must still select one success, one failure, and one gap, each with a non-empty description. Each pattern must cite evidence from its corresponding group. If contrast is weak, use evidence_type=weak_contrast_fallback and confidence=low, and choose the most conservative supported pair and clarification.
|
||||
Return every field in this exact shape: {{"successful_pattern":{{"description":"one High-group behavior to preserve or promote","evidence_ids":["trace_id:E###"]}},"failure_pattern":{{"description":"one contrasting Low-group behavior to mitigate","evidence_ids":["trace_id:E###"]}},"contrast":"direct relationship between the selected success and failure","root_cause":"skill-level cause","source":"compared evidence","skill_mitigatable":true,"confidence":"low|medium|high","selected_gap":{{"description":"one supported behavioral clarification","success_behavior_to_preserve":"the selected successful behavior","failure_behavior_to_mitigate":"the selected failure behavior","target_heading":"existing skill heading","evidence_type":"contrast type","confidence":"low|medium|high","evidence_ids":["trace_id:E###"],"reason":"why this one gap preserves the success while mitigating the failure"}}}}.
|
||||
|
||||
High IDs: {json.dumps(high_ids)}
|
||||
Low IDs: {json.dumps(low_ids)}
|
||||
Scores: {json.dumps(score_rows, ensure_ascii=False)}
|
||||
Map results: {json.dumps(maps, ensure_ascii=False)}
|
||||
Patch history: {json.dumps(history, ensure_ascii=False)}
|
||||
Current SKILL.md:
|
||||
{skill_text}""",
|
||||
)
|
||||
for field, label in (
|
||||
("successful_pattern", "successful pattern"),
|
||||
("failure_pattern", "failure pattern"),
|
||||
):
|
||||
pattern = result.get(field)
|
||||
if not isinstance(pattern, dict) or not str(pattern.get("description", "")).strip():
|
||||
raise ValueError(f"reducer returned no usable {label}")
|
||||
evidence_ids = pattern.get("evidence_ids")
|
||||
if not isinstance(evidence_ids, list) or not evidence_ids:
|
||||
raise ValueError(f"reducer returned no evidence for {label}")
|
||||
gap = result.get("selected_gap")
|
||||
if not isinstance(gap, dict) or not str(gap.get("description", "")).strip():
|
||||
raise ValueError("reducer returned no usable contrastive gap")
|
||||
for field in ("success_behavior_to_preserve", "failure_behavior_to_mitigate"):
|
||||
if not str(gap.get(field, "")).strip():
|
||||
raise ValueError(f"reducer selected_gap missing {field}")
|
||||
return result
|
||||
|
||||
def generate_patches(
|
||||
self, skill_path: Path, reduction: dict[str, Any], history: list[dict[str, Any]], error: str = ""
|
||||
) -> list[Patch]:
|
||||
text = skill_path.read_text(encoding="utf-8")
|
||||
result = self.client.json(
|
||||
"Generate a small ordered patch bundle for a skill document. Return JSON only.",
|
||||
f"""Generate one required success-oriented patch and, only when it adds distinct value, one optional failure-oriented patch. Both patches must address the same selected_gap; do not introduce unrelated improvements.
|
||||
|
||||
The required promote_success patch must express the selected successful behavior as a clear, actionable recommended workflow or stopping condition in the most appropriate existing section.
|
||||
|
||||
The optional mitigate_failure patch is allowed only when it adds non-duplicative detection, recovery, or exception-handling guidance. Omit it when it would merely negate, restate, or cross-reference the promote_success patch. If included, it must remain useful independently rather than existing only to repeat the preferred path.
|
||||
|
||||
Return patches in application order: promote_success first, then optional mitigate_failure. Each patch is one contiguous text replacement. For every patch, old_text must be a non-empty, uniquely occurring verbatim substring of the original SKILL.md and patches must target non-overlapping substrings so they can be applied sequentially. new_text must replace old_text locally and preserve general applicability. Do not rewrite the whole document; keep textual growth and behavioral scope minimal.
|
||||
|
||||
The patch bundle must preserve efficient successful behavior. It must not add significant tool cost, introduce mandatory tool or API calls, broaden the existing external search scope, require exhaustive checking when targeted checking is sufficient, or delay creation of a required deliverable. Prefer prioritization, bounded stopping criteria, and writing or updating required outputs as soon as the core result is supported. Do not turn a trace-specific failure into an unconditional every/all/always/never/only-after rule unless the task itself inherently requires that rule.
|
||||
|
||||
Return {{"patches":[{{"role":"promote_success|mitigate_failure","edit_type":string,"target_heading":string,"old_text":string,"new_text":string,"evidence_ids":[string],"evidence_type":string,"confidence":string,"reason":string}}]}}. The patches array must contain one or two items and must always begin with promote_success.
|
||||
Selected analysis: {json.dumps(reduction, ensure_ascii=False)}
|
||||
History: {json.dumps(history, ensure_ascii=False)}
|
||||
Previous application error: {error}
|
||||
SKILL.md:
|
||||
{text}""",
|
||||
)
|
||||
values = result.get("patches")
|
||||
if not isinstance(values, list) or not 1 <= len(values) <= 2:
|
||||
raise ValueError("patch generator must return one or two patches")
|
||||
if not all(isinstance(value, dict) for value in values):
|
||||
raise ValueError("every generated patch must be an object")
|
||||
expected_roles = ["promote_success", "mitigate_failure"]
|
||||
roles = [value.get("role") for value in values]
|
||||
if roles != expected_roles[: len(values)]:
|
||||
raise ValueError(
|
||||
"patch roles must be promote_success followed by optional mitigate_failure"
|
||||
)
|
||||
patches = [Patch.from_dict(value) for value in values]
|
||||
if len(patches) == 2 and patches[0].new_text.strip() == patches[1].new_text.strip():
|
||||
raise ValueError("failure patch duplicates the success patch")
|
||||
skill_hash = sha256_file(skill_path)
|
||||
for patch in patches:
|
||||
patch.skill_hash = skill_hash
|
||||
return patches
|
||||
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import Patch
|
||||
from ..storage import atomic_write_text, sha256_file
|
||||
|
||||
|
||||
def _validate_skill_text(text: str) -> None:
|
||||
if not text.strip():
|
||||
raise ValueError("SKILL.md is empty")
|
||||
if text.startswith("---"):
|
||||
end = text.find("\n---", 3)
|
||||
if end < 0:
|
||||
raise ValueError("SKILL.md has an unterminated YAML frontmatter")
|
||||
frontmatter = text[3:end].strip()
|
||||
try:
|
||||
import yaml
|
||||
|
||||
parsed = yaml.safe_load(frontmatter) if frontmatter else {}
|
||||
except Exception as exc:
|
||||
raise ValueError(f"invalid SKILL.md frontmatter: {exc}") from exc
|
||||
if parsed is not None and not isinstance(parsed, dict):
|
||||
raise ValueError("SKILL.md frontmatter must be a mapping")
|
||||
|
||||
|
||||
def apply_patches(
|
||||
current_package: Path, candidate_package: Path, patches: list[Patch]
|
||||
) -> str:
|
||||
skill = current_package / "SKILL.md"
|
||||
if not skill.is_file():
|
||||
raise ValueError(f"missing {skill}")
|
||||
if not patches:
|
||||
raise ValueError("at least one patch is required")
|
||||
actual_hash = sha256_file(skill)
|
||||
text = skill.read_text(encoding="utf-8")
|
||||
for index, patch in enumerate(patches, start=1):
|
||||
if not patch.skill_hash:
|
||||
raise ValueError(f"patch {index} missing generation-time skill_hash")
|
||||
if actual_hash != patch.skill_hash:
|
||||
raise ValueError(
|
||||
f"patch {index} was not generated from the current SKILL.md"
|
||||
)
|
||||
if not patch.old_text:
|
||||
raise ValueError(f"patch {index} old_text must not be empty")
|
||||
matches = text.count(patch.old_text)
|
||||
if matches != 1:
|
||||
raise ValueError(
|
||||
f"patch {index} old_text must match exactly once; found {matches}"
|
||||
)
|
||||
changed = text.replace(patch.old_text, patch.new_text, 1)
|
||||
if changed == text:
|
||||
raise ValueError(f"patch {index} does not change SKILL.md")
|
||||
_validate_skill_text(changed)
|
||||
text = changed
|
||||
token = uuid.uuid4().hex
|
||||
temporary = candidate_package.with_name(f".{candidate_package.name}.{token}.tmp")
|
||||
backup = candidate_package.with_name(f".{candidate_package.name}.{token}.bak")
|
||||
shutil.copytree(current_package, temporary)
|
||||
try:
|
||||
atomic_write_text(temporary / "SKILL.md", text)
|
||||
if candidate_package.exists():
|
||||
candidate_package.rename(backup)
|
||||
try:
|
||||
temporary.rename(candidate_package)
|
||||
except Exception:
|
||||
if backup.exists() and not candidate_package.exists():
|
||||
backup.rename(candidate_package)
|
||||
raise
|
||||
finally:
|
||||
if temporary.exists():
|
||||
shutil.rmtree(temporary)
|
||||
if backup.exists() and candidate_package.exists():
|
||||
shutil.rmtree(backup)
|
||||
return sha256_file(candidate_package / "SKILL.md")
|
||||
@@ -0,0 +1,21 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
|
||||
|
||||
def relative_high_low(
|
||||
score_by_id: dict[str, float], count: int = 3, seed: str | int = 0
|
||||
) -> tuple[list[str], list[str]]:
|
||||
"""稳定选择互不重叠的相对高分组和低分组。"""
|
||||
|
||||
if len(score_by_id) < count * 2:
|
||||
raise ValueError(f"need at least {count * 2} traces for disjoint High/Low groups")
|
||||
trace_ids = list(score_by_id)
|
||||
random.Random(str(seed)).shuffle(trace_ids)
|
||||
high = sorted(trace_ids, key=score_by_id.__getitem__, reverse=True)[:count]
|
||||
high_set = set(high)
|
||||
low = sorted(
|
||||
(trace_id for trace_id in trace_ids if trace_id not in high_set),
|
||||
key=score_by_id.__getitem__,
|
||||
)[:count]
|
||||
return high, low
|
||||
@@ -0,0 +1,234 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
from ..models import RolloutTrace
|
||||
|
||||
|
||||
_TOOL_CALL_RE = re.compile(r"^Tool call (?P<name>[^:\n]+):[ \t]*", re.MULTILINE)
|
||||
_TOOL_RESULT_RE = re.compile(
|
||||
r"^Tool result(?: \((?P<name>[^;\n)]+)(?:;[ \t]*(?P<status>[^)\n]+))?\))?:?[ \t]*",
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
|
||||
def _content(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return json.dumps(value, ensure_ascii=False)
|
||||
|
||||
|
||||
def opencode_events_to_state(
|
||||
lines: Iterable[str], probe: str
|
||||
) -> tuple[list[dict[str, str]], bool, int]:
|
||||
state: list[dict[str, str]] = [{"role": "user", "content": probe}]
|
||||
invoked = False
|
||||
parsed = 0
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not isinstance(event, dict):
|
||||
continue
|
||||
parsed += 1
|
||||
kind = str(event.get("type", event.get("event", ""))).lower()
|
||||
part = event.get("part") if isinstance(event.get("part"), dict) else event
|
||||
tool = part.get("tool") or part.get("name") or event.get("tool") or event.get("name")
|
||||
title = part.get("title") or event.get("title") or ""
|
||||
state_data = part.get("state") if isinstance(part.get("state"), dict) else {}
|
||||
haystack = " ".join([str(kind), str(tool or ""), str(title), _content(part)])
|
||||
if str(tool or "").lower() == "skill" or "<skill_content" in haystack.lower():
|
||||
invoked = True
|
||||
if tool or "tool" in kind:
|
||||
arguments = state_data.get("input", part.get("input", part.get("arguments", {})))
|
||||
output = state_data.get("output", part.get("output", part.get("result", "")))
|
||||
state.append({"role": "assistant", "content": f"Tool call {tool or title}: {_content(arguments)}"})
|
||||
if output not in (None, ""):
|
||||
state.append({"role": "user", "content": f"Tool result: {_content(output)}"})
|
||||
continue
|
||||
text = part.get("text", part.get("content", event.get("message", "")))
|
||||
if text not in (None, ""):
|
||||
role = str(event.get("role", part.get("role", "assistant")))
|
||||
if role not in {"assistant", "user", "system"}:
|
||||
role = "assistant"
|
||||
state.append({"role": role, "content": _content(text)})
|
||||
return state, invoked, parsed
|
||||
|
||||
|
||||
def read_event_file(path: Path, probe: str) -> tuple[list[dict[str, str]], bool, int]:
|
||||
with path.open(encoding="utf-8", errors="replace") as handle:
|
||||
return opencode_events_to_state(handle, probe)
|
||||
|
||||
|
||||
def _split_tool_results(content: str) -> tuple[str, list[tuple[str, str, str]]]:
|
||||
matches = list(_TOOL_RESULT_RE.finditer(content))
|
||||
if not matches:
|
||||
return content, []
|
||||
prefix = content[:matches[0].start()].strip()
|
||||
results = []
|
||||
for index, match in enumerate(matches):
|
||||
end = matches[index + 1].start() if index + 1 < len(matches) else len(content)
|
||||
results.append((
|
||||
(match.group("name") or "unknown").strip(),
|
||||
(match.group("status") or "unknown").strip(),
|
||||
content[match.end():end].strip(),
|
||||
))
|
||||
return prefix, results
|
||||
|
||||
|
||||
def _excerpt(content: str, limit: int) -> str:
|
||||
content = content.strip()
|
||||
if len(content) <= limit:
|
||||
return content
|
||||
marker = "\n[... content omitted ...]\n"
|
||||
if limit <= len(marker) + 2:
|
||||
return content[:limit]
|
||||
omitted = len(content) - (limit - len(marker))
|
||||
while True:
|
||||
marker = f"\n[... {omitted} chars omitted ...]\n"
|
||||
available = limit - len(marker)
|
||||
updated = len(content) - available
|
||||
if updated == omitted:
|
||||
break
|
||||
omitted = updated
|
||||
head = (available + 1) // 2
|
||||
tail = available // 2
|
||||
return content[:head] + marker + content[-tail:]
|
||||
|
||||
|
||||
def _termination(trace: RolloutTrace) -> str:
|
||||
value = trace.metadata.get("termination")
|
||||
if value not in (None, ""):
|
||||
return str(value)
|
||||
if trace.timed_out is True:
|
||||
return "timeout"
|
||||
if trace.exit_code == 0:
|
||||
return "completed"
|
||||
if trace.exit_code is not None:
|
||||
return "error"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _runtime_facts(trace: RolloutTrace) -> str:
|
||||
metadata = trace.metadata
|
||||
facts = {
|
||||
"trace_id": trace.trace_id,
|
||||
"termination": _termination(trace),
|
||||
"timed_out": trace.timed_out,
|
||||
"exit_code": trace.exit_code,
|
||||
"agent_execution_seconds": metadata.get("agent_execution_seconds"),
|
||||
"tool_calls": metadata.get("tool_calls"),
|
||||
"skill_invoked": trace.skill_invoked,
|
||||
"timeout_reason": metadata.get("timeout_reason"),
|
||||
"error_category": metadata.get("error_category"),
|
||||
"partial_trajectory": metadata.get("partial_trajectory"),
|
||||
}
|
||||
return "RUNTIME_FACTS " + json.dumps(facts, ensure_ascii=False, separators=(",", ":"))
|
||||
|
||||
|
||||
def _allocate_excerpt_budget(caps: list[int], available: int) -> list[int]:
|
||||
allocations = [0] * len(caps)
|
||||
active = [index for index, cap in enumerate(caps) if cap > 0]
|
||||
while active and available > 0:
|
||||
share = max(1, available // len(active))
|
||||
progressed = False
|
||||
for index in active.copy():
|
||||
amount = min(share, caps[index] - allocations[index], available)
|
||||
allocations[index] += amount
|
||||
available -= amount
|
||||
progressed = progressed or amount > 0
|
||||
if allocations[index] >= caps[index]:
|
||||
active.remove(index)
|
||||
if available == 0:
|
||||
break
|
||||
if not progressed:
|
||||
break
|
||||
return allocations
|
||||
|
||||
|
||||
def compact_trace(trace: RolloutTrace, total: int = 30000) -> str:
|
||||
"""Render runtime facts and a complete action ledger for one Map call."""
|
||||
entries: list[dict[str, Any]] = []
|
||||
for index, message in enumerate(trace.state):
|
||||
event_id = f"E{index:03d}"
|
||||
role = str(message.get("role", "unknown"))
|
||||
content = _content(message.get("content", ""))
|
||||
|
||||
tool_call = _TOOL_CALL_RE.match(content)
|
||||
if tool_call:
|
||||
entries.append({
|
||||
"id": f"{event_id}.T01",
|
||||
"kind": "tool_call",
|
||||
"role": role,
|
||||
"name": tool_call.group("name").strip(),
|
||||
"status": "unknown",
|
||||
"content": content[tool_call.end():].strip(),
|
||||
})
|
||||
continue
|
||||
|
||||
prefix, tool_results = _split_tool_results(content) if index > 0 else (content, [])
|
||||
if prefix:
|
||||
entries.append({
|
||||
"id": event_id,
|
||||
"kind": "message",
|
||||
"role": role,
|
||||
"content": prefix,
|
||||
})
|
||||
for tool_index, (name, status, result) in enumerate(tool_results, 1):
|
||||
entries.append({
|
||||
"id": f"{event_id}.T{tool_index:02d}",
|
||||
"kind": "tool_result",
|
||||
"role": role,
|
||||
"name": name,
|
||||
"status": status,
|
||||
"content": result,
|
||||
})
|
||||
|
||||
message_entries = [entry for entry in entries if entry["kind"] == "message"]
|
||||
first_user = next((entry for entry in message_entries if entry["role"] == "user"), None)
|
||||
final_assistant = next(
|
||||
(entry for entry in reversed(message_entries) if entry["role"] == "assistant"),
|
||||
None,
|
||||
)
|
||||
|
||||
skeletons = []
|
||||
caps = []
|
||||
for entry in entries:
|
||||
if entry["kind"] == "message":
|
||||
labels = ["message", f"role={entry['role']}"]
|
||||
if entry is first_user:
|
||||
labels.append("task")
|
||||
if entry is final_assistant:
|
||||
labels.append("final")
|
||||
skeleton = f"{entry['id']} [" + " ".join(labels) + "]"
|
||||
cap = 2500 if entry is first_user else 3000 if entry is final_assistant else 600
|
||||
else:
|
||||
name = str(entry["name"])[:80]
|
||||
status = str(entry["status"])[:40]
|
||||
skeleton = f"{entry['id']} [{entry['kind']} name={name} status={status}]"
|
||||
cap = 600
|
||||
skeletons.append(skeleton)
|
||||
caps.append(min(cap, len(str(entry["content"]))))
|
||||
|
||||
facts = _runtime_facts(trace)
|
||||
fixed_size = (
|
||||
len(facts)
|
||||
+ sum(len(skeleton) + 1 for skeleton in skeletons)
|
||||
+ sum(1 for cap in caps if cap > 0)
|
||||
)
|
||||
allocations = _allocate_excerpt_budget(caps, max(0, total - fixed_size))
|
||||
rendered = [facts]
|
||||
for entry, skeleton, allocation in zip(entries, skeletons, allocations):
|
||||
rendered.append(skeleton)
|
||||
if allocation:
|
||||
rendered.append(_excerpt(str(entry["content"]), allocation))
|
||||
return "\n".join(rendered)
|
||||
Reference in New Issue
Block a user