Files
SkillCompiler/data/skills-bench/docs/paper-figures/scripts/06_harness_parity.py
T
2026-09-04 14:58:42 +08:00

185 lines
6.9 KiBLFS
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""06 Harness parity — 1×4 grid, one model per sub-panel.
Each sub-panel: x = 4 harnesses (Claude Code / Codex / Gemini CLI / OpenCode),
y = pass rate. Three lines per sub-panel — one per condition (no-skill /
curated / self-gen). The native harness for each model is marked with ★.
================================================================================
FAKE DATA FORMAT
================================================================================
Module-level constants:
MODELS: list[(name: str, native_harness: str)] — 4 paired models
HARNESSES: list[str] — 4 harness names
CONDITIONS: list[str] — ["without", "withskills", "withgenerate"]
`fake_data()` returns dict keyed by (model, harness, condition) → pass_rate.
Per-model base rates are stored in `base_per_model` (inside the function).
The deterministic story knobs are:
- Native harness for the model produces a healthy curated lift.
- OpenCode produces a consistent curated lift for all families.
- Codex on any model: curated ≈ no-skill (skills ignored, ~+0.02).
- Cross-family non-OpenCode (e.g., Opus on Gemini CLI) loses ~10 pp baseline.
Each per-cell value adds Gaussian noise with sigma 0.008 (no-skill) /
0.010 (curated) / 0.012 (self-gen), and is clipped to [0.10, 0.88].
================================================================================
"""
from __future__ import annotations
from pathlib import Path
import matplotlib.pyplot as plt
import matplotlib.ticker as mticker
from matplotlib.lines import Line2D
import numpy as np
from utils import (
CONDITION_COLORS,
CONDITION_LABELS,
apply_style,
model_marker_size,
save_figure,
)
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "06_harness_parity.pdf"
# Each (model, "native_for") tag tells which harness is the model's native one;
# we use it to add a small ★ marker on that x position.
MODELS = [
("Opus 4.7", "Claude Code"),
("Sonnet 4.6", "Claude Code"),
("GPT-5.5 Thinking", "Codex"),
("Gemini 3.1 Pro", "Gemini CLI"),
]
HARNESSES = ["Claude Code", "Codex", "Gemini CLI", "OpenCode"]
HARNESS_X = {h: i for i, h in enumerate(HARNESSES)}
CONDITIONS = ["without", "withskills", "withgenerate"]
def fake_data(seed: int = 71) -> dict:
"""Returns dict[(model, harness, condition)] -> pass rate.
Story knobs:
* Native harness for the model's family produces a healthy curated lift.
* OpenCode produces consistent curated lift for all families.
* Codex on any model: curated ≈ no-skill (skills ignored).
* Cross-family native (e.g., Opus on Gemini CLI) loses ~10pp baseline.
"""
rng = np.random.default_rng(seed)
base_per_model = {
"Opus 4.7": 0.48,
"Sonnet 4.6": 0.44,
"GPT-5.5 Thinking": 0.46,
"Gemini 3.1 Pro": 0.47,
}
def cross_family_penalty(model_native: str, harness: str) -> float:
if harness == "OpenCode":
return 0.0
if harness == model_native:
return 0.0
return -0.10
def skill_lift(harness: str, native_match: bool) -> float:
if harness == "Codex":
return 0.02
if harness == "OpenCode":
return 0.13
return 0.12 if native_match else 0.04
out: dict = {}
for model, native in MODELS:
b = base_per_model[model]
for h in HARNESSES:
penalty = cross_family_penalty(native, h)
no = b + penalty + rng.normal(0, 0.008)
cur = no + skill_lift(h, native_match=(h == native or h == "OpenCode")) + rng.normal(0, 0.010)
sg = no - 0.005 + rng.normal(0, 0.012)
out[(model, h, "without")] = float(np.clip(no, 0.10, 0.85))
out[(model, h, "withskills")] = float(np.clip(cur, 0.10, 0.88))
out[(model, h, "withgenerate")] = float(np.clip(sg, 0.10, 0.85))
return out
CONDITION_MARKER = {"without": "o", "withskills": "^", "withgenerate": "s"}
def _draw_panel(ax: plt.Axes, model: str, native: str, data: dict) -> None:
tier_size = model_marker_size(model)
xs = np.arange(len(HARNESSES))
for cond in CONDITIONS:
color = CONDITION_COLORS[cond]
ys = np.array([data[(model, h, cond)] for h in HARNESSES])
lw = 1.7 if cond == "withskills" else 1.2
alpha = 1.0 if cond == "withskills" else 0.78
ls = "-" if cond != "withgenerate" else (0, (3, 2))
ax.plot(xs, ys, color=color, lw=lw, linestyle=ls, alpha=alpha, zorder=2)
marker = CONDITION_MARKER[cond]
face = color if cond == "withskills" else "white"
edge = "white" if cond == "withskills" else color
ax.scatter(xs, ys, s=tier_size * 0.55, marker=marker,
facecolor=face, edgecolor=edge, linewidth=1.0,
alpha=alpha, zorder=3)
# Native-harness star marker.
nx = HARNESS_X[native]
ax.annotate(
"★", xy=(nx, 0.10), xytext=(nx, 0.10), ha="center", va="bottom",
fontsize=10, color="#f59e0b", fontweight="bold", clip_on=False,
)
ax.set_title(model, fontsize=8.8, pad=4, loc="left", color="#111827")
ax.set_xticks(xs)
ax.set_xticklabels([h.replace(" ", "\n") for h in HARNESSES], fontsize=6.6)
ax.set_xlim(-0.45, len(HARNESSES) - 0.55)
ax.set_ylim(0.10, 0.78)
ax.tick_params(labelsize=7)
ax.yaxis.set_major_formatter(mticker.PercentFormatter(xmax=1.0, decimals=0))
ax.grid(axis="y", alpha=0.35)
ax.set_axisbelow(True)
def main() -> None:
apply_style()
data = fake_data()
fig, axes = plt.subplots(1, 4, figsize=(13.4, 3.2), sharey=True)
axes_flat = axes.flatten()
for ax, (model, native) in zip(axes_flat, MODELS):
_draw_panel(ax, model, native, data)
axes[0].set_ylabel("Pass rate")
cond_handles = [
Line2D([0], [0], color=CONDITION_COLORS[c], lw=1.8,
linestyle="-" if c != "withgenerate" else (0, (3, 2)),
marker=CONDITION_MARKER[c], markersize=7,
markerfacecolor=CONDITION_COLORS[c] if c == "withskills" else "white",
markeredgecolor=CONDITION_COLORS[c] if c != "withskills" else "white",
markeredgewidth=1.0, label=CONDITION_LABELS[c])
for c in CONDITIONS
]
cond_handles.append(
Line2D([0], [0], marker="*", linestyle="none", markersize=11,
color="#f59e0b", label="Native harness for this model")
)
# Suptitle removed — LaTeX caption already names the comparison.
fig.legend(
handles=cond_handles, loc="lower center",
bbox_to_anchor=(0.5, 0.92), ncol=4, frameon=False,
fontsize=8.6, handletextpad=0.5, columnspacing=1.6,
)
fig.subplots_adjust(top=0.86, bottom=0.20, left=0.05, right=0.99,
wspace=0.10)
save_figure(fig, OUTPUT_PATH)
print(f"[06] wrote {OUTPUT_PATH}")
if __name__ == "__main__":
main()