"""06 Harness parity — 1×4 grid, one model per sub-panel. Each sub-panel: x = 4 harnesses (Claude Code / Codex / Gemini CLI / OpenCode), y = pass rate. Three lines per sub-panel — one per condition (no-skill / curated / self-gen). The native harness for each model is marked with ★. ================================================================================ FAKE DATA FORMAT ================================================================================ Module-level constants: MODELS: list[(name: str, native_harness: str)] — 4 paired models HARNESSES: list[str] — 4 harness names CONDITIONS: list[str] — ["without", "withskills", "withgenerate"] `fake_data()` returns dict keyed by (model, harness, condition) → pass_rate. Per-model base rates are stored in `base_per_model` (inside the function). The deterministic story knobs are: - Native harness for the model produces a healthy curated lift. - OpenCode produces a consistent curated lift for all families. - Codex on any model: curated ≈ no-skill (skills ignored, ~+0.02). - Cross-family non-OpenCode (e.g., Opus on Gemini CLI) loses ~10 pp baseline. Each per-cell value adds Gaussian noise with sigma 0.008 (no-skill) / 0.010 (curated) / 0.012 (self-gen), and is clipped to [0.10, 0.88]. ================================================================================ """ from __future__ import annotations from pathlib import Path import matplotlib.pyplot as plt import matplotlib.ticker as mticker from matplotlib.lines import Line2D import numpy as np from utils import ( CONDITION_COLORS, CONDITION_LABELS, apply_style, model_marker_size, save_figure, ) OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "06_harness_parity.pdf" # Each (model, "native_for") tag tells which harness is the model's native one; # we use it to add a small ★ marker on that x position. MODELS = [ ("Opus 4.7", "Claude Code"), ("Sonnet 4.6", "Claude Code"), ("GPT-5.5 Thinking", "Codex"), ("Gemini 3.1 Pro", "Gemini CLI"), ] HARNESSES = ["Claude Code", "Codex", "Gemini CLI", "OpenCode"] HARNESS_X = {h: i for i, h in enumerate(HARNESSES)} CONDITIONS = ["without", "withskills", "withgenerate"] def fake_data(seed: int = 71) -> dict: """Returns dict[(model, harness, condition)] -> pass rate. Story knobs: * Native harness for the model's family produces a healthy curated lift. * OpenCode produces consistent curated lift for all families. * Codex on any model: curated ≈ no-skill (skills ignored). * Cross-family native (e.g., Opus on Gemini CLI) loses ~10pp baseline. """ rng = np.random.default_rng(seed) base_per_model = { "Opus 4.7": 0.48, "Sonnet 4.6": 0.44, "GPT-5.5 Thinking": 0.46, "Gemini 3.1 Pro": 0.47, } def cross_family_penalty(model_native: str, harness: str) -> float: if harness == "OpenCode": return 0.0 if harness == model_native: return 0.0 return -0.10 def skill_lift(harness: str, native_match: bool) -> float: if harness == "Codex": return 0.02 if harness == "OpenCode": return 0.13 return 0.12 if native_match else 0.04 out: dict = {} for model, native in MODELS: b = base_per_model[model] for h in HARNESSES: penalty = cross_family_penalty(native, h) no = b + penalty + rng.normal(0, 0.008) cur = no + skill_lift(h, native_match=(h == native or h == "OpenCode")) + rng.normal(0, 0.010) sg = no - 0.005 + rng.normal(0, 0.012) out[(model, h, "without")] = float(np.clip(no, 0.10, 0.85)) out[(model, h, "withskills")] = float(np.clip(cur, 0.10, 0.88)) out[(model, h, "withgenerate")] = float(np.clip(sg, 0.10, 0.85)) return out CONDITION_MARKER = {"without": "o", "withskills": "^", "withgenerate": "s"} def _draw_panel(ax: plt.Axes, model: str, native: str, data: dict) -> None: tier_size = model_marker_size(model) xs = np.arange(len(HARNESSES)) for cond in CONDITIONS: color = CONDITION_COLORS[cond] ys = np.array([data[(model, h, cond)] for h in HARNESSES]) lw = 1.7 if cond == "withskills" else 1.2 alpha = 1.0 if cond == "withskills" else 0.78 ls = "-" if cond != "withgenerate" else (0, (3, 2)) ax.plot(xs, ys, color=color, lw=lw, linestyle=ls, alpha=alpha, zorder=2) marker = CONDITION_MARKER[cond] face = color if cond == "withskills" else "white" edge = "white" if cond == "withskills" else color ax.scatter(xs, ys, s=tier_size * 0.55, marker=marker, facecolor=face, edgecolor=edge, linewidth=1.0, alpha=alpha, zorder=3) # Native-harness star marker. nx = HARNESS_X[native] ax.annotate( "★", xy=(nx, 0.10), xytext=(nx, 0.10), ha="center", va="bottom", fontsize=10, color="#f59e0b", fontweight="bold", clip_on=False, ) ax.set_title(model, fontsize=8.8, pad=4, loc="left", color="#111827") ax.set_xticks(xs) ax.set_xticklabels([h.replace(" ", "\n") for h in HARNESSES], fontsize=6.6) ax.set_xlim(-0.45, len(HARNESSES) - 0.55) ax.set_ylim(0.10, 0.78) ax.tick_params(labelsize=7) ax.yaxis.set_major_formatter(mticker.PercentFormatter(xmax=1.0, decimals=0)) ax.grid(axis="y", alpha=0.35) ax.set_axisbelow(True) def main() -> None: apply_style() data = fake_data() fig, axes = plt.subplots(1, 4, figsize=(13.4, 3.2), sharey=True) axes_flat = axes.flatten() for ax, (model, native) in zip(axes_flat, MODELS): _draw_panel(ax, model, native, data) axes[0].set_ylabel("Pass rate") cond_handles = [ Line2D([0], [0], color=CONDITION_COLORS[c], lw=1.8, linestyle="-" if c != "withgenerate" else (0, (3, 2)), marker=CONDITION_MARKER[c], markersize=7, markerfacecolor=CONDITION_COLORS[c] if c == "withskills" else "white", markeredgecolor=CONDITION_COLORS[c] if c != "withskills" else "white", markeredgewidth=1.0, label=CONDITION_LABELS[c]) for c in CONDITIONS ] cond_handles.append( Line2D([0], [0], marker="*", linestyle="none", markersize=11, color="#f59e0b", label="Native harness for this model") ) # Suptitle removed — LaTeX caption already names the comparison. fig.legend( handles=cond_handles, loc="lower center", bbox_to_anchor=(0.5, 0.92), ncol=4, frameon=False, fontsize=8.6, handletextpad=0.5, columnspacing=1.6, ) fig.subplots_adjust(top=0.86, bottom=0.20, left=0.05, right=0.99, wspace=0.10) save_figure(fig, OUTPUT_PATH) print(f"[06] wrote {OUTPUT_PATH}") if __name__ == "__main__": main()