185 lines
6.9 KiBLFS
Python
185 lines
6.9 KiBLFS
Python
"""06 Harness parity — 1×4 grid, one model per sub-panel.
|
||
|
||
Each sub-panel: x = 4 harnesses (Claude Code / Codex / Gemini CLI / OpenCode),
|
||
y = pass rate. Three lines per sub-panel — one per condition (no-skill /
|
||
curated / self-gen). The native harness for each model is marked with ★.
|
||
|
||
================================================================================
|
||
FAKE DATA FORMAT
|
||
================================================================================
|
||
Module-level constants:
|
||
MODELS: list[(name: str, native_harness: str)] — 4 paired models
|
||
HARNESSES: list[str] — 4 harness names
|
||
CONDITIONS: list[str] — ["without", "withskills", "withgenerate"]
|
||
|
||
`fake_data()` returns dict keyed by (model, harness, condition) → pass_rate.
|
||
Per-model base rates are stored in `base_per_model` (inside the function).
|
||
The deterministic story knobs are:
|
||
- Native harness for the model produces a healthy curated lift.
|
||
- OpenCode produces a consistent curated lift for all families.
|
||
- Codex on any model: curated ≈ no-skill (skills ignored, ~+0.02).
|
||
- Cross-family non-OpenCode (e.g., Opus on Gemini CLI) loses ~10 pp baseline.
|
||
Each per-cell value adds Gaussian noise with sigma 0.008 (no-skill) /
|
||
0.010 (curated) / 0.012 (self-gen), and is clipped to [0.10, 0.88].
|
||
================================================================================
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from pathlib import Path
|
||
|
||
import matplotlib.pyplot as plt
|
||
import matplotlib.ticker as mticker
|
||
from matplotlib.lines import Line2D
|
||
import numpy as np
|
||
|
||
from utils import (
|
||
CONDITION_COLORS,
|
||
CONDITION_LABELS,
|
||
apply_style,
|
||
model_marker_size,
|
||
save_figure,
|
||
)
|
||
|
||
|
||
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "06_harness_parity.pdf"
|
||
|
||
# Each (model, "native_for") tag tells which harness is the model's native one;
|
||
# we use it to add a small ★ marker on that x position.
|
||
MODELS = [
|
||
("Opus 4.7", "Claude Code"),
|
||
("Sonnet 4.6", "Claude Code"),
|
||
("GPT-5.5 Thinking", "Codex"),
|
||
("Gemini 3.1 Pro", "Gemini CLI"),
|
||
]
|
||
|
||
HARNESSES = ["Claude Code", "Codex", "Gemini CLI", "OpenCode"]
|
||
HARNESS_X = {h: i for i, h in enumerate(HARNESSES)}
|
||
CONDITIONS = ["without", "withskills", "withgenerate"]
|
||
|
||
|
||
def fake_data(seed: int = 71) -> dict:
|
||
"""Returns dict[(model, harness, condition)] -> pass rate.
|
||
|
||
Story knobs:
|
||
* Native harness for the model's family produces a healthy curated lift.
|
||
* OpenCode produces consistent curated lift for all families.
|
||
* Codex on any model: curated ≈ no-skill (skills ignored).
|
||
* Cross-family native (e.g., Opus on Gemini CLI) loses ~10pp baseline.
|
||
"""
|
||
rng = np.random.default_rng(seed)
|
||
base_per_model = {
|
||
"Opus 4.7": 0.48,
|
||
"Sonnet 4.6": 0.44,
|
||
"GPT-5.5 Thinking": 0.46,
|
||
"Gemini 3.1 Pro": 0.47,
|
||
}
|
||
|
||
def cross_family_penalty(model_native: str, harness: str) -> float:
|
||
if harness == "OpenCode":
|
||
return 0.0
|
||
if harness == model_native:
|
||
return 0.0
|
||
return -0.10
|
||
|
||
def skill_lift(harness: str, native_match: bool) -> float:
|
||
if harness == "Codex":
|
||
return 0.02
|
||
if harness == "OpenCode":
|
||
return 0.13
|
||
return 0.12 if native_match else 0.04
|
||
|
||
out: dict = {}
|
||
for model, native in MODELS:
|
||
b = base_per_model[model]
|
||
for h in HARNESSES:
|
||
penalty = cross_family_penalty(native, h)
|
||
no = b + penalty + rng.normal(0, 0.008)
|
||
cur = no + skill_lift(h, native_match=(h == native or h == "OpenCode")) + rng.normal(0, 0.010)
|
||
sg = no - 0.005 + rng.normal(0, 0.012)
|
||
out[(model, h, "without")] = float(np.clip(no, 0.10, 0.85))
|
||
out[(model, h, "withskills")] = float(np.clip(cur, 0.10, 0.88))
|
||
out[(model, h, "withgenerate")] = float(np.clip(sg, 0.10, 0.85))
|
||
return out
|
||
|
||
|
||
CONDITION_MARKER = {"without": "o", "withskills": "^", "withgenerate": "s"}
|
||
|
||
|
||
def _draw_panel(ax: plt.Axes, model: str, native: str, data: dict) -> None:
|
||
tier_size = model_marker_size(model)
|
||
xs = np.arange(len(HARNESSES))
|
||
|
||
for cond in CONDITIONS:
|
||
color = CONDITION_COLORS[cond]
|
||
ys = np.array([data[(model, h, cond)] for h in HARNESSES])
|
||
lw = 1.7 if cond == "withskills" else 1.2
|
||
alpha = 1.0 if cond == "withskills" else 0.78
|
||
ls = "-" if cond != "withgenerate" else (0, (3, 2))
|
||
ax.plot(xs, ys, color=color, lw=lw, linestyle=ls, alpha=alpha, zorder=2)
|
||
marker = CONDITION_MARKER[cond]
|
||
face = color if cond == "withskills" else "white"
|
||
edge = "white" if cond == "withskills" else color
|
||
ax.scatter(xs, ys, s=tier_size * 0.55, marker=marker,
|
||
facecolor=face, edgecolor=edge, linewidth=1.0,
|
||
alpha=alpha, zorder=3)
|
||
|
||
# Native-harness star marker.
|
||
nx = HARNESS_X[native]
|
||
ax.annotate(
|
||
"★", xy=(nx, 0.10), xytext=(nx, 0.10), ha="center", va="bottom",
|
||
fontsize=10, color="#f59e0b", fontweight="bold", clip_on=False,
|
||
)
|
||
|
||
ax.set_title(model, fontsize=8.8, pad=4, loc="left", color="#111827")
|
||
ax.set_xticks(xs)
|
||
ax.set_xticklabels([h.replace(" ", "\n") for h in HARNESSES], fontsize=6.6)
|
||
ax.set_xlim(-0.45, len(HARNESSES) - 0.55)
|
||
ax.set_ylim(0.10, 0.78)
|
||
ax.tick_params(labelsize=7)
|
||
ax.yaxis.set_major_formatter(mticker.PercentFormatter(xmax=1.0, decimals=0))
|
||
ax.grid(axis="y", alpha=0.35)
|
||
ax.set_axisbelow(True)
|
||
|
||
|
||
def main() -> None:
|
||
apply_style()
|
||
data = fake_data()
|
||
|
||
fig, axes = plt.subplots(1, 4, figsize=(13.4, 3.2), sharey=True)
|
||
axes_flat = axes.flatten()
|
||
|
||
for ax, (model, native) in zip(axes_flat, MODELS):
|
||
_draw_panel(ax, model, native, data)
|
||
|
||
axes[0].set_ylabel("Pass rate")
|
||
|
||
cond_handles = [
|
||
Line2D([0], [0], color=CONDITION_COLORS[c], lw=1.8,
|
||
linestyle="-" if c != "withgenerate" else (0, (3, 2)),
|
||
marker=CONDITION_MARKER[c], markersize=7,
|
||
markerfacecolor=CONDITION_COLORS[c] if c == "withskills" else "white",
|
||
markeredgecolor=CONDITION_COLORS[c] if c != "withskills" else "white",
|
||
markeredgewidth=1.0, label=CONDITION_LABELS[c])
|
||
for c in CONDITIONS
|
||
]
|
||
cond_handles.append(
|
||
Line2D([0], [0], marker="*", linestyle="none", markersize=11,
|
||
color="#f59e0b", label="Native harness for this model")
|
||
)
|
||
# Suptitle removed — LaTeX caption already names the comparison.
|
||
fig.legend(
|
||
handles=cond_handles, loc="lower center",
|
||
bbox_to_anchor=(0.5, 0.92), ncol=4, frameon=False,
|
||
fontsize=8.6, handletextpad=0.5, columnspacing=1.6,
|
||
)
|
||
|
||
fig.subplots_adjust(top=0.86, bottom=0.20, left=0.05, right=0.99,
|
||
wspace=0.10)
|
||
save_figure(fig, OUTPUT_PATH)
|
||
print(f"[06] wrote {OUTPUT_PATH}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|