204 lines
8.3 KiBLFS
Python
204 lines
8.3 KiBLFS
Python
"""02 Condition Dumbbell — per-model pass rate under three skill conditions.
|
||
|
||
For each of 12 OpenCode-sweep models, three vertical markers connected by a
|
||
thin line: hollow circle (no-skill), filled triangle (curated), hollow square
|
||
(self-gen). Wilson 95% CI as whiskers; "+Δpp" annotation above the curated
|
||
marker. Models sorted left-to-right by curated pass rate.
|
||
|
||
================================================================================
|
||
FAKE DATA FORMAT
|
||
================================================================================
|
||
This script is fully synthetic. The data source is the module-level constant
|
||
`MODELS`, a list of (model_key: str, display_name: str) tuples for the 12-model
|
||
OpenCode sweep.
|
||
|
||
For each (model, condition) pair, `_synthetic_table()` produces a row:
|
||
model_key: str — internal id (used for sorting / lookup)
|
||
model: str — display name
|
||
condition: str — one of {"without", "withskills", "selfgen"}
|
||
pass_rate: float — drawn from base ± offset + N(0, 0.015), clipped to [0.12, 0.85]
|
||
base decreases linearly with model rank;
|
||
offset = -0.10 (without) / +0.07 (withskills) / -0.02 (selfgen)
|
||
n_tasks_target: int — 84 (used for Wilson CI half-width)
|
||
n_repeats_target: int — 3
|
||
The Wilson 95% CI half-width is computed from (pass_rate, n=84*3) and added as
|
||
columns ci_low / ci_high.
|
||
================================================================================
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from pathlib import Path
|
||
|
||
import matplotlib.pyplot as plt
|
||
from matplotlib.lines import Line2D
|
||
from matplotlib.patches import Patch
|
||
import numpy as np
|
||
import pandas as pd
|
||
|
||
from utils import (
|
||
CONDITION_COLORS as UTL_CONDITION_COLORS,
|
||
apply_style,
|
||
pct_formatter,
|
||
save_figure,
|
||
)
|
||
|
||
|
||
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "02_condition_bars.pdf"
|
||
|
||
CONDITION_ORDER = ["without", "withskills", "selfgen"]
|
||
CONDITION_LABELS = {
|
||
"without": "No Skills",
|
||
"withskills": "Curated Skills",
|
||
"selfgen": "Self-Generated",
|
||
}
|
||
CONDITION_COLORS = {
|
||
"without": UTL_CONDITION_COLORS["without"],
|
||
"withskills": UTL_CONDITION_COLORS["withskills"],
|
||
"selfgen": UTL_CONDITION_COLORS["withgenerate"],
|
||
}
|
||
CONDITION_MARKER = {"without": "o", "withskills": "^", "selfgen": "s"}
|
||
|
||
|
||
# ── Fake-data definition ────────────────────────────────────────────────────
|
||
MODELS: list[tuple[str, str]] = [
|
||
("opus-4-7", "Opus 4.7"),
|
||
("sonnet-4.6", "Sonnet 4.6"),
|
||
("haiku-4.5", "Haiku 4.5"),
|
||
("gpt-5-5-thinking", "GPT-5.5 Thinking"),
|
||
("gpt-5-4-mini", "GPT-5.4 Mini"),
|
||
("gemini-3-1-pro", "Gemini 3.1 Pro"),
|
||
("gemini-3-1-flash", "Gemini 3.1 Flash"),
|
||
("deepseek-v4", "DeepSeek V4"),
|
||
("kimi-k2-6", "Kimi K2.6"),
|
||
("glm-4-7", "GLM 4.7"),
|
||
("qwen-3-6-max", "Qwen 3.6 Max"),
|
||
("minimax-m2", "MiniMax M2"),
|
||
]
|
||
|
||
|
||
def _wilson_halfwidth(p: pd.Series, n: pd.Series, z: float = 1.96) -> pd.Series:
|
||
p = pd.to_numeric(p, errors="coerce").clip(0, 1)
|
||
n = pd.to_numeric(n, errors="coerce")
|
||
denom = 1 + z**2 / n
|
||
half = (z * np.sqrt(p * (1 - p) / n + z**2 / (4 * n**2))) / denom
|
||
return half.where(n > 1, np.nan).fillna(np.nan)
|
||
|
||
|
||
def _synthetic_table() -> pd.DataFrame:
|
||
rng = np.random.default_rng(11)
|
||
rows = []
|
||
for i, (k, m) in enumerate(MODELS):
|
||
base = 0.55 - 0.02 * i
|
||
for c in CONDITION_ORDER:
|
||
offset = {"without": -0.10, "withskills": 0.07, "selfgen": -0.02}[c]
|
||
rows.append({
|
||
"model_key": k, "model": m, "condition": c,
|
||
"pass_rate": float(np.clip(base + offset + rng.normal(0, 0.015), 0.12, 0.85)),
|
||
"n_tasks_target": 84, "n_repeats_target": 3,
|
||
})
|
||
df = pd.DataFrame(rows)
|
||
n = df["n_tasks_target"] * df["n_repeats_target"]
|
||
df["ci_low"] = _wilson_halfwidth(df["pass_rate"], n)
|
||
df["ci_high"] = df["ci_low"]
|
||
return df
|
||
|
||
|
||
def _model_order(df: pd.DataFrame) -> list[tuple[str, str]]:
|
||
cur = df[df["condition"].eq("withskills")].copy()
|
||
cur = cur.sort_values(["pass_rate", "model"], ascending=[False, True], kind="stable")
|
||
return list(cur[["model_key", "model"]].itertuples(index=False, name=None))
|
||
|
||
|
||
def _model_pass(df: pd.DataFrame, key: str, cond: str) -> tuple[float, float]:
|
||
rows = df[(df["model_key"].eq(key)) & (df["condition"].eq(cond))]
|
||
if rows.empty:
|
||
return float("nan"), float("nan")
|
||
return float(rows["pass_rate"].iloc[0]), float(rows["ci_high"].iloc[0])
|
||
|
||
|
||
def main() -> None:
|
||
apply_style()
|
||
plt.rcParams.update({"axes.labelsize": 9.5, "xtick.labelsize": 7.6,
|
||
"ytick.labelsize": 8.0, "legend.fontsize": 7.8})
|
||
df = _synthetic_table()
|
||
order = _model_order(df)
|
||
pos = {k: i for i, (k, _) in enumerate(order)}
|
||
|
||
fig, ax = plt.subplots(figsize=(11.4, 5.4))
|
||
fig.subplots_adjust(left=0.075, right=0.99, top=0.86, bottom=0.28)
|
||
|
||
for key, _model in order:
|
||
x = pos[key]
|
||
p_no, ci_no = _model_pass(df, key, "without")
|
||
p_cu, ci_cu = _model_pass(df, key, "withskills")
|
||
p_sg, ci_sg = _model_pass(df, key, "selfgen")
|
||
ys = [y for y in (p_no, p_cu, p_sg) if np.isfinite(y)]
|
||
if not ys:
|
||
continue
|
||
ax.plot([x, x], [min(ys), max(ys)], color="#9ca3af",
|
||
linewidth=1.2, alpha=0.7, zorder=2)
|
||
|
||
for cond, p, ci in [("without", p_no, ci_no),
|
||
("withskills", p_cu, ci_cu),
|
||
("selfgen", p_sg, ci_sg)]:
|
||
if not np.isfinite(p):
|
||
continue
|
||
color = CONDITION_COLORS[cond]
|
||
face = color if cond == "withskills" else "white"
|
||
edge = "white" if cond == "withskills" else color
|
||
ax.scatter(x, p, s=70, marker=CONDITION_MARKER[cond],
|
||
facecolor=face, edgecolor=edge, linewidth=1.2, zorder=4)
|
||
if np.isfinite(ci):
|
||
ax.errorbar(x, p, yerr=ci, fmt="none", ecolor=color,
|
||
elinewidth=0.9, capsize=2, capthick=0.8,
|
||
alpha=0.7, zorder=3)
|
||
|
||
if np.isfinite(p_no) and np.isfinite(p_cu):
|
||
delta_pp = (p_cu - p_no) * 100
|
||
color = "#15803d" if delta_pp >= 0 else "#b91c1c"
|
||
ax.annotate(f"{delta_pp:+.0f}", xy=(x, max(p_no, p_cu)),
|
||
xytext=(0, 9), textcoords="offset points",
|
||
ha="center", va="bottom",
|
||
fontsize=6.8, fontweight="bold", color=color)
|
||
|
||
labels = [m for _, m in order]
|
||
ax.set_xticks(range(len(order)))
|
||
ax.set_xticklabels(labels, rotation=42, ha="right", rotation_mode="anchor")
|
||
ax.set_ylabel("Pass rate")
|
||
ax.set_xlabel("Model series (sorted by curated pass-rate)")
|
||
ax.yaxis.set_major_formatter(pct_formatter())
|
||
ax.set_ylim(0.10, 0.85)
|
||
ax.set_xlim(-0.6, len(order) - 0.4)
|
||
ax.grid(True, axis="y", alpha=0.45)
|
||
ax.set_axisbelow(True)
|
||
ax.set_title("OpenCode SOTA model performance by skills condition", pad=10)
|
||
|
||
cond_handles = [
|
||
Line2D([0], [0], marker=CONDITION_MARKER[c], linestyle="none", markersize=8,
|
||
markerfacecolor=CONDITION_COLORS[c] if c == "withskills" else "white",
|
||
markeredgecolor="white" if c == "withskills" else CONDITION_COLORS[c],
|
||
markeredgewidth=1.2, label=CONDITION_LABELS[c])
|
||
for c in CONDITION_ORDER
|
||
]
|
||
extra = [
|
||
Line2D([0], [0], color="#9ca3af", linewidth=1.4, label="connector"),
|
||
Line2D([0], [0], marker="|", color="#374151", linewidth=0,
|
||
markersize=7, markeredgewidth=1.2, label="Wilson 95% CI"),
|
||
Patch(facecolor="#15803d", alpha=0.85, label="Δ above curated (+pp)"),
|
||
Patch(facecolor="#b91c1c", alpha=0.85, label="Δ above curated (−pp)"),
|
||
]
|
||
legend = ax.legend(
|
||
handles=cond_handles + extra,
|
||
loc="upper center", bbox_to_anchor=(0.50, 0.99),
|
||
ncol=len(cond_handles + extra),
|
||
frameon=True, title="Encoding",
|
||
title_fontsize=8, columnspacing=1.0, handlelength=1.4, fontsize=7.4,
|
||
)
|
||
save_figure(fig, OUTPUT_PATH, bbox_extra_artists=(legend,))
|
||
print(f"[02] wrote {OUTPUT_PATH}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|