Files
2026-09-04 14:58:42 +08:00

204 lines
8.3 KiBLFS
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""02 Condition Dumbbell — per-model pass rate under three skill conditions.
For each of 12 OpenCode-sweep models, three vertical markers connected by a
thin line: hollow circle (no-skill), filled triangle (curated), hollow square
(self-gen). Wilson 95% CI as whiskers; "+Δpp" annotation above the curated
marker. Models sorted left-to-right by curated pass rate.
================================================================================
FAKE DATA FORMAT
================================================================================
This script is fully synthetic. The data source is the module-level constant
`MODELS`, a list of (model_key: str, display_name: str) tuples for the 12-model
OpenCode sweep.
For each (model, condition) pair, `_synthetic_table()` produces a row:
model_key: str — internal id (used for sorting / lookup)
model: str — display name
condition: str — one of {"without", "withskills", "selfgen"}
pass_rate: float — drawn from base ± offset + N(0, 0.015), clipped to [0.12, 0.85]
base decreases linearly with model rank;
offset = -0.10 (without) / +0.07 (withskills) / -0.02 (selfgen)
n_tasks_target: int — 84 (used for Wilson CI half-width)
n_repeats_target: int — 3
The Wilson 95% CI half-width is computed from (pass_rate, n=84*3) and added as
columns ci_low / ci_high.
================================================================================
"""
from __future__ import annotations
from pathlib import Path
import matplotlib.pyplot as plt
from matplotlib.lines import Line2D
from matplotlib.patches import Patch
import numpy as np
import pandas as pd
from utils import (
CONDITION_COLORS as UTL_CONDITION_COLORS,
apply_style,
pct_formatter,
save_figure,
)
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "02_condition_bars.pdf"
CONDITION_ORDER = ["without", "withskills", "selfgen"]
CONDITION_LABELS = {
"without": "No Skills",
"withskills": "Curated Skills",
"selfgen": "Self-Generated",
}
CONDITION_COLORS = {
"without": UTL_CONDITION_COLORS["without"],
"withskills": UTL_CONDITION_COLORS["withskills"],
"selfgen": UTL_CONDITION_COLORS["withgenerate"],
}
CONDITION_MARKER = {"without": "o", "withskills": "^", "selfgen": "s"}
# ── Fake-data definition ────────────────────────────────────────────────────
MODELS: list[tuple[str, str]] = [
("opus-4-7", "Opus 4.7"),
("sonnet-4.6", "Sonnet 4.6"),
("haiku-4.5", "Haiku 4.5"),
("gpt-5-5-thinking", "GPT-5.5 Thinking"),
("gpt-5-4-mini", "GPT-5.4 Mini"),
("gemini-3-1-pro", "Gemini 3.1 Pro"),
("gemini-3-1-flash", "Gemini 3.1 Flash"),
("deepseek-v4", "DeepSeek V4"),
("kimi-k2-6", "Kimi K2.6"),
("glm-4-7", "GLM 4.7"),
("qwen-3-6-max", "Qwen 3.6 Max"),
("minimax-m2", "MiniMax M2"),
]
def _wilson_halfwidth(p: pd.Series, n: pd.Series, z: float = 1.96) -> pd.Series:
p = pd.to_numeric(p, errors="coerce").clip(0, 1)
n = pd.to_numeric(n, errors="coerce")
denom = 1 + z**2 / n
half = (z * np.sqrt(p * (1 - p) / n + z**2 / (4 * n**2))) / denom
return half.where(n > 1, np.nan).fillna(np.nan)
def _synthetic_table() -> pd.DataFrame:
rng = np.random.default_rng(11)
rows = []
for i, (k, m) in enumerate(MODELS):
base = 0.55 - 0.02 * i
for c in CONDITION_ORDER:
offset = {"without": -0.10, "withskills": 0.07, "selfgen": -0.02}[c]
rows.append({
"model_key": k, "model": m, "condition": c,
"pass_rate": float(np.clip(base + offset + rng.normal(0, 0.015), 0.12, 0.85)),
"n_tasks_target": 84, "n_repeats_target": 3,
})
df = pd.DataFrame(rows)
n = df["n_tasks_target"] * df["n_repeats_target"]
df["ci_low"] = _wilson_halfwidth(df["pass_rate"], n)
df["ci_high"] = df["ci_low"]
return df
def _model_order(df: pd.DataFrame) -> list[tuple[str, str]]:
cur = df[df["condition"].eq("withskills")].copy()
cur = cur.sort_values(["pass_rate", "model"], ascending=[False, True], kind="stable")
return list(cur[["model_key", "model"]].itertuples(index=False, name=None))
def _model_pass(df: pd.DataFrame, key: str, cond: str) -> tuple[float, float]:
rows = df[(df["model_key"].eq(key)) & (df["condition"].eq(cond))]
if rows.empty:
return float("nan"), float("nan")
return float(rows["pass_rate"].iloc[0]), float(rows["ci_high"].iloc[0])
def main() -> None:
apply_style()
plt.rcParams.update({"axes.labelsize": 9.5, "xtick.labelsize": 7.6,
"ytick.labelsize": 8.0, "legend.fontsize": 7.8})
df = _synthetic_table()
order = _model_order(df)
pos = {k: i for i, (k, _) in enumerate(order)}
fig, ax = plt.subplots(figsize=(11.4, 5.4))
fig.subplots_adjust(left=0.075, right=0.99, top=0.86, bottom=0.28)
for key, _model in order:
x = pos[key]
p_no, ci_no = _model_pass(df, key, "without")
p_cu, ci_cu = _model_pass(df, key, "withskills")
p_sg, ci_sg = _model_pass(df, key, "selfgen")
ys = [y for y in (p_no, p_cu, p_sg) if np.isfinite(y)]
if not ys:
continue
ax.plot([x, x], [min(ys), max(ys)], color="#9ca3af",
linewidth=1.2, alpha=0.7, zorder=2)
for cond, p, ci in [("without", p_no, ci_no),
("withskills", p_cu, ci_cu),
("selfgen", p_sg, ci_sg)]:
if not np.isfinite(p):
continue
color = CONDITION_COLORS[cond]
face = color if cond == "withskills" else "white"
edge = "white" if cond == "withskills" else color
ax.scatter(x, p, s=70, marker=CONDITION_MARKER[cond],
facecolor=face, edgecolor=edge, linewidth=1.2, zorder=4)
if np.isfinite(ci):
ax.errorbar(x, p, yerr=ci, fmt="none", ecolor=color,
elinewidth=0.9, capsize=2, capthick=0.8,
alpha=0.7, zorder=3)
if np.isfinite(p_no) and np.isfinite(p_cu):
delta_pp = (p_cu - p_no) * 100
color = "#15803d" if delta_pp >= 0 else "#b91c1c"
ax.annotate(f"{delta_pp:+.0f}", xy=(x, max(p_no, p_cu)),
xytext=(0, 9), textcoords="offset points",
ha="center", va="bottom",
fontsize=6.8, fontweight="bold", color=color)
labels = [m for _, m in order]
ax.set_xticks(range(len(order)))
ax.set_xticklabels(labels, rotation=42, ha="right", rotation_mode="anchor")
ax.set_ylabel("Pass rate")
ax.set_xlabel("Model series (sorted by curated pass-rate)")
ax.yaxis.set_major_formatter(pct_formatter())
ax.set_ylim(0.10, 0.85)
ax.set_xlim(-0.6, len(order) - 0.4)
ax.grid(True, axis="y", alpha=0.45)
ax.set_axisbelow(True)
ax.set_title("OpenCode SOTA model performance by skills condition", pad=10)
cond_handles = [
Line2D([0], [0], marker=CONDITION_MARKER[c], linestyle="none", markersize=8,
markerfacecolor=CONDITION_COLORS[c] if c == "withskills" else "white",
markeredgecolor="white" if c == "withskills" else CONDITION_COLORS[c],
markeredgewidth=1.2, label=CONDITION_LABELS[c])
for c in CONDITION_ORDER
]
extra = [
Line2D([0], [0], color="#9ca3af", linewidth=1.4, label="connector"),
Line2D([0], [0], marker="|", color="#374151", linewidth=0,
markersize=7, markeredgewidth=1.2, label="Wilson 95% CI"),
Patch(facecolor="#15803d", alpha=0.85, label="Δ above curated (+pp)"),
Patch(facecolor="#b91c1c", alpha=0.85, label="Δ above curated (−pp)"),
]
legend = ax.legend(
handles=cond_handles + extra,
loc="upper center", bbox_to_anchor=(0.50, 0.99),
ncol=len(cond_handles + extra),
frameon=True, title="Encoding",
title_fontsize=8, columnspacing=1.0, handlelength=1.4, fontsize=7.4,
)
save_figure(fig, OUTPUT_PATH, bbox_extra_artists=(legend,))
print(f"[02] wrote {OUTPUT_PATH}")
if __name__ == "__main__":
main()