Files
SkillCompiler/data/skills-bench/docs/paper-figures/scripts/01_pareto_vector.py
T
2026-09-04 14:58:42 +08:00

223 lines
9.9 KiBLFS
Python

"""01 Pareto Vector Field — cost vs pass-rate Pareto with skill-effect arrows.
Per model, two points are drawn:
no-skill point (hollow circle) → with-skills point (filled triangle)
plus the no-skills and with-skills Pareto frontier overlays.
================================================================================
FAKE DATA FORMAT
================================================================================
This script is fully synthetic. The data source is the module-level constant
`SYNTHETIC_MODELS`, a list of tuples:
(model_name: str, provider: str, cost_no_skill_cents: float,
pass_rate_no_skill: float)
For each model, `_synthetic_points()` derives the with-skills point by adding
a uniform random Δpass_rate ∈ [0.06, 0.20] and Δcost (relative) ∈ [0.04, 0.22]
on top of the no-skill values. The output table has columns:
model, model_key, provider, condition (without|withskills),
cost_cents, pass_rate
where every row is one (model, condition) point.
================================================================================
"""
from __future__ import annotations
from pathlib import Path
import matplotlib.pyplot as plt
import matplotlib.ticker as mticker
from matplotlib.lines import Line2D
from matplotlib.patches import FancyArrowPatch
import numpy as np
import pandas as pd
from utils import (
PROVIDER_COLORS,
apply_style,
model_marker_size,
save_figure,
)
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "01_pareto_vector.pdf"
CONDITION_ORDER = ["without", "withskills"]
# ── Fake-data definition ────────────────────────────────────────────────────
SYNTHETIC_MODELS: list[tuple[str, str, float, float]] = [
# (model name, provider, no-skill cost in cents, no-skill pass rate)
("Opus 4.7", "Anthropic", 95.0, 0.50),
("Sonnet 4.6", "Anthropic", 32.0, 0.46),
("Haiku 4.5", "Anthropic", 9.0, 0.36),
("GPT-5.5 Thinking", "OpenAI", 78.0, 0.46),
("GPT-5.4 Mini", "OpenAI", 14.0, 0.40),
("GPT-5.2", "OpenAI", 22.0, 0.34),
("Gemini 3.1 Pro", "Google", 60.0, 0.49),
("Gemini 3.1 Flash", "Google", 6.0, 0.42),
("DeepSeek V4", "DeepSeek", 8.0, 0.47),
("Kimi K2.6", "Kimi", 11.0, 0.36),
("GLM 4.7", "GLM", 5.0, 0.34),
("Qwen 3.6 Max", "Qwen", 4.0, 0.32),
("MiniMax M2", "Minimax", 7.0, 0.30),
]
def _synthetic_points(seed: int = 19) -> pd.DataFrame:
rng = np.random.default_rng(seed)
rows = []
for name, prov, cost_no, perf_no in SYNTHETIC_MODELS:
d_perf = rng.uniform(0.06, 0.20)
d_cost = rng.uniform(0.04, 0.22)
rows.append({"model": name, "model_key": name, "provider": prov,
"condition": "without", "cost_cents": cost_no, "pass_rate": perf_no})
rows.append({"model": name, "model_key": name, "provider": prov,
"condition": "withskills", "cost_cents": cost_no * (1 + d_cost),
"pass_rate": min(perf_no + d_perf, 0.82)})
return pd.DataFrame(rows)
def _per_model_table(df: pd.DataFrame) -> pd.DataFrame:
pivot = df.pivot_table(index=["model", "model_key", "provider"],
columns="condition",
values=["cost_cents", "pass_rate"],
aggfunc="first").reset_index()
pivot.columns = [c[0] if isinstance(c, tuple) and c[1] == "" else c for c in pivot.columns]
return pivot
def _frontier(xs: np.ndarray, ys: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
order = np.lexsort((-ys, xs))
xs, ys = xs[order], ys[order]
fx, fy, best = [], [], -np.inf
for x, y in zip(xs, ys):
if y > best + 1e-9:
fx.append(x); fy.append(y); best = y
return np.array(fx), np.array(fy)
def _plot_frontier(ax, table, condition, *, color, linewidth, linestyle,
marker, marker_size, zorder):
points = table[[("cost_cents", condition), ("pass_rate", condition)]].dropna()
if points.empty:
return
fx, fy = _frontier(points[("cost_cents", condition)].to_numpy(),
points[("pass_rate", condition)].to_numpy())
ax.plot(fx, fy, color=color, linewidth=linewidth, linestyle=linestyle, zorder=zorder)
ax.scatter(fx, fy, s=marker_size, marker=marker, facecolor="none",
edgecolor=color, linewidth=1.1, zorder=zorder + 0.25)
def _segment_point(start, end, t):
x = float(np.exp(np.log(start[0]) * (1 - t) + np.log(end[0]) * t))
y = float(start[1] * (1 - t) + end[1] * t)
return x, y
def _draw_midpoint_arrow(ax, start, end, *, color, lw, alpha,
linestyle=(0, (2.5, 2.5)), mutation_scale, zorder, span=0.16):
ax.plot([start[0], end[0]], [start[1], end[1]],
color=color, lw=lw, alpha=alpha, linestyle=linestyle, zorder=zorder)
a0 = _segment_point(start, end, 0.5 - span / 2)
a1 = _segment_point(start, end, 0.5 + span / 2)
ax.add_patch(FancyArrowPatch(a0, a1, arrowstyle="-|>", mutation_scale=mutation_scale,
color=color, lw=lw, alpha=alpha, linestyle=linestyle,
zorder=zorder + 0.15))
def main() -> None:
apply_style()
df = _synthetic_points()
table = _per_model_table(df)
fig, ax = plt.subplots(figsize=(11.2, 4.85))
ax.set_xscale("log")
for _, row in table.iterrows():
color = PROVIDER_COLORS.get(row["provider"], PROVIDER_COLORS["Unknown"])
tier = model_marker_size(row["model"])
cost_no = row[("cost_cents", "without")]
perf_no = row[("pass_rate", "without")]
cost_cu = row[("cost_cents", "withskills")]
perf_cu = row[("pass_rate", "withskills")]
if pd.notna(cost_no) and pd.notna(perf_no):
ax.scatter(cost_no, perf_no, s=tier, marker="o",
facecolor="white", edgecolor=color, linewidth=1.4, zorder=3)
if pd.notna(cost_cu) and pd.notna(perf_cu):
ax.scatter(cost_cu, perf_cu, s=int(tier * 1.15), marker="^",
facecolor=color, edgecolor="white", linewidth=0.8, zorder=4)
if pd.notna(cost_no) and pd.notna(perf_no):
_draw_midpoint_arrow(ax, (cost_no, perf_no), (cost_cu, perf_cu),
color=color, lw=1.15, alpha=0.58,
mutation_scale=9, zorder=1.8)
_plot_frontier(ax, table, "without", color="#4b5563", linewidth=2.25,
linestyle=(0, (8, 3)), marker="o", marker_size=90, zorder=2.65)
_plot_frontier(ax, table, "withskills", color="#111827", linewidth=2.45,
linestyle="-", marker="*", marker_size=135, zorder=2.8)
label_models = {"Opus 4.7", "GPT-5.5 Thinking", "Gemini 3.1 Pro", "Sonnet 4.6"}
for _, row in table.iterrows():
if row["model"] in label_models and pd.notna(row[("cost_cents", "withskills")]):
color = PROVIDER_COLORS.get(row["provider"], "#4b5563")
ax.text(row[("cost_cents", "withskills")] * 1.08,
row[("pass_rate", "withskills")] + 0.005,
row["model"], fontsize=10.6, color=color,
fontweight="bold", va="center", clip_on=False)
ax.set_xlabel("Average cost per trial (cents, log scale)")
ax.set_ylabel("Pass rate")
ax.yaxis.set_major_formatter(mticker.PercentFormatter(xmax=1.0, decimals=0))
ax.xaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v:g}"))
xs_all = df["cost_cents"].dropna()
ys_all = df["pass_rate"].dropna()
ax.set_xlim(xs_all.min() * 0.7, xs_all.max() * 1.25)
ax.set_ylim(max(0.0, ys_all.min() - 0.04), min(0.86, ys_all.max() + 0.055))
ax.grid(True, which="major", axis="both", alpha=0.45)
ax.grid(True, which="minor", axis="x", alpha=0.18)
cond_handles = [
Line2D([0], [0], marker="o", linestyle="none", markersize=7,
markerfacecolor="white", markeredgecolor="#374151",
markeredgewidth=1.2, label="No skills"),
Line2D([0], [0], marker="^", linestyle="none", markersize=8,
markerfacecolor="#475569", markeredgecolor="white",
markeredgewidth=0.8, label="With skills"),
Line2D([0], [0], color="#6b7280", linewidth=1.45,
linestyle=(0, (8, 3)), label="No-skills Pareto"),
Line2D([0], [0], color="#111827", linewidth=2.1, label="With-skills Pareto"),
]
leg_cond = ax.legend(handles=cond_handles, loc="lower right",
bbox_to_anchor=(0.995, 0.005),
fontsize=10.2, frameon=True, framealpha=0.95,
title="Condition", title_fontsize=10.8,
borderaxespad=0.5, borderpad=0.45,
handletextpad=0.55, labelspacing=0.34)
leg_cond.get_frame().set_edgecolor("#e5e7eb")
ax.add_artist(leg_cond)
providers = sorted(df["provider"].dropna().unique().tolist())
prov_handles = [
Line2D([0], [0], marker="o", linestyle="none", markersize=7,
markerfacecolor=PROVIDER_COLORS.get(p, "#4b5563"),
markeredgecolor="white", markeredgewidth=0.6, label=p)
for p in providers
]
leg_prov = ax.legend(handles=prov_handles, loc="lower right",
bbox_to_anchor=(0.820, 0.005),
fontsize=10.2, ncol=3, title="Provider",
title_fontsize=10.8, frameon=True, framealpha=0.95,
borderaxespad=0.5, borderpad=0.45,
handletextpad=0.45, columnspacing=0.75, labelspacing=0.34)
leg_prov.get_frame().set_edgecolor("#e5e7eb")
fig.subplots_adjust(top=0.99, bottom=0.17, left=0.085, right=0.985)
save_figure(fig, OUTPUT_PATH)
print(f"[01] wrote {OUTPUT_PATH}")
if __name__ == "__main__":
main()