223 lines
9.9 KiBLFS
Python
223 lines
9.9 KiBLFS
Python
"""01 Pareto Vector Field — cost vs pass-rate Pareto with skill-effect arrows.
|
|
|
|
Per model, two points are drawn:
|
|
no-skill point (hollow circle) → with-skills point (filled triangle)
|
|
plus the no-skills and with-skills Pareto frontier overlays.
|
|
|
|
================================================================================
|
|
FAKE DATA FORMAT
|
|
================================================================================
|
|
This script is fully synthetic. The data source is the module-level constant
|
|
`SYNTHETIC_MODELS`, a list of tuples:
|
|
|
|
(model_name: str, provider: str, cost_no_skill_cents: float,
|
|
pass_rate_no_skill: float)
|
|
|
|
For each model, `_synthetic_points()` derives the with-skills point by adding
|
|
a uniform random Δpass_rate ∈ [0.06, 0.20] and Δcost (relative) ∈ [0.04, 0.22]
|
|
on top of the no-skill values. The output table has columns:
|
|
model, model_key, provider, condition (without|withskills),
|
|
cost_cents, pass_rate
|
|
where every row is one (model, condition) point.
|
|
================================================================================
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import matplotlib.pyplot as plt
|
|
import matplotlib.ticker as mticker
|
|
from matplotlib.lines import Line2D
|
|
from matplotlib.patches import FancyArrowPatch
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
from utils import (
|
|
PROVIDER_COLORS,
|
|
apply_style,
|
|
model_marker_size,
|
|
save_figure,
|
|
)
|
|
|
|
|
|
OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "01_pareto_vector.pdf"
|
|
|
|
CONDITION_ORDER = ["without", "withskills"]
|
|
|
|
|
|
# ── Fake-data definition ────────────────────────────────────────────────────
|
|
SYNTHETIC_MODELS: list[tuple[str, str, float, float]] = [
|
|
# (model name, provider, no-skill cost in cents, no-skill pass rate)
|
|
("Opus 4.7", "Anthropic", 95.0, 0.50),
|
|
("Sonnet 4.6", "Anthropic", 32.0, 0.46),
|
|
("Haiku 4.5", "Anthropic", 9.0, 0.36),
|
|
("GPT-5.5 Thinking", "OpenAI", 78.0, 0.46),
|
|
("GPT-5.4 Mini", "OpenAI", 14.0, 0.40),
|
|
("GPT-5.2", "OpenAI", 22.0, 0.34),
|
|
("Gemini 3.1 Pro", "Google", 60.0, 0.49),
|
|
("Gemini 3.1 Flash", "Google", 6.0, 0.42),
|
|
("DeepSeek V4", "DeepSeek", 8.0, 0.47),
|
|
("Kimi K2.6", "Kimi", 11.0, 0.36),
|
|
("GLM 4.7", "GLM", 5.0, 0.34),
|
|
("Qwen 3.6 Max", "Qwen", 4.0, 0.32),
|
|
("MiniMax M2", "Minimax", 7.0, 0.30),
|
|
]
|
|
|
|
|
|
def _synthetic_points(seed: int = 19) -> pd.DataFrame:
|
|
rng = np.random.default_rng(seed)
|
|
rows = []
|
|
for name, prov, cost_no, perf_no in SYNTHETIC_MODELS:
|
|
d_perf = rng.uniform(0.06, 0.20)
|
|
d_cost = rng.uniform(0.04, 0.22)
|
|
rows.append({"model": name, "model_key": name, "provider": prov,
|
|
"condition": "without", "cost_cents": cost_no, "pass_rate": perf_no})
|
|
rows.append({"model": name, "model_key": name, "provider": prov,
|
|
"condition": "withskills", "cost_cents": cost_no * (1 + d_cost),
|
|
"pass_rate": min(perf_no + d_perf, 0.82)})
|
|
return pd.DataFrame(rows)
|
|
|
|
|
|
def _per_model_table(df: pd.DataFrame) -> pd.DataFrame:
|
|
pivot = df.pivot_table(index=["model", "model_key", "provider"],
|
|
columns="condition",
|
|
values=["cost_cents", "pass_rate"],
|
|
aggfunc="first").reset_index()
|
|
pivot.columns = [c[0] if isinstance(c, tuple) and c[1] == "" else c for c in pivot.columns]
|
|
return pivot
|
|
|
|
|
|
def _frontier(xs: np.ndarray, ys: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
|
|
order = np.lexsort((-ys, xs))
|
|
xs, ys = xs[order], ys[order]
|
|
fx, fy, best = [], [], -np.inf
|
|
for x, y in zip(xs, ys):
|
|
if y > best + 1e-9:
|
|
fx.append(x); fy.append(y); best = y
|
|
return np.array(fx), np.array(fy)
|
|
|
|
|
|
def _plot_frontier(ax, table, condition, *, color, linewidth, linestyle,
|
|
marker, marker_size, zorder):
|
|
points = table[[("cost_cents", condition), ("pass_rate", condition)]].dropna()
|
|
if points.empty:
|
|
return
|
|
fx, fy = _frontier(points[("cost_cents", condition)].to_numpy(),
|
|
points[("pass_rate", condition)].to_numpy())
|
|
ax.plot(fx, fy, color=color, linewidth=linewidth, linestyle=linestyle, zorder=zorder)
|
|
ax.scatter(fx, fy, s=marker_size, marker=marker, facecolor="none",
|
|
edgecolor=color, linewidth=1.1, zorder=zorder + 0.25)
|
|
|
|
|
|
def _segment_point(start, end, t):
|
|
x = float(np.exp(np.log(start[0]) * (1 - t) + np.log(end[0]) * t))
|
|
y = float(start[1] * (1 - t) + end[1] * t)
|
|
return x, y
|
|
|
|
|
|
def _draw_midpoint_arrow(ax, start, end, *, color, lw, alpha,
|
|
linestyle=(0, (2.5, 2.5)), mutation_scale, zorder, span=0.16):
|
|
ax.plot([start[0], end[0]], [start[1], end[1]],
|
|
color=color, lw=lw, alpha=alpha, linestyle=linestyle, zorder=zorder)
|
|
a0 = _segment_point(start, end, 0.5 - span / 2)
|
|
a1 = _segment_point(start, end, 0.5 + span / 2)
|
|
ax.add_patch(FancyArrowPatch(a0, a1, arrowstyle="-|>", mutation_scale=mutation_scale,
|
|
color=color, lw=lw, alpha=alpha, linestyle=linestyle,
|
|
zorder=zorder + 0.15))
|
|
|
|
|
|
def main() -> None:
|
|
apply_style()
|
|
df = _synthetic_points()
|
|
table = _per_model_table(df)
|
|
|
|
fig, ax = plt.subplots(figsize=(11.2, 4.85))
|
|
ax.set_xscale("log")
|
|
|
|
for _, row in table.iterrows():
|
|
color = PROVIDER_COLORS.get(row["provider"], PROVIDER_COLORS["Unknown"])
|
|
tier = model_marker_size(row["model"])
|
|
cost_no = row[("cost_cents", "without")]
|
|
perf_no = row[("pass_rate", "without")]
|
|
cost_cu = row[("cost_cents", "withskills")]
|
|
perf_cu = row[("pass_rate", "withskills")]
|
|
if pd.notna(cost_no) and pd.notna(perf_no):
|
|
ax.scatter(cost_no, perf_no, s=tier, marker="o",
|
|
facecolor="white", edgecolor=color, linewidth=1.4, zorder=3)
|
|
if pd.notna(cost_cu) and pd.notna(perf_cu):
|
|
ax.scatter(cost_cu, perf_cu, s=int(tier * 1.15), marker="^",
|
|
facecolor=color, edgecolor="white", linewidth=0.8, zorder=4)
|
|
if pd.notna(cost_no) and pd.notna(perf_no):
|
|
_draw_midpoint_arrow(ax, (cost_no, perf_no), (cost_cu, perf_cu),
|
|
color=color, lw=1.15, alpha=0.58,
|
|
mutation_scale=9, zorder=1.8)
|
|
|
|
_plot_frontier(ax, table, "without", color="#4b5563", linewidth=2.25,
|
|
linestyle=(0, (8, 3)), marker="o", marker_size=90, zorder=2.65)
|
|
_plot_frontier(ax, table, "withskills", color="#111827", linewidth=2.45,
|
|
linestyle="-", marker="*", marker_size=135, zorder=2.8)
|
|
|
|
label_models = {"Opus 4.7", "GPT-5.5 Thinking", "Gemini 3.1 Pro", "Sonnet 4.6"}
|
|
for _, row in table.iterrows():
|
|
if row["model"] in label_models and pd.notna(row[("cost_cents", "withskills")]):
|
|
color = PROVIDER_COLORS.get(row["provider"], "#4b5563")
|
|
ax.text(row[("cost_cents", "withskills")] * 1.08,
|
|
row[("pass_rate", "withskills")] + 0.005,
|
|
row["model"], fontsize=10.6, color=color,
|
|
fontweight="bold", va="center", clip_on=False)
|
|
|
|
ax.set_xlabel("Average cost per trial (cents, log scale)")
|
|
ax.set_ylabel("Pass rate")
|
|
ax.yaxis.set_major_formatter(mticker.PercentFormatter(xmax=1.0, decimals=0))
|
|
ax.xaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v:g}"))
|
|
xs_all = df["cost_cents"].dropna()
|
|
ys_all = df["pass_rate"].dropna()
|
|
ax.set_xlim(xs_all.min() * 0.7, xs_all.max() * 1.25)
|
|
ax.set_ylim(max(0.0, ys_all.min() - 0.04), min(0.86, ys_all.max() + 0.055))
|
|
ax.grid(True, which="major", axis="both", alpha=0.45)
|
|
ax.grid(True, which="minor", axis="x", alpha=0.18)
|
|
|
|
cond_handles = [
|
|
Line2D([0], [0], marker="o", linestyle="none", markersize=7,
|
|
markerfacecolor="white", markeredgecolor="#374151",
|
|
markeredgewidth=1.2, label="No skills"),
|
|
Line2D([0], [0], marker="^", linestyle="none", markersize=8,
|
|
markerfacecolor="#475569", markeredgecolor="white",
|
|
markeredgewidth=0.8, label="With skills"),
|
|
Line2D([0], [0], color="#6b7280", linewidth=1.45,
|
|
linestyle=(0, (8, 3)), label="No-skills Pareto"),
|
|
Line2D([0], [0], color="#111827", linewidth=2.1, label="With-skills Pareto"),
|
|
]
|
|
leg_cond = ax.legend(handles=cond_handles, loc="lower right",
|
|
bbox_to_anchor=(0.995, 0.005),
|
|
fontsize=10.2, frameon=True, framealpha=0.95,
|
|
title="Condition", title_fontsize=10.8,
|
|
borderaxespad=0.5, borderpad=0.45,
|
|
handletextpad=0.55, labelspacing=0.34)
|
|
leg_cond.get_frame().set_edgecolor("#e5e7eb")
|
|
ax.add_artist(leg_cond)
|
|
|
|
providers = sorted(df["provider"].dropna().unique().tolist())
|
|
prov_handles = [
|
|
Line2D([0], [0], marker="o", linestyle="none", markersize=7,
|
|
markerfacecolor=PROVIDER_COLORS.get(p, "#4b5563"),
|
|
markeredgecolor="white", markeredgewidth=0.6, label=p)
|
|
for p in providers
|
|
]
|
|
leg_prov = ax.legend(handles=prov_handles, loc="lower right",
|
|
bbox_to_anchor=(0.820, 0.005),
|
|
fontsize=10.2, ncol=3, title="Provider",
|
|
title_fontsize=10.8, frameon=True, framealpha=0.95,
|
|
borderaxespad=0.5, borderpad=0.45,
|
|
handletextpad=0.45, columnspacing=0.75, labelspacing=0.34)
|
|
leg_prov.get_frame().set_edgecolor("#e5e7eb")
|
|
|
|
fig.subplots_adjust(top=0.99, bottom=0.17, left=0.085, right=0.985)
|
|
save_figure(fig, OUTPUT_PATH)
|
|
print(f"[01] wrote {OUTPUT_PATH}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|