"""09 Domain failure panels (appendix figure). Two-panel: A) failure burden (left) and B) failure composition (right) across 11 SkillsBench task domains. Sorted by total failure count desc. ================================================================================ FAKE DATA FORMAT ================================================================================ Module-level constants: DOMAIN_PALETTE: dict[str, hex_color] — 11 domains with full display names DOMAIN_PANEL_LABELS: dict[str, str] — friendlier y-axis labels FAILURE_COLS: list[str] — 6 failure category keys `synthetic_domain_data()` returns a DataFrame with one row per domain: domain: str — display name (key of DOMAIN_PALETTE) domain_label: str — same as domain : int — count of failures in that category (sums to per-domain failure_total) domain_total: int — failure_total * 4 (placeholder configuration count) failure_total: int — sum of the 6 failure-category columns Per-domain totals are linspace(900, 280) across the 11 domains; per-category weights are drawn from Dirichlet([3.5, 2.0, 1.6, 0.9, 0.7, 0.5]) so the Task-Reasoning category dominates each bar. ================================================================================ """ from __future__ import annotations from pathlib import Path import matplotlib.pyplot as plt import matplotlib.ticker as mtick import numpy as np import pandas as pd from utils import ( FAILURE_COLORS, FAILURE_LABELS, apply_style, save_figure, ) OUTPUT_PATH = Path(__file__).resolve().parent.parent / "figures" / "09_domain_failure_panels.pdf" FAILURE_COLS = list(FAILURE_LABELS.keys()) # Shared 11-domain palette mirroring script 04_task_heterogeneity. DOMAIN_PALETTE = { "Office / Document Automation": "#2563eb", "Software Engineering": "#059669", "Web / Visualization": "#d97706", "Security / Auditing / Fuzzing": "#dc2626", "Manufacturing / Lab Tables": "#7c3aed", "Business Automation / API": "#0891b2", "Graph / Dialog Parsing": "#64748b", "Scientific Computing": "#db2777", "Optimization / Planning / Control": "#ea580c", "Video / Audio / OCR / Multimodal": "#16a34a", "Retrieval / Analysis / Reporting": "#4f46e5", } # Friendlier labels for the y-axis (mirrors the legacy script). DOMAIN_PANEL_LABELS = { "Office / Document Automation": "Excel / Office / Document Automation", "Software Engineering": "Software Engineering / Code Repair", "Web / Visualization": "Web / Visualization / Three.js", "Security / Auditing / Fuzzing": "Network / Security / Auditing / Fuzzing", "Manufacturing / Lab Tables": "Manufacturing / Lab / Industry Tables", "Business Automation / API": "Business Automation / Travel / API", } def _stacked_percent(values: np.ndarray) -> np.ndarray: arr = np.asarray(values, dtype=float) denom = arr.sum(axis=1, keepdims=True) denom[denom == 0] = 1 return arr / denom def synthetic_domain_data() -> pd.DataFrame: rng = np.random.default_rng(42) rows = [] domains = list(DOMAIN_PALETTE.keys()) base_totals = np.linspace(900, 280, len(domains)) for d, total in zip(domains, base_totals): weights = rng.dirichlet([3.5, 2.0, 1.6, 0.9, 0.7, 0.5]) counts = (weights * total).round().astype(int) rows.append({ "domain": d, "domain_label": d, **{c: int(v) for c, v in zip(FAILURE_COLS, counts)}, "domain_total": int(total) * 4, "failure_total": int(counts.sum()), }) df = pd.DataFrame(rows).sort_values("failure_total", ascending=False).reset_index(drop=True) df["panel_label"] = df["domain_label"].replace(DOMAIN_PANEL_LABELS) return df def plot_panels(df: pd.DataFrame) -> plt.Figure: labels = df["panel_label"].tolist() totals = df["failure_total"].to_numpy(dtype=float) counts = df[FAILURE_COLS].to_numpy(dtype=float) pct = _stacked_percent(counts) * 100.0 y = np.arange(len(df)) fig, (ax_total, ax_comp) = plt.subplots( 1, 2, figsize=(13.5, 5.6), sharey=True, gridspec_kw={"width_ratios": [1.0, 1.35], "wspace": 0.05}, ) gradient = plt.colormaps["Blues"](np.linspace(0.92, 0.42, len(df))) ax_total.barh(y, totals, height=0.68, color=gradient, edgecolor="none") for yi, total in zip(y, totals): ax_total.text( total + max(totals) * 0.01, yi, f"{int(total):,}", va="center", ha="left", fontsize=9, color="#333333", ) ax_total.set_yticks(y) ax_total.set_yticklabels(labels) ax_total.invert_yaxis() ax_total.set_xlim(0, max(totals) * 1.12) ax_total.set_xlabel("Total failures (summed across configurations)") ax_total.set_title("A. Failure burden by skill domain", loc="left", pad=10) ax_total.tick_params(axis="y", length=0) ax_total.xaxis.set_major_locator(mtick.MaxNLocator(integer=True, nbins=5)) ax_total.grid(axis="x", linestyle=":", alpha=0.45) ax_total.set_axisbelow(True) # Tint y-tick labels by domain palette so the two panels share a consistent # domain encoding (the horizontal bars in panel A still use the Blues # gradient, which encodes magnitude rather than identity). for tick_label, panel_label in zip(ax_total.get_yticklabels(), labels): # The palette key is the canonical domain_label, panel_label is the # potentially-renamed display string — fall back to canonical. canonical = next( (dom for dom, mapped in DOMAIN_PANEL_LABELS.items() if mapped == panel_label), panel_label, ) tick_label.set_color(DOMAIN_PALETTE.get(canonical, "#1f2937")) left = np.zeros(len(df)) for col_idx, col in enumerate(FAILURE_COLS): widths = pct[:, col_idx] ax_comp.barh( y, widths, left=left, height=0.68, color=FAILURE_COLORS[col], edgecolor="none", label=FAILURE_LABELS[col], ) for yi, width, start in zip(y, widths, left): if width >= 6: # Authentication grey is light enough that white text disappears. text_color = "white" if col != "authentication_access" else "#1a1a1a" ax_comp.text( start + width / 2, yi, f"{width:.0f}%", va="center", ha="center", fontsize=8.5, color=text_color, ) left += widths ax_comp.set_xlim(0, 100) ax_comp.set_xticks([0, 25, 50, 75, 100]) ax_comp.xaxis.set_major_formatter(mtick.PercentFormatter(decimals=0)) ax_comp.set_xlabel("Failure composition (% of domain total)") ax_comp.set_title("B. Failure composition by skill domain", loc="left", pad=10) ax_comp.tick_params(axis="y", length=0) ax_comp.grid(axis="x", linestyle=":", alpha=0.45) ax_comp.set_axisbelow(True) handles, legend_labels = ax_comp.get_legend_handles_labels() legend = fig.legend( handles, legend_labels, loc="lower center", ncol=6, frameon=False, bbox_to_anchor=(0.5, -0.02), fontsize=9.5, handlelength=1.2, handleheight=1.0, columnspacing=1.6, ) title = fig.suptitle( "SkillsBench: failure burden and composition across 11 skill domains", y=0.975, fontsize=13.5, fontweight="semibold", x=0.05, ha="left", ) fig.subplots_adjust(left=0.23, right=0.985, top=0.82, bottom=0.18, wspace=0.06) fig._legend_artist = legend fig._title_artist = title return fig def main() -> None: apply_style() plt.rcParams.update({ "font.size": 10, "axes.titlesize": 12, "axes.titleweight": "semibold", "axes.labelsize": 10, "xtick.labelsize": 9, "ytick.labelsize": 9, }) df = synthetic_domain_data() fig = plot_panels(df) save_figure(fig, OUTPUT_PATH, bbox_extra_artists=(fig._legend_artist, fig._title_artist)) print(f"[09] wrote {OUTPUT_PATH}") if __name__ == "__main__": main()