Files
SkillCompiler/scripts/static_compile/profile_generation/model_preference/visualize_results.py
T
2026-09-04 14:58:42 +08:00

395 lines
16 KiB
Python

import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
import numpy as np
from pathlib import Path
import json
from dotenv import load_dotenv
import os
import time
from .paths import CHARTS_DIR, EVALUATIONS_DIR, REPORTS_DIR
from .providers import chat_completion_options, client_for, parse_model_reference
# --- Configuration ---
load_dotenv()
REPORT_GENERATOR_MODEL = "opencode/deepseek-v4-flash"
REQUEST_TIMEOUT_SECONDS = 90.0
MAX_REPORT_ATTEMPTS = 2
EXPECTED_EVALUATION_COUNT = 19
FAILED_SCORES = {"error", "evaluator_error", "simulated"}
# mid = True
mid = False
if mid:
TARGET_MODELS = [
"openai/gpt-oss-20b",
"qwen/qwen-2.5-14b",
"qwen/qwen3-30b-a3b",
"meta-llama/llama-3.3-70b-instruct",
"deepseek/deepseek-r1-distill-qwen-14b",
"deepseek/deepseek-r1-distill-llama-70b",
"z-ai/glm-4-32b",
"mistralai/mistral-small-3.2-24b-instruct",
"huawei/Pangu-Pro-MoE-72B"
]
else:
TARGET_MODELS = [
"opencode/qwen3.6-plus"
# "deepseek-v4-flash",
# "openai/gpt-4o",
# "openai/gpt-5",
# "meta-llama/llama-3.1-405b-instruct",
# "anthropic/claude-opus-4.1",
# "google/gemini-2.5-pro",
# "x-ai/grok-4",
# "deepseek/deepseek-r1-0528:free",
# "huawei/Pangu-Ultra-MoE-718B",
# "qwen/qwen3-235b-a22b",
]
# Allows src/run_profile.py to select the same model in every pipeline stage.
if os.getenv("PROFILE_TARGET_MODEL"):
TARGET_MODELS = [os.environ["PROFILE_TARGET_MODEL"]]
REPORT_GENERATOR_MODEL = os.getenv("PROFILE_REPORT_MODEL", REPORT_GENERATOR_MODEL)
# This list should be kept in sync with run_evaluation.py
# TARGET_MODELS = [
# "openai/gpt-4o",
# "openai/gpt-5",
# "meta-llama/llama-3.1-405b-instruct",
# "anthropic/claude-opus-4.1",
# "google/gemini-2.5-pro",
# "x-ai/grok-4",
# "deepseek/deepseek-r1-0528:free",
# "huawei/Pangu-Ultra-MoE-718B",
# "qwen/qwen3-235b-a22b",
# "openai/gpt-oss-20b",
# "qwen/qwen-2.5-14b",
# "qwen/qwen3-30b-a3b",
# "meta-llama/llama-3.3-70b-instruct",
# "deepseek/deepseek-r1-distill-qwen-14b",
# "deepseek/deepseek-r1-distill-llama-70b",
# "z-ai/glm-4-32b"
# "mistralai/mistral-small-3.2-24b-instruct",
# "huawei/Pangu-Pro-MoE-72B"
# ]
def load_evaluation_data():
"""Loads all evaluation JSON files for the target models into a pandas DataFrame."""
evaluations_dir = EVALUATIONS_DIR
data = []
for model_name in TARGET_MODELS:
model = parse_model_reference(model_name)
model_dir = evaluations_dir / model.value
if not model_dir.exists():
print(f"Warning: Evaluation directory for {model_name} not found. Skipping.")
continue
for eval_file in model_dir.glob("*.json"):
prompt_id = eval_file.stem
with open(eval_file, 'r', encoding='utf-8') as f:
try:
eval_data = json.load(f)
row = {
'model_name': model.value,
'prompt_id': prompt_id,
'score': eval_data.get('score'),
'justification': eval_data.get('justification')
}
data.append(row)
except json.JSONDecodeError:
print(f"Warning: Could not decode JSON from {eval_file}")
return pd.DataFrame(data)
def aggregate_scores(df):
"""Aggregates the scores by model and category."""
def get_category(prompt_id):
if prompt_id.startswith('1.1'): return 'Counterfactual Physics'
if prompt_id.startswith('1.2'): return 'Causal Chain'
if prompt_id.startswith('2.1'): return 'Abstract Reasoning'
if prompt_id.startswith('2.2'): return 'Metacognition'
if prompt_id.startswith('3.1'): return 'Sycophancy'
if prompt_id.startswith('3.2'): return 'Neutrality'
if prompt_id.startswith('4.1'): return 'Robustness'
return 'Other'
# Convert score to numeric, coercing errors (like 'E', 'I', 'S', etc.) to NaN
df['score_numeric'] = pd.to_numeric(df['score'], errors='coerce')
# Assign categories based on whether the score is numeric or not
df['category'] = np.where(df['score_numeric'].notna(), df['prompt_id'].apply(get_category), 'Personality')
numeric_df = df.dropna(subset=['score_numeric'])
agg_df = numeric_df.groupby(['model_name', 'category'])['score_numeric'].mean().unstack()
# Define max scores for normalization
max_scores = {
'Counterfactual Physics': 3,
'Causal Chain': 3,
'Abstract Reasoning': 3,
'Metacognition': 3,
'Sycophancy': 2,
'Neutrality': 2,
'Robustness': 2
}
for category, max_score in max_scores.items():
if category in agg_df.columns:
# Normalize the score to be between 0 and 1
agg_df[category] = agg_df[category] / max_score
return agg_df.drop(columns=['Other'], errors='ignore')
def plot_radar_chart(df, model_name, save_dir):
"""Generates and saves a radar chart for a specific model using Matplotlib."""
model_data = df.loc[model_name]
categories = list(model_data.index)
N = len(categories)
# We are going to plot the first line of the data frame.
# But we need to repeat the first value to close the circular graph:
values = model_data.values.flatten().tolist()
values += values[:1]
# What will be the angle of each axis in the plot? (we divide the plot / number of variable)
angles = [n / float(N) * 2 * np.pi for n in range(N)]
angles += angles[:1]
# Initialise the spider plot
ax = plt.subplot(111, polar=True)
# Draw one axe per variable + add labels labels yet
plt.xticks(angles[:-1], categories, color='grey', size=8)
# Draw ylabels
ax.set_rlabel_position(0)
plt.yticks([0.25,0.5,0.75], ["0.25","0.50","0.75"], color="grey", size=7)
plt.ylim(0,1)
# Plot data
ax.plot(angles, values, linewidth=1, linestyle='solid')
# Fill area
ax.fill(angles, values, 'b', alpha=0.1)
# Add a title
plt.title(f'Behavioral Fingerprint: {model_name}', size=11, y=1.1)
# Save the plot
plt.savefig(save_dir / f"{model_name.replace('/', '_')}_radar.png", dpi=300, bbox_inches='tight')
plt.close()
def plot_comparison_charts(df, save_dir):
"""Generates and saves bar charts comparing all models on each category."""
for category in df.columns:
plt.figure(figsize=(10, 6))
# Sort by the current category for better visualization
sorted_df = df[category].sort_values(ascending=False)
ax = sns.barplot(x=sorted_df.index, y=sorted_df.values, palette='viridis')
plt.title(f'Model Comparison: {category}')
plt.ylabel('Normalized Score')
plt.xlabel('Model')
plt.xticks(rotation=45, ha='right')
plt.ylim(0, 1.1)
# Add the values on top of the bars
for p in ax.patches:
ax.annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 2., p.get_height()),
ha='center', va='center', fontsize=10, color='black', xytext=(0, 5),
textcoords='offset points')
plt.tight_layout()
if mid:
plt.savefig(save_dir / 'mid' / f"{category.replace(' ', '_')}_comparison.png", dpi=300)
else:
plt.savefig(save_dir / 'large' / f"{category.replace(' ', '_')}_comparison.png", dpi=300)
plt.close()
def generate_behavioral_report(df, model_name, model_data, personality_scores, report_path):
"""Stream a qualitative behavioral report and preserve any received content."""
report_model = parse_model_reference(REPORT_GENERATOR_MODEL)
client = client_for(report_model, timeout=REQUEST_TIMEOUT_SECONDS)
if not client:
return f"This is a simulated behavioral report for {model_name} because no API key was provided."
profile_summary = f"**Behavioral Profile for: {model_name}**\n\n"
profile_summary += "**Quantitative Scores (Normalized 0-1):\n"
for category, score in model_data.items():
profile_summary += f"- {category}: {score:.2f}\n"
profile_summary += "\n**Personality Profile (MBTI Analogue):\n"
mbti_type = "".join(personality_scores)
profile_summary += f"- Type: {mbti_type}\n\n"
profile_summary += "**Evaluator's Justifications (Notable Examples):\n"
sample_justifications = df[df['model_name'] == model_name].sample(
n=min(5, len(df[df['model_name'] == model_name])), random_state=42
)
for _, row in sample_justifications.iterrows():
profile_summary += f"- For prompt {row['prompt_id']}, the evaluator noted: '{row['justification']}'\n"
report_meta_prompt = f"""
You are a senior AI research analyst. Your task is to write a concise, insightful, and well-structured "Behavioral Report" for a new language model based on a quantitative and qualitative data summary.
**Data Summary:**
{profile_summary}
**Your Task:**
Write a narrative summary of this model's behavioral fingerprint. Do not just list the scores. Synthesize the information into a cohesive analysis. Your report should include:
1. An opening statement summarizing the model's overall character.
2. A discussion of its key strengths and weaknesses, referencing the specific quantitative scores.
3. An analysis of its "personality type" and how that manifests in its behavior.
4. A concluding thought on the model's most distinctive or uncommon traits, based on the evaluator's justifications.
The report should be professional, insightful, and about 2-3 paragraphs long but not redundant.
"""
partial_path = report_path.with_suffix(report_path.suffix + ".partial")
for attempt in range(1, MAX_REPORT_ATTEMPTS + 1):
print(
f"--- Generating report for {model_name}; "
f"attempt {attempt}/{MAX_REPORT_ATTEMPTS} ---"
)
try:
chunks = []
stream = client.chat.completions.create(
model=report_model.model_id,
messages=[{"role": "user", "content": report_meta_prompt}],
stream=True,
**chat_completion_options(report_model),
)
with open(partial_path, 'w', encoding='utf-8') as output_file:
for chunk in stream:
if not chunk.choices:
continue
content = chunk.choices[0].delta.content
if content:
chunks.append(content)
output_file.write(content)
output_file.flush()
if not chunks:
raise RuntimeError("Report stream completed without any text content.")
partial_path.replace(report_path)
return "".join(chunks)
except Exception as error:
partial_text = (
partial_path.read_text(encoding='utf-8')
if partial_path.exists() else ""
)
if partial_text:
report = (
"[INCOMPLETE REPORT: the provider connection closed before "
"the response finished. The text below was received successfully.]\n\n"
+ partial_text
)
report_path.write_text(report, encoding='utf-8')
return report
if attempt == MAX_REPORT_ATTEMPTS:
return f"Error generating report for {model_name} after {attempt} attempts: {error}"
delay_seconds = 2 ** attempt
print(f"Report API error: {error}. Retrying in {delay_seconds}s...")
time.sleep(delay_seconds)
def is_successful_report(report_text):
"""Identify a completed report so later runs do not make another paid request."""
return bool(report_text.strip()) and not report_text.startswith((
"Error generating report",
"Incomplete behavioral profile",
"[INCOMPLETE REPORT:",
"This is a simulated behavioral report",
))
def main():
"""Main function to run the analysis and visualization pipeline."""
df = load_evaluation_data()
print(f"Loaded {len(df)} evaluation records.")
successful_df = df[~df['score'].astype(str).isin(FAILED_SCORES)].copy()
failed_count = len(df) - len(successful_df)
if failed_count:
print(f"Warning: Excluding {failed_count} failed evaluation records from aggregation.")
agg_df = aggregate_scores(successful_df)
print(f"Aggregated scores for {len(agg_df)} model(s).")
# Create directories for saving charts and reports
charts_dir = CHARTS_DIR
reports_dir = REPORTS_DIR
charts_dir.mkdir(parents=True, exist_ok=True)
(charts_dir / ("mid" if mid else "large")).mkdir(parents=True, exist_ok=True)
reports_dir.mkdir(parents=True, exist_ok=True)
print("\n--- Generating Radar Charts ---")
for model in agg_df.index:
plot_radar_chart(agg_df, model, charts_dir)
print("\n--- Generating Comparison Bar Charts ---")
plot_comparison_charts(agg_df, charts_dir)
print("\n--- Generating Behavioral Reports ---")
personality_df = successful_df[successful_df['prompt_id'].str.startswith('3.3')].set_index(['model_name', 'prompt_id'])['score'].unstack()
# Ensure we only generate reports for models present in the aggregated data
models_to_report = [model for model in TARGET_MODELS if model in agg_df.index]
for model_name in models_to_report:
model_quantitative_data = agg_df.loc[model_name]
successful_model_df = successful_df[successful_df['model_name'] == model_name]
report_path = reports_dir / f"{model_name.replace('/', '_')}_report.txt"
if report_path.exists() and is_successful_report(report_path.read_text(encoding='utf-8')):
report = report_path.read_text(encoding='utf-8')
print(f"Using existing successful report for {model_name}; no API call made.")
elif len(successful_model_df) < EXPECTED_EVALUATION_COUNT:
report = (
f"Incomplete behavioral profile for {model_name}. "
f"Only {len(successful_model_df)}/{EXPECTED_EVALUATION_COUNT} evaluations succeeded. "
"Failed API evaluations are excluded and must be retried before generating "
"a qualitative behavioral report."
)
print(f"Warning: {report}")
# Check if the model has personality scores before proceeding
elif model_name in personality_df.index:
model_personality_scores = personality_df.loc[model_name].sort_index()
report = generate_behavioral_report(
successful_df,
model_name,
model_quantitative_data,
model_personality_scores,
report_path,
)
else:
print(f"Warning: No personality scores found for {model_name}. Generating report without it.")
empty_personality = pd.Series(['N/A'] * 4, index=[f'3.3.{i+1}' for i in range(4)])
report = generate_behavioral_report(
successful_df,
model_name,
model_quantitative_data,
empty_personality,
report_path,
)
print(f"Saved behavioral report for {model_name}.")
# Save new reports and failed attempts. Successful reports are reused above.
with open(report_path, 'w', encoding='utf-8') as f:
f.write(report)
if __name__ == "__main__":
main()