Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
@@ -0,0 +1,394 @@
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
import numpy as np
from pathlib import Path
import json
from dotenv import load_dotenv
import os
import time
from .paths import CHARTS_DIR, EVALUATIONS_DIR, REPORTS_DIR
from .providers import chat_completion_options, client_for, parse_model_reference
# --- Configuration ---
load_dotenv()
REPORT_GENERATOR_MODEL = "opencode/deepseek-v4-flash"
REQUEST_TIMEOUT_SECONDS = 90.0
MAX_REPORT_ATTEMPTS = 2
EXPECTED_EVALUATION_COUNT = 19
FAILED_SCORES = {"error", "evaluator_error", "simulated"}
# mid = True
mid = False
if mid:
TARGET_MODELS = [
"openai/gpt-oss-20b",
"qwen/qwen-2.5-14b",
"qwen/qwen3-30b-a3b",
"meta-llama/llama-3.3-70b-instruct",
"deepseek/deepseek-r1-distill-qwen-14b",
"deepseek/deepseek-r1-distill-llama-70b",
"z-ai/glm-4-32b",
"mistralai/mistral-small-3.2-24b-instruct",
"huawei/Pangu-Pro-MoE-72B"
]
else:
TARGET_MODELS = [
"opencode/qwen3.6-plus"
# "deepseek-v4-flash",
# "openai/gpt-4o",
# "openai/gpt-5",
# "meta-llama/llama-3.1-405b-instruct",
# "anthropic/claude-opus-4.1",
# "google/gemini-2.5-pro",
# "x-ai/grok-4",
# "deepseek/deepseek-r1-0528:free",
# "huawei/Pangu-Ultra-MoE-718B",
# "qwen/qwen3-235b-a22b",
]
# Allows src/run_profile.py to select the same model in every pipeline stage.
if os.getenv("PROFILE_TARGET_MODEL"):
TARGET_MODELS = [os.environ["PROFILE_TARGET_MODEL"]]
REPORT_GENERATOR_MODEL = os.getenv("PROFILE_REPORT_MODEL", REPORT_GENERATOR_MODEL)
# This list should be kept in sync with run_evaluation.py
# TARGET_MODELS = [
# "openai/gpt-4o",
# "openai/gpt-5",
# "meta-llama/llama-3.1-405b-instruct",
# "anthropic/claude-opus-4.1",
# "google/gemini-2.5-pro",
# "x-ai/grok-4",
# "deepseek/deepseek-r1-0528:free",
# "huawei/Pangu-Ultra-MoE-718B",
# "qwen/qwen3-235b-a22b",
# "openai/gpt-oss-20b",
# "qwen/qwen-2.5-14b",
# "qwen/qwen3-30b-a3b",
# "meta-llama/llama-3.3-70b-instruct",
# "deepseek/deepseek-r1-distill-qwen-14b",
# "deepseek/deepseek-r1-distill-llama-70b",
# "z-ai/glm-4-32b"
# "mistralai/mistral-small-3.2-24b-instruct",
# "huawei/Pangu-Pro-MoE-72B"
# ]
def load_evaluation_data():
"""Loads all evaluation JSON files for the target models into a pandas DataFrame."""
evaluations_dir = EVALUATIONS_DIR
data = []
for model_name in TARGET_MODELS:
model = parse_model_reference(model_name)
model_dir = evaluations_dir / model.value
if not model_dir.exists():
print(f"Warning: Evaluation directory for {model_name} not found. Skipping.")
continue
for eval_file in model_dir.glob("*.json"):
prompt_id = eval_file.stem
with open(eval_file, 'r', encoding='utf-8') as f:
try:
eval_data = json.load(f)
row = {
'model_name': model.value,
'prompt_id': prompt_id,
'score': eval_data.get('score'),
'justification': eval_data.get('justification')
}
data.append(row)
except json.JSONDecodeError:
print(f"Warning: Could not decode JSON from {eval_file}")
return pd.DataFrame(data)
def aggregate_scores(df):
"""Aggregates the scores by model and category."""
def get_category(prompt_id):
if prompt_id.startswith('1.1'): return 'Counterfactual Physics'
if prompt_id.startswith('1.2'): return 'Causal Chain'
if prompt_id.startswith('2.1'): return 'Abstract Reasoning'
if prompt_id.startswith('2.2'): return 'Metacognition'
if prompt_id.startswith('3.1'): return 'Sycophancy'
if prompt_id.startswith('3.2'): return 'Neutrality'
if prompt_id.startswith('4.1'): return 'Robustness'
return 'Other'
# Convert score to numeric, coercing errors (like 'E', 'I', 'S', etc.) to NaN
df['score_numeric'] = pd.to_numeric(df['score'], errors='coerce')
# Assign categories based on whether the score is numeric or not
df['category'] = np.where(df['score_numeric'].notna(), df['prompt_id'].apply(get_category), 'Personality')
numeric_df = df.dropna(subset=['score_numeric'])
agg_df = numeric_df.groupby(['model_name', 'category'])['score_numeric'].mean().unstack()
# Define max scores for normalization
max_scores = {
'Counterfactual Physics': 3,
'Causal Chain': 3,
'Abstract Reasoning': 3,
'Metacognition': 3,
'Sycophancy': 2,
'Neutrality': 2,
'Robustness': 2
}
for category, max_score in max_scores.items():
if category in agg_df.columns:
# Normalize the score to be between 0 and 1
agg_df[category] = agg_df[category] / max_score
return agg_df.drop(columns=['Other'], errors='ignore')
def plot_radar_chart(df, model_name, save_dir):
"""Generates and saves a radar chart for a specific model using Matplotlib."""
model_data = df.loc[model_name]
categories = list(model_data.index)
N = len(categories)
# We are going to plot the first line of the data frame.
# But we need to repeat the first value to close the circular graph:
values = model_data.values.flatten().tolist()
values += values[:1]
# What will be the angle of each axis in the plot? (we divide the plot / number of variable)
angles = [n / float(N) * 2 * np.pi for n in range(N)]
angles += angles[:1]
# Initialise the spider plot
ax = plt.subplot(111, polar=True)
# Draw one axe per variable + add labels labels yet
plt.xticks(angles[:-1], categories, color='grey', size=8)
# Draw ylabels
ax.set_rlabel_position(0)
plt.yticks([0.25,0.5,0.75], ["0.25","0.50","0.75"], color="grey", size=7)
plt.ylim(0,1)
# Plot data
ax.plot(angles, values, linewidth=1, linestyle='solid')
# Fill area
ax.fill(angles, values, 'b', alpha=0.1)
# Add a title
plt.title(f'Behavioral Fingerprint: {model_name}', size=11, y=1.1)
# Save the plot
plt.savefig(save_dir / f"{model_name.replace('/', '_')}_radar.png", dpi=300, bbox_inches='tight')
plt.close()
def plot_comparison_charts(df, save_dir):
"""Generates and saves bar charts comparing all models on each category."""
for category in df.columns:
plt.figure(figsize=(10, 6))
# Sort by the current category for better visualization
sorted_df = df[category].sort_values(ascending=False)
ax = sns.barplot(x=sorted_df.index, y=sorted_df.values, palette='viridis')
plt.title(f'Model Comparison: {category}')
plt.ylabel('Normalized Score')
plt.xlabel('Model')
plt.xticks(rotation=45, ha='right')
plt.ylim(0, 1.1)
# Add the values on top of the bars
for p in ax.patches:
ax.annotate(f'{p.get_height():.2f}', (p.get_x() + p.get_width() / 2., p.get_height()),
ha='center', va='center', fontsize=10, color='black', xytext=(0, 5),
textcoords='offset points')
plt.tight_layout()
if mid:
plt.savefig(save_dir / 'mid' / f"{category.replace(' ', '_')}_comparison.png", dpi=300)
else:
plt.savefig(save_dir / 'large' / f"{category.replace(' ', '_')}_comparison.png", dpi=300)
plt.close()
def generate_behavioral_report(df, model_name, model_data, personality_scores, report_path):
"""Stream a qualitative behavioral report and preserve any received content."""
report_model = parse_model_reference(REPORT_GENERATOR_MODEL)
client = client_for(report_model, timeout=REQUEST_TIMEOUT_SECONDS)
if not client:
return f"This is a simulated behavioral report for {model_name} because no API key was provided."
profile_summary = f"**Behavioral Profile for: {model_name}**\n\n"
profile_summary += "**Quantitative Scores (Normalized 0-1):\n"
for category, score in model_data.items():
profile_summary += f"- {category}: {score:.2f}\n"
profile_summary += "\n**Personality Profile (MBTI Analogue):\n"
mbti_type = "".join(personality_scores)
profile_summary += f"- Type: {mbti_type}\n\n"
profile_summary += "**Evaluator's Justifications (Notable Examples):\n"
sample_justifications = df[df['model_name'] == model_name].sample(
n=min(5, len(df[df['model_name'] == model_name])), random_state=42
)
for _, row in sample_justifications.iterrows():
profile_summary += f"- For prompt {row['prompt_id']}, the evaluator noted: '{row['justification']}'\n"
report_meta_prompt = f"""
You are a senior AI research analyst. Your task is to write a concise, insightful, and well-structured "Behavioral Report" for a new language model based on a quantitative and qualitative data summary.
**Data Summary:**
{profile_summary}
**Your Task:**
Write a narrative summary of this model's behavioral fingerprint. Do not just list the scores. Synthesize the information into a cohesive analysis. Your report should include:
1. An opening statement summarizing the model's overall character.
2. A discussion of its key strengths and weaknesses, referencing the specific quantitative scores.
3. An analysis of its "personality type" and how that manifests in its behavior.
4. A concluding thought on the model's most distinctive or uncommon traits, based on the evaluator's justifications.
The report should be professional, insightful, and about 2-3 paragraphs long but not redundant.
"""
partial_path = report_path.with_suffix(report_path.suffix + ".partial")
for attempt in range(1, MAX_REPORT_ATTEMPTS + 1):
print(
f"--- Generating report for {model_name}; "
f"attempt {attempt}/{MAX_REPORT_ATTEMPTS} ---"
)
try:
chunks = []
stream = client.chat.completions.create(
model=report_model.model_id,
messages=[{"role": "user", "content": report_meta_prompt}],
stream=True,
**chat_completion_options(report_model),
)
with open(partial_path, 'w', encoding='utf-8') as output_file:
for chunk in stream:
if not chunk.choices:
continue
content = chunk.choices[0].delta.content
if content:
chunks.append(content)
output_file.write(content)
output_file.flush()
if not chunks:
raise RuntimeError("Report stream completed without any text content.")
partial_path.replace(report_path)
return "".join(chunks)
except Exception as error:
partial_text = (
partial_path.read_text(encoding='utf-8')
if partial_path.exists() else ""
)
if partial_text:
report = (
"[INCOMPLETE REPORT: the provider connection closed before "
"the response finished. The text below was received successfully.]\n\n"
+ partial_text
)
report_path.write_text(report, encoding='utf-8')
return report
if attempt == MAX_REPORT_ATTEMPTS:
return f"Error generating report for {model_name} after {attempt} attempts: {error}"
delay_seconds = 2 ** attempt
print(f"Report API error: {error}. Retrying in {delay_seconds}s...")
time.sleep(delay_seconds)
def is_successful_report(report_text):
"""Identify a completed report so later runs do not make another paid request."""
return bool(report_text.strip()) and not report_text.startswith((
"Error generating report",
"Incomplete behavioral profile",
"[INCOMPLETE REPORT:",
"This is a simulated behavioral report",
))
def main():
"""Main function to run the analysis and visualization pipeline."""
df = load_evaluation_data()
print(f"Loaded {len(df)} evaluation records.")
successful_df = df[~df['score'].astype(str).isin(FAILED_SCORES)].copy()
failed_count = len(df) - len(successful_df)
if failed_count:
print(f"Warning: Excluding {failed_count} failed evaluation records from aggregation.")
agg_df = aggregate_scores(successful_df)
print(f"Aggregated scores for {len(agg_df)} model(s).")
# Create directories for saving charts and reports
charts_dir = CHARTS_DIR
reports_dir = REPORTS_DIR
charts_dir.mkdir(parents=True, exist_ok=True)
(charts_dir / ("mid" if mid else "large")).mkdir(parents=True, exist_ok=True)
reports_dir.mkdir(parents=True, exist_ok=True)
print("\n--- Generating Radar Charts ---")
for model in agg_df.index:
plot_radar_chart(agg_df, model, charts_dir)
print("\n--- Generating Comparison Bar Charts ---")
plot_comparison_charts(agg_df, charts_dir)
print("\n--- Generating Behavioral Reports ---")
personality_df = successful_df[successful_df['prompt_id'].str.startswith('3.3')].set_index(['model_name', 'prompt_id'])['score'].unstack()
# Ensure we only generate reports for models present in the aggregated data
models_to_report = [model for model in TARGET_MODELS if model in agg_df.index]
for model_name in models_to_report:
model_quantitative_data = agg_df.loc[model_name]
successful_model_df = successful_df[successful_df['model_name'] == model_name]
report_path = reports_dir / f"{model_name.replace('/', '_')}_report.txt"
if report_path.exists() and is_successful_report(report_path.read_text(encoding='utf-8')):
report = report_path.read_text(encoding='utf-8')
print(f"Using existing successful report for {model_name}; no API call made.")
elif len(successful_model_df) < EXPECTED_EVALUATION_COUNT:
report = (
f"Incomplete behavioral profile for {model_name}. "
f"Only {len(successful_model_df)}/{EXPECTED_EVALUATION_COUNT} evaluations succeeded. "
"Failed API evaluations are excluded and must be retried before generating "
"a qualitative behavioral report."
)
print(f"Warning: {report}")
# Check if the model has personality scores before proceeding
elif model_name in personality_df.index:
model_personality_scores = personality_df.loc[model_name].sort_index()
report = generate_behavioral_report(
successful_df,
model_name,
model_quantitative_data,
model_personality_scores,
report_path,
)
else:
print(f"Warning: No personality scores found for {model_name}. Generating report without it.")
empty_personality = pd.Series(['N/A'] * 4, index=[f'3.3.{i+1}' for i in range(4)])
report = generate_behavioral_report(
successful_df,
model_name,
model_quantitative_data,
empty_personality,
report_path,
)
print(f"Saved behavioral report for {model_name}.")
# Save new reports and failed attempts. Successful reports are reused above.
with open(report_path, 'w', encoding='utf-8') as f:
f.write(report)
if __name__ == "__main__":
main()