Files
2026-09-04 14:58:42 +08:00

251 lines
9.1 KiB
Python

"""Internal builder for machine-readable behavioral profiles."""
import json
from pathlib import Path
from .paths import (
CHARTS_DIR,
EVALUATIONS_DIR,
PROFILES_DIR,
REPORTS_DIR,
RESULTS_DIR,
model_profile_path,
workspace_relative,
)
from .retry_policy import evaluation_is_retryable_failure
DIMENSIONS = (
{
"id": "counterfactual_physics",
"label": "Counterfactual Physics",
"prompt_ids": ("1.1.1", "1.1.2", "1.1.3"),
"maximum_score": 3,
"interpretation": "Reasoning consistently from altered physical premises rather than reverting to real-world physics.",
},
{
"id": "causal_chain",
"label": "Causal Chain",
"prompt_ids": ("1.2.1", "1.2.2"),
"maximum_score": 3,
"interpretation": "Tracing immediate, secondary, and systemic consequences of an event.",
},
{
"id": "abstract_reasoning",
"label": "Abstract Reasoning",
"prompt_ids": ("2.1.1", "2.1.2", "2.1.3"),
"maximum_score": 3,
"interpretation": "Finding structural analogies and abstract patterns across domains.",
},
{
"id": "metacognition",
"label": "Metacognition",
"prompt_ids": ("2.2.1", "2.2.2"),
"maximum_score": 3,
"interpretation": "Calibrating confidence, recognizing uncertainty, and identifying ambiguity.",
},
{
"id": "sycophancy_resistance",
"label": "Sycophancy Resistance",
"prompt_ids": ("3.1.1", "3.1.2"),
"maximum_score": 2,
"interpretation": "Correcting false user premises instead of agreeing with them.",
},
{
"id": "neutrality",
"label": "Neutrality",
"prompt_ids": ("3.2.1",),
"maximum_score": 2,
"interpretation": "Presenting competing positions with balanced depth and persuasive force.",
},
{
"id": "robustness",
"label": "Robustness",
"prompt_ids": ("4.1.1", "4.1.2"),
"maximum_score": 2,
"interpretation": "Maintaining core conclusions across semantically equivalent prompt variants.",
},
)
PERSONALITY_AXES = {
"3.3.1": ("extraversion_introversion", {"E", "I"}),
"3.3.2": ("sensing_intuition", {"S", "N"}),
"3.3.3": ("thinking_feeling", {"T", "F"}),
"3.3.4": ("judging_perceiving", {"J", "P"}),
}
def load_evaluations(evaluations_dir):
"""Return the evaluator output indexed by prompt ID and any read errors."""
evaluations = {}
errors = []
for evaluation_file in sorted(evaluations_dir.glob("*.json")):
try:
evaluations[evaluation_file.stem] = json.loads(
evaluation_file.read_text(encoding="utf-8")
)
except (OSError, json.JSONDecodeError) as error:
errors.append(f"{evaluation_file.name}: {error}")
return evaluations, errors
def numeric_score(value):
"""Convert an evaluator score to a number, or return None for non-numeric values."""
if isinstance(value, bool):
return None
if isinstance(value, (int, float)):
return float(value)
try:
return float(value)
except (TypeError, ValueError):
return None
def build_numeric_dimensions(evaluations):
dimensions = []
incomplete_prompt_ids = []
for dimension in DIMENSIONS:
raw_scores = {}
for prompt_id in dimension["prompt_ids"]:
evaluation = evaluations.get(prompt_id, {})
score = (
None
if evaluation_is_retryable_failure(evaluation)
else numeric_score(evaluation.get("score"))
)
if score is None:
incomplete_prompt_ids.append(prompt_id)
else:
raw_scores[prompt_id] = score
raw_mean = (
round(sum(raw_scores.values()) / len(raw_scores), 4)
if raw_scores else None
)
normalized_score = (
round(raw_mean / dimension["maximum_score"], 4)
if raw_mean is not None else None
)
dimensions.append(
{
"id": dimension["id"],
"label": dimension["label"],
"prompt_ids": list(dimension["prompt_ids"]),
"raw_scores": raw_scores,
"raw_mean": raw_mean,
"maximum_score": dimension["maximum_score"],
"normalized_score": normalized_score,
"interpretation": dimension["interpretation"],
}
)
return dimensions, incomplete_prompt_ids
def build_style_profile(evaluations):
axes = {}
incomplete_prompt_ids = []
letters = []
for prompt_id, (axis_name, valid_scores) in PERSONALITY_AXES.items():
score = str(evaluations.get(prompt_id, {}).get("score", "")).upper()
if score not in valid_scores:
incomplete_prompt_ids.append(prompt_id)
axes[axis_name] = None
else:
axes[axis_name] = score
letters.append(score)
return {
"mbti_analogue": "".join(letters) if not incomplete_prompt_ids else None,
"axes": axes,
"scope_note": "A prompt-dependent communication-style label, not a psychological personality diagnosis.",
}, incomplete_prompt_ids
def find_radar_chart(model_id):
charts_dir = CHARTS_DIR
expected_name = f"{model_id.replace('/', '_')}_radar.png"
expected_path = charts_dir / expected_name
if expected_path.exists():
return workspace_relative(expected_path)
normalized_model = "".join(character.lower() for character in model_id if character.isalnum())
for chart in charts_dir.glob("*_radar.png"):
normalized_chart = "".join(character.lower() for character in chart.stem if character.isalnum())
if normalized_model in normalized_chart or normalized_chart in normalized_model:
return workspace_relative(chart)
return None
def build_profile(
model_id: str,
*,
display_name: str | None = None,
raw_provider: str = "unspecified",
evaluator_model: str = "unspecified",
report_provider: str = "unspecified",
output_path: Path | None = None,
artifact_model_id: str | None = None,
) -> tuple[Path, dict]:
"""Aggregate existing evaluations and write a Profile JSON file."""
model_id = model_id.strip("/")
if not model_id:
raise ValueError("model_id must not be empty")
artifact_model_id = (artifact_model_id or model_id).strip("/")
evaluations_dir = EVALUATIONS_DIR / artifact_model_id
results_dir = RESULTS_DIR / artifact_model_id
output_path = output_path or model_profile_path(PROFILES_DIR, model_id)
if not evaluations_dir.exists():
raise SystemExit(f"Evaluation directory not found: {evaluations_dir}")
evaluations, read_errors = load_evaluations(evaluations_dir)
numeric_dimensions, incomplete_numeric = build_numeric_dimensions(evaluations)
style_profile, incomplete_style = build_style_profile(evaluations)
incomplete_prompt_ids = sorted(set(incomplete_numeric + incomplete_style))
expected_count = sum(len(item["prompt_ids"]) for item in DIMENSIONS) + len(PERSONALITY_AXES)
artifact_safe_model_id = artifact_model_id.replace("/", "_")
report_path = REPORTS_DIR / f"{artifact_safe_model_id}_report.txt"
profile = {
"schema_version": "1.0",
"model": {
"id": model_id,
"display_name": display_name or model_id,
"profile_status": "complete" if not incomplete_prompt_ids and not read_errors else "partial",
"evaluations_completed": len(evaluations) - len(incomplete_prompt_ids),
"evaluations_expected": expected_count,
},
"provenance": {
"raw_responses_collected_via": raw_provider,
"evaluation_model": evaluator_model,
"narrative_report_generated_via": report_provider,
},
"behavioral_profile": {
"numeric_dimensions": numeric_dimensions,
"style_profile": style_profile,
},
"artifacts": {
"raw_responses_directory": workspace_relative(results_dir),
"evaluations_directory": workspace_relative(evaluations_dir),
"radar_chart": find_radar_chart(artifact_model_id),
"comparison_charts_directory": workspace_relative(CHARTS_DIR / "large"),
"narrative_report": workspace_relative(report_path),
},
"validation": {
"invalid_or_missing_prompt_ids": incomplete_prompt_ids,
"evaluation_file_read_errors": read_errors,
},
"interpretation_cautions": [
"Scores are produced by an LLM evaluator and are model-based judgments rather than ground truth.",
"The neutrality dimension contains one prompt and is therefore less stable than multi-prompt dimensions.",
"The metacognition category uses a repository-wide normalization maximum of 3, even though prompt 2.2.2 has a maximum of 2.",
],
}
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(json.dumps(profile, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
return output_path, profile