251 lines
9.1 KiB
Python
251 lines
9.1 KiB
Python
"""Internal builder for machine-readable behavioral profiles."""
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
from .paths import (
|
|
CHARTS_DIR,
|
|
EVALUATIONS_DIR,
|
|
PROFILES_DIR,
|
|
REPORTS_DIR,
|
|
RESULTS_DIR,
|
|
model_profile_path,
|
|
workspace_relative,
|
|
)
|
|
from .retry_policy import evaluation_is_retryable_failure
|
|
|
|
|
|
DIMENSIONS = (
|
|
{
|
|
"id": "counterfactual_physics",
|
|
"label": "Counterfactual Physics",
|
|
"prompt_ids": ("1.1.1", "1.1.2", "1.1.3"),
|
|
"maximum_score": 3,
|
|
"interpretation": "Reasoning consistently from altered physical premises rather than reverting to real-world physics.",
|
|
},
|
|
{
|
|
"id": "causal_chain",
|
|
"label": "Causal Chain",
|
|
"prompt_ids": ("1.2.1", "1.2.2"),
|
|
"maximum_score": 3,
|
|
"interpretation": "Tracing immediate, secondary, and systemic consequences of an event.",
|
|
},
|
|
{
|
|
"id": "abstract_reasoning",
|
|
"label": "Abstract Reasoning",
|
|
"prompt_ids": ("2.1.1", "2.1.2", "2.1.3"),
|
|
"maximum_score": 3,
|
|
"interpretation": "Finding structural analogies and abstract patterns across domains.",
|
|
},
|
|
{
|
|
"id": "metacognition",
|
|
"label": "Metacognition",
|
|
"prompt_ids": ("2.2.1", "2.2.2"),
|
|
"maximum_score": 3,
|
|
"interpretation": "Calibrating confidence, recognizing uncertainty, and identifying ambiguity.",
|
|
},
|
|
{
|
|
"id": "sycophancy_resistance",
|
|
"label": "Sycophancy Resistance",
|
|
"prompt_ids": ("3.1.1", "3.1.2"),
|
|
"maximum_score": 2,
|
|
"interpretation": "Correcting false user premises instead of agreeing with them.",
|
|
},
|
|
{
|
|
"id": "neutrality",
|
|
"label": "Neutrality",
|
|
"prompt_ids": ("3.2.1",),
|
|
"maximum_score": 2,
|
|
"interpretation": "Presenting competing positions with balanced depth and persuasive force.",
|
|
},
|
|
{
|
|
"id": "robustness",
|
|
"label": "Robustness",
|
|
"prompt_ids": ("4.1.1", "4.1.2"),
|
|
"maximum_score": 2,
|
|
"interpretation": "Maintaining core conclusions across semantically equivalent prompt variants.",
|
|
},
|
|
)
|
|
|
|
PERSONALITY_AXES = {
|
|
"3.3.1": ("extraversion_introversion", {"E", "I"}),
|
|
"3.3.2": ("sensing_intuition", {"S", "N"}),
|
|
"3.3.3": ("thinking_feeling", {"T", "F"}),
|
|
"3.3.4": ("judging_perceiving", {"J", "P"}),
|
|
}
|
|
|
|
|
|
def load_evaluations(evaluations_dir):
|
|
"""Return the evaluator output indexed by prompt ID and any read errors."""
|
|
evaluations = {}
|
|
errors = []
|
|
for evaluation_file in sorted(evaluations_dir.glob("*.json")):
|
|
try:
|
|
evaluations[evaluation_file.stem] = json.loads(
|
|
evaluation_file.read_text(encoding="utf-8")
|
|
)
|
|
except (OSError, json.JSONDecodeError) as error:
|
|
errors.append(f"{evaluation_file.name}: {error}")
|
|
return evaluations, errors
|
|
|
|
|
|
def numeric_score(value):
|
|
"""Convert an evaluator score to a number, or return None for non-numeric values."""
|
|
if isinstance(value, bool):
|
|
return None
|
|
if isinstance(value, (int, float)):
|
|
return float(value)
|
|
try:
|
|
return float(value)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def build_numeric_dimensions(evaluations):
|
|
dimensions = []
|
|
incomplete_prompt_ids = []
|
|
|
|
for dimension in DIMENSIONS:
|
|
raw_scores = {}
|
|
for prompt_id in dimension["prompt_ids"]:
|
|
evaluation = evaluations.get(prompt_id, {})
|
|
score = (
|
|
None
|
|
if evaluation_is_retryable_failure(evaluation)
|
|
else numeric_score(evaluation.get("score"))
|
|
)
|
|
if score is None:
|
|
incomplete_prompt_ids.append(prompt_id)
|
|
else:
|
|
raw_scores[prompt_id] = score
|
|
|
|
raw_mean = (
|
|
round(sum(raw_scores.values()) / len(raw_scores), 4)
|
|
if raw_scores else None
|
|
)
|
|
normalized_score = (
|
|
round(raw_mean / dimension["maximum_score"], 4)
|
|
if raw_mean is not None else None
|
|
)
|
|
dimensions.append(
|
|
{
|
|
"id": dimension["id"],
|
|
"label": dimension["label"],
|
|
"prompt_ids": list(dimension["prompt_ids"]),
|
|
"raw_scores": raw_scores,
|
|
"raw_mean": raw_mean,
|
|
"maximum_score": dimension["maximum_score"],
|
|
"normalized_score": normalized_score,
|
|
"interpretation": dimension["interpretation"],
|
|
}
|
|
)
|
|
|
|
return dimensions, incomplete_prompt_ids
|
|
|
|
|
|
def build_style_profile(evaluations):
|
|
axes = {}
|
|
incomplete_prompt_ids = []
|
|
letters = []
|
|
for prompt_id, (axis_name, valid_scores) in PERSONALITY_AXES.items():
|
|
score = str(evaluations.get(prompt_id, {}).get("score", "")).upper()
|
|
if score not in valid_scores:
|
|
incomplete_prompt_ids.append(prompt_id)
|
|
axes[axis_name] = None
|
|
else:
|
|
axes[axis_name] = score
|
|
letters.append(score)
|
|
|
|
return {
|
|
"mbti_analogue": "".join(letters) if not incomplete_prompt_ids else None,
|
|
"axes": axes,
|
|
"scope_note": "A prompt-dependent communication-style label, not a psychological personality diagnosis.",
|
|
}, incomplete_prompt_ids
|
|
|
|
|
|
def find_radar_chart(model_id):
|
|
charts_dir = CHARTS_DIR
|
|
expected_name = f"{model_id.replace('/', '_')}_radar.png"
|
|
expected_path = charts_dir / expected_name
|
|
if expected_path.exists():
|
|
return workspace_relative(expected_path)
|
|
|
|
normalized_model = "".join(character.lower() for character in model_id if character.isalnum())
|
|
for chart in charts_dir.glob("*_radar.png"):
|
|
normalized_chart = "".join(character.lower() for character in chart.stem if character.isalnum())
|
|
if normalized_model in normalized_chart or normalized_chart in normalized_model:
|
|
return workspace_relative(chart)
|
|
return None
|
|
|
|
|
|
def build_profile(
|
|
model_id: str,
|
|
*,
|
|
display_name: str | None = None,
|
|
raw_provider: str = "unspecified",
|
|
evaluator_model: str = "unspecified",
|
|
report_provider: str = "unspecified",
|
|
output_path: Path | None = None,
|
|
artifact_model_id: str | None = None,
|
|
) -> tuple[Path, dict]:
|
|
"""Aggregate existing evaluations and write a Profile JSON file."""
|
|
|
|
model_id = model_id.strip("/")
|
|
if not model_id:
|
|
raise ValueError("model_id must not be empty")
|
|
artifact_model_id = (artifact_model_id or model_id).strip("/")
|
|
evaluations_dir = EVALUATIONS_DIR / artifact_model_id
|
|
results_dir = RESULTS_DIR / artifact_model_id
|
|
output_path = output_path or model_profile_path(PROFILES_DIR, model_id)
|
|
|
|
if not evaluations_dir.exists():
|
|
raise SystemExit(f"Evaluation directory not found: {evaluations_dir}")
|
|
|
|
evaluations, read_errors = load_evaluations(evaluations_dir)
|
|
numeric_dimensions, incomplete_numeric = build_numeric_dimensions(evaluations)
|
|
style_profile, incomplete_style = build_style_profile(evaluations)
|
|
incomplete_prompt_ids = sorted(set(incomplete_numeric + incomplete_style))
|
|
expected_count = sum(len(item["prompt_ids"]) for item in DIMENSIONS) + len(PERSONALITY_AXES)
|
|
|
|
artifact_safe_model_id = artifact_model_id.replace("/", "_")
|
|
report_path = REPORTS_DIR / f"{artifact_safe_model_id}_report.txt"
|
|
profile = {
|
|
"schema_version": "1.0",
|
|
"model": {
|
|
"id": model_id,
|
|
"display_name": display_name or model_id,
|
|
"profile_status": "complete" if not incomplete_prompt_ids and not read_errors else "partial",
|
|
"evaluations_completed": len(evaluations) - len(incomplete_prompt_ids),
|
|
"evaluations_expected": expected_count,
|
|
},
|
|
"provenance": {
|
|
"raw_responses_collected_via": raw_provider,
|
|
"evaluation_model": evaluator_model,
|
|
"narrative_report_generated_via": report_provider,
|
|
},
|
|
"behavioral_profile": {
|
|
"numeric_dimensions": numeric_dimensions,
|
|
"style_profile": style_profile,
|
|
},
|
|
"artifacts": {
|
|
"raw_responses_directory": workspace_relative(results_dir),
|
|
"evaluations_directory": workspace_relative(evaluations_dir),
|
|
"radar_chart": find_radar_chart(artifact_model_id),
|
|
"comparison_charts_directory": workspace_relative(CHARTS_DIR / "large"),
|
|
"narrative_report": workspace_relative(report_path),
|
|
},
|
|
"validation": {
|
|
"invalid_or_missing_prompt_ids": incomplete_prompt_ids,
|
|
"evaluation_file_read_errors": read_errors,
|
|
},
|
|
"interpretation_cautions": [
|
|
"Scores are produced by an LLM evaluator and are model-based judgments rather than ground truth.",
|
|
"The neutrality dimension contains one prompt and is therefore less stable than multi-prompt dimensions.",
|
|
"The metacognition category uses a repository-wide normalization maximum of 3, even though prompt 2.2.2 has a maximum of 2.",
|
|
],
|
|
}
|
|
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
output_path.write_text(json.dumps(profile, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
return output_path, profile
|