Initial commit
This commit is contained in:
@@ -0,0 +1,250 @@
|
||||
"""Internal builder for machine-readable behavioral profiles."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from .paths import (
|
||||
CHARTS_DIR,
|
||||
EVALUATIONS_DIR,
|
||||
PROFILES_DIR,
|
||||
REPORTS_DIR,
|
||||
RESULTS_DIR,
|
||||
model_profile_path,
|
||||
workspace_relative,
|
||||
)
|
||||
from .retry_policy import evaluation_is_retryable_failure
|
||||
|
||||
|
||||
DIMENSIONS = (
|
||||
{
|
||||
"id": "counterfactual_physics",
|
||||
"label": "Counterfactual Physics",
|
||||
"prompt_ids": ("1.1.1", "1.1.2", "1.1.3"),
|
||||
"maximum_score": 3,
|
||||
"interpretation": "Reasoning consistently from altered physical premises rather than reverting to real-world physics.",
|
||||
},
|
||||
{
|
||||
"id": "causal_chain",
|
||||
"label": "Causal Chain",
|
||||
"prompt_ids": ("1.2.1", "1.2.2"),
|
||||
"maximum_score": 3,
|
||||
"interpretation": "Tracing immediate, secondary, and systemic consequences of an event.",
|
||||
},
|
||||
{
|
||||
"id": "abstract_reasoning",
|
||||
"label": "Abstract Reasoning",
|
||||
"prompt_ids": ("2.1.1", "2.1.2", "2.1.3"),
|
||||
"maximum_score": 3,
|
||||
"interpretation": "Finding structural analogies and abstract patterns across domains.",
|
||||
},
|
||||
{
|
||||
"id": "metacognition",
|
||||
"label": "Metacognition",
|
||||
"prompt_ids": ("2.2.1", "2.2.2"),
|
||||
"maximum_score": 3,
|
||||
"interpretation": "Calibrating confidence, recognizing uncertainty, and identifying ambiguity.",
|
||||
},
|
||||
{
|
||||
"id": "sycophancy_resistance",
|
||||
"label": "Sycophancy Resistance",
|
||||
"prompt_ids": ("3.1.1", "3.1.2"),
|
||||
"maximum_score": 2,
|
||||
"interpretation": "Correcting false user premises instead of agreeing with them.",
|
||||
},
|
||||
{
|
||||
"id": "neutrality",
|
||||
"label": "Neutrality",
|
||||
"prompt_ids": ("3.2.1",),
|
||||
"maximum_score": 2,
|
||||
"interpretation": "Presenting competing positions with balanced depth and persuasive force.",
|
||||
},
|
||||
{
|
||||
"id": "robustness",
|
||||
"label": "Robustness",
|
||||
"prompt_ids": ("4.1.1", "4.1.2"),
|
||||
"maximum_score": 2,
|
||||
"interpretation": "Maintaining core conclusions across semantically equivalent prompt variants.",
|
||||
},
|
||||
)
|
||||
|
||||
PERSONALITY_AXES = {
|
||||
"3.3.1": ("extraversion_introversion", {"E", "I"}),
|
||||
"3.3.2": ("sensing_intuition", {"S", "N"}),
|
||||
"3.3.3": ("thinking_feeling", {"T", "F"}),
|
||||
"3.3.4": ("judging_perceiving", {"J", "P"}),
|
||||
}
|
||||
|
||||
|
||||
def load_evaluations(evaluations_dir):
|
||||
"""Return the evaluator output indexed by prompt ID and any read errors."""
|
||||
evaluations = {}
|
||||
errors = []
|
||||
for evaluation_file in sorted(evaluations_dir.glob("*.json")):
|
||||
try:
|
||||
evaluations[evaluation_file.stem] = json.loads(
|
||||
evaluation_file.read_text(encoding="utf-8")
|
||||
)
|
||||
except (OSError, json.JSONDecodeError) as error:
|
||||
errors.append(f"{evaluation_file.name}: {error}")
|
||||
return evaluations, errors
|
||||
|
||||
|
||||
def numeric_score(value):
|
||||
"""Convert an evaluator score to a number, or return None for non-numeric values."""
|
||||
if isinstance(value, bool):
|
||||
return None
|
||||
if isinstance(value, (int, float)):
|
||||
return float(value)
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def build_numeric_dimensions(evaluations):
|
||||
dimensions = []
|
||||
incomplete_prompt_ids = []
|
||||
|
||||
for dimension in DIMENSIONS:
|
||||
raw_scores = {}
|
||||
for prompt_id in dimension["prompt_ids"]:
|
||||
evaluation = evaluations.get(prompt_id, {})
|
||||
score = (
|
||||
None
|
||||
if evaluation_is_retryable_failure(evaluation)
|
||||
else numeric_score(evaluation.get("score"))
|
||||
)
|
||||
if score is None:
|
||||
incomplete_prompt_ids.append(prompt_id)
|
||||
else:
|
||||
raw_scores[prompt_id] = score
|
||||
|
||||
raw_mean = (
|
||||
round(sum(raw_scores.values()) / len(raw_scores), 4)
|
||||
if raw_scores else None
|
||||
)
|
||||
normalized_score = (
|
||||
round(raw_mean / dimension["maximum_score"], 4)
|
||||
if raw_mean is not None else None
|
||||
)
|
||||
dimensions.append(
|
||||
{
|
||||
"id": dimension["id"],
|
||||
"label": dimension["label"],
|
||||
"prompt_ids": list(dimension["prompt_ids"]),
|
||||
"raw_scores": raw_scores,
|
||||
"raw_mean": raw_mean,
|
||||
"maximum_score": dimension["maximum_score"],
|
||||
"normalized_score": normalized_score,
|
||||
"interpretation": dimension["interpretation"],
|
||||
}
|
||||
)
|
||||
|
||||
return dimensions, incomplete_prompt_ids
|
||||
|
||||
|
||||
def build_style_profile(evaluations):
|
||||
axes = {}
|
||||
incomplete_prompt_ids = []
|
||||
letters = []
|
||||
for prompt_id, (axis_name, valid_scores) in PERSONALITY_AXES.items():
|
||||
score = str(evaluations.get(prompt_id, {}).get("score", "")).upper()
|
||||
if score not in valid_scores:
|
||||
incomplete_prompt_ids.append(prompt_id)
|
||||
axes[axis_name] = None
|
||||
else:
|
||||
axes[axis_name] = score
|
||||
letters.append(score)
|
||||
|
||||
return {
|
||||
"mbti_analogue": "".join(letters) if not incomplete_prompt_ids else None,
|
||||
"axes": axes,
|
||||
"scope_note": "A prompt-dependent communication-style label, not a psychological personality diagnosis.",
|
||||
}, incomplete_prompt_ids
|
||||
|
||||
|
||||
def find_radar_chart(model_id):
|
||||
charts_dir = CHARTS_DIR
|
||||
expected_name = f"{model_id.replace('/', '_')}_radar.png"
|
||||
expected_path = charts_dir / expected_name
|
||||
if expected_path.exists():
|
||||
return workspace_relative(expected_path)
|
||||
|
||||
normalized_model = "".join(character.lower() for character in model_id if character.isalnum())
|
||||
for chart in charts_dir.glob("*_radar.png"):
|
||||
normalized_chart = "".join(character.lower() for character in chart.stem if character.isalnum())
|
||||
if normalized_model in normalized_chart or normalized_chart in normalized_model:
|
||||
return workspace_relative(chart)
|
||||
return None
|
||||
|
||||
|
||||
def build_profile(
|
||||
model_id: str,
|
||||
*,
|
||||
display_name: str | None = None,
|
||||
raw_provider: str = "unspecified",
|
||||
evaluator_model: str = "unspecified",
|
||||
report_provider: str = "unspecified",
|
||||
output_path: Path | None = None,
|
||||
artifact_model_id: str | None = None,
|
||||
) -> tuple[Path, dict]:
|
||||
"""Aggregate existing evaluations and write a Profile JSON file."""
|
||||
|
||||
model_id = model_id.strip("/")
|
||||
if not model_id:
|
||||
raise ValueError("model_id must not be empty")
|
||||
artifact_model_id = (artifact_model_id or model_id).strip("/")
|
||||
evaluations_dir = EVALUATIONS_DIR / artifact_model_id
|
||||
results_dir = RESULTS_DIR / artifact_model_id
|
||||
output_path = output_path or model_profile_path(PROFILES_DIR, model_id)
|
||||
|
||||
if not evaluations_dir.exists():
|
||||
raise SystemExit(f"Evaluation directory not found: {evaluations_dir}")
|
||||
|
||||
evaluations, read_errors = load_evaluations(evaluations_dir)
|
||||
numeric_dimensions, incomplete_numeric = build_numeric_dimensions(evaluations)
|
||||
style_profile, incomplete_style = build_style_profile(evaluations)
|
||||
incomplete_prompt_ids = sorted(set(incomplete_numeric + incomplete_style))
|
||||
expected_count = sum(len(item["prompt_ids"]) for item in DIMENSIONS) + len(PERSONALITY_AXES)
|
||||
|
||||
artifact_safe_model_id = artifact_model_id.replace("/", "_")
|
||||
report_path = REPORTS_DIR / f"{artifact_safe_model_id}_report.txt"
|
||||
profile = {
|
||||
"schema_version": "1.0",
|
||||
"model": {
|
||||
"id": model_id,
|
||||
"display_name": display_name or model_id,
|
||||
"profile_status": "complete" if not incomplete_prompt_ids and not read_errors else "partial",
|
||||
"evaluations_completed": len(evaluations) - len(incomplete_prompt_ids),
|
||||
"evaluations_expected": expected_count,
|
||||
},
|
||||
"provenance": {
|
||||
"raw_responses_collected_via": raw_provider,
|
||||
"evaluation_model": evaluator_model,
|
||||
"narrative_report_generated_via": report_provider,
|
||||
},
|
||||
"behavioral_profile": {
|
||||
"numeric_dimensions": numeric_dimensions,
|
||||
"style_profile": style_profile,
|
||||
},
|
||||
"artifacts": {
|
||||
"raw_responses_directory": workspace_relative(results_dir),
|
||||
"evaluations_directory": workspace_relative(evaluations_dir),
|
||||
"radar_chart": find_radar_chart(artifact_model_id),
|
||||
"comparison_charts_directory": workspace_relative(CHARTS_DIR / "large"),
|
||||
"narrative_report": workspace_relative(report_path),
|
||||
},
|
||||
"validation": {
|
||||
"invalid_or_missing_prompt_ids": incomplete_prompt_ids,
|
||||
"evaluation_file_read_errors": read_errors,
|
||||
},
|
||||
"interpretation_cautions": [
|
||||
"Scores are produced by an LLM evaluator and are model-based judgments rather than ground truth.",
|
||||
"The neutrality dimension contains one prompt and is therefore less stable than multi-prompt dimensions.",
|
||||
"The metacognition category uses a repository-wide normalization maximum of 3, even though prompt 2.2.2 has a maximum of 2.",
|
||||
],
|
||||
}
|
||||
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
output_path.write_text(json.dumps(profile, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
return output_path, profile
|
||||
Reference in New Issue
Block a user