175 lines
6.2 KiB
Python
175 lines
6.2 KiB
Python
"""Run the complete behavioral-fingerprinting pipeline for one target model.
|
|
|
|
Usage:
|
|
python src/run_profile.py opencode/deepseek-v4-flash
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
|
|
from .paths import (
|
|
EVALUATIONS_DIR,
|
|
PROFILES_DIR,
|
|
WORKSPACE_ROOT,
|
|
model_profile_path,
|
|
)
|
|
from .profile_builder import DIMENSIONS, PERSONALITY_AXES, build_profile
|
|
from .providers import parse_model_reference, provider_label
|
|
from .retry_policy import evaluation_is_retryable_failure
|
|
|
|
|
|
COLLECTION_MODULE = (
|
|
"scripts.static_compile.profile_generation.model_preference.run_experiment"
|
|
)
|
|
EVALUATION_MODULE = (
|
|
"scripts.static_compile.profile_generation.model_preference.run_evaluation"
|
|
)
|
|
VISUALIZATION_MODULE = (
|
|
"scripts.static_compile.profile_generation.model_preference.visualize_results"
|
|
)
|
|
|
|
|
|
def parse_args():
|
|
parser = argparse.ArgumentParser(
|
|
description="Collect responses, evaluate them, visualize results, and build one profile."
|
|
)
|
|
parser.add_argument(
|
|
"target_model",
|
|
help="Target model in provider/model-id format, for example: opencode/qwen3.6-plus",
|
|
)
|
|
parser.add_argument(
|
|
"--refresh",
|
|
action="store_true",
|
|
help="Run collection and evaluation stages even when evaluation JSON already exists; successful cached responses and evaluations are retained.",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def run_step(name, command, environment):
|
|
print(f"\n{'=' * 80}\n{name}\n{'=' * 80}", flush=True)
|
|
subprocess.run(command, check=True, env=environment, cwd=WORKSPACE_ROOT)
|
|
|
|
|
|
def profile_output_path(model_id: str):
|
|
return model_profile_path(PROFILES_DIR, model_id)
|
|
|
|
|
|
def existing_profile_metadata(model_id: str) -> tuple[str, dict[str, str]]:
|
|
"""Preserve provenance when rebuilding a Profile from cached evaluations."""
|
|
|
|
output_path = profile_output_path(model_id)
|
|
if not output_path.is_file():
|
|
return model_id, {}
|
|
try:
|
|
existing = json.loads(output_path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return model_id, {}
|
|
model = existing.get("model")
|
|
provenance = existing.get("provenance")
|
|
display_name = (
|
|
model.get("display_name")
|
|
if isinstance(model, dict) and isinstance(model.get("display_name"), str)
|
|
else model_id
|
|
)
|
|
return display_name, provenance if isinstance(provenance, dict) else {}
|
|
|
|
|
|
def evaluation_cache_is_complete(evaluation_dir) -> bool:
|
|
"""Return whether every expected evaluation exists and is reusable."""
|
|
|
|
expected_prompt_ids = {
|
|
prompt_id
|
|
for dimension in DIMENSIONS
|
|
for prompt_id in dimension["prompt_ids"]
|
|
} | set(PERSONALITY_AXES)
|
|
for prompt_id in expected_prompt_ids:
|
|
evaluation_path = evaluation_dir / f"{prompt_id}.json"
|
|
try:
|
|
evaluation = json.loads(evaluation_path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return False
|
|
if not isinstance(evaluation, dict) or evaluation_is_retryable_failure(evaluation):
|
|
return False
|
|
return True
|
|
|
|
|
|
def write_profile(
|
|
model_id: str,
|
|
*,
|
|
environment: dict[str, str],
|
|
cached: bool,
|
|
artifact_model_id: str | None = None,
|
|
) -> None:
|
|
display_name, prior_provenance = existing_profile_metadata(model_id)
|
|
if cached:
|
|
raw_provider = str(prior_provenance.get("raw_responses_collected_via", "unspecified"))
|
|
if raw_provider == "unspecified":
|
|
raw_provider = provider_label(parse_model_reference(model_id).provider)
|
|
evaluator_model = str(prior_provenance.get("evaluation_model", "unspecified"))
|
|
report_provider = str(prior_provenance.get("narrative_report_generated_via", "unspecified"))
|
|
else:
|
|
raw_provider = provider_label(parse_model_reference(model_id).provider)
|
|
evaluator_model = environment["PROFILE_EVALUATOR_MODEL"]
|
|
report_provider = provider_label(
|
|
parse_model_reference(environment["PROFILE_REPORT_MODEL"]).provider
|
|
)
|
|
output_path, profile = build_profile(
|
|
model_id,
|
|
display_name=display_name,
|
|
raw_provider=raw_provider,
|
|
evaluator_model=evaluator_model,
|
|
report_provider=report_provider,
|
|
artifact_model_id=artifact_model_id,
|
|
)
|
|
print(f"Wrote {profile['model']['profile_status']} profile to {output_path}")
|
|
for prompt_id in profile["validation"]["invalid_or_missing_prompt_ids"]:
|
|
print(f"- Missing or invalid score: {prompt_id}")
|
|
for error in profile["validation"]["evaluation_file_read_errors"]:
|
|
print(f"- Could not read evaluation: {error}")
|
|
|
|
|
|
def main():
|
|
args = parse_args()
|
|
try:
|
|
target_model = parse_model_reference(args.target_model).value
|
|
except ValueError as error:
|
|
raise SystemExit(f"error: {error}") from error
|
|
|
|
environment = os.environ.copy()
|
|
environment["PROFILE_TARGET_MODEL"] = target_model
|
|
# One command-level model routes every external call in this pipeline.
|
|
environment["PROFILE_EVALUATOR_MODEL"] = target_model
|
|
environment["PROFILE_REPORT_MODEL"] = target_model
|
|
|
|
evaluation_dir = EVALUATIONS_DIR / target_model
|
|
has_complete_cache = (
|
|
evaluation_dir.is_dir() and evaluation_cache_is_complete(evaluation_dir)
|
|
)
|
|
if has_complete_cache and not args.refresh:
|
|
print(
|
|
f"Found existing evaluations in {evaluation_dir}; "
|
|
"rebuilding Profile only. Use --refresh to rerun live stages."
|
|
)
|
|
write_profile(
|
|
target_model,
|
|
environment=environment,
|
|
cached=True,
|
|
artifact_model_id=target_model,
|
|
)
|
|
return
|
|
|
|
python = sys.executable
|
|
run_step("1/4 Collecting target-model responses", [python, "-m", COLLECTION_MODULE], environment)
|
|
run_step("2/4 Evaluating responses", [python, "-m", EVALUATION_MODULE], environment)
|
|
run_step("3/4 Generating charts and narrative report", [python, "-m", VISUALIZATION_MODULE], environment)
|
|
print(f"\n{'=' * 80}\n4/4 Building structured profile JSON\n{'=' * 80}")
|
|
write_profile(target_model, environment=environment, cached=False)
|
|
print(f"\nComplete. Profile: {profile_output_path(target_model)}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|