"""Run the complete behavioral-fingerprinting pipeline for one target model. Usage: python src/run_profile.py opencode/deepseek-v4-flash """ import argparse import json import os import subprocess import sys from .paths import ( EVALUATIONS_DIR, PROFILES_DIR, WORKSPACE_ROOT, model_profile_path, ) from .profile_builder import DIMENSIONS, PERSONALITY_AXES, build_profile from .providers import parse_model_reference, provider_label from .retry_policy import evaluation_is_retryable_failure COLLECTION_MODULE = ( "scripts.static_compile.profile_generation.model_preference.run_experiment" ) EVALUATION_MODULE = ( "scripts.static_compile.profile_generation.model_preference.run_evaluation" ) VISUALIZATION_MODULE = ( "scripts.static_compile.profile_generation.model_preference.visualize_results" ) def parse_args(): parser = argparse.ArgumentParser( description="Collect responses, evaluate them, visualize results, and build one profile." ) parser.add_argument( "target_model", help="Target model in provider/model-id format, for example: opencode/qwen3.6-plus", ) parser.add_argument( "--refresh", action="store_true", help="Run collection and evaluation stages even when evaluation JSON already exists; successful cached responses and evaluations are retained.", ) return parser.parse_args() def run_step(name, command, environment): print(f"\n{'=' * 80}\n{name}\n{'=' * 80}", flush=True) subprocess.run(command, check=True, env=environment, cwd=WORKSPACE_ROOT) def profile_output_path(model_id: str): return model_profile_path(PROFILES_DIR, model_id) def existing_profile_metadata(model_id: str) -> tuple[str, dict[str, str]]: """Preserve provenance when rebuilding a Profile from cached evaluations.""" output_path = profile_output_path(model_id) if not output_path.is_file(): return model_id, {} try: existing = json.loads(output_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError): return model_id, {} model = existing.get("model") provenance = existing.get("provenance") display_name = ( model.get("display_name") if isinstance(model, dict) and isinstance(model.get("display_name"), str) else model_id ) return display_name, provenance if isinstance(provenance, dict) else {} def evaluation_cache_is_complete(evaluation_dir) -> bool: """Return whether every expected evaluation exists and is reusable.""" expected_prompt_ids = { prompt_id for dimension in DIMENSIONS for prompt_id in dimension["prompt_ids"] } | set(PERSONALITY_AXES) for prompt_id in expected_prompt_ids: evaluation_path = evaluation_dir / f"{prompt_id}.json" try: evaluation = json.loads(evaluation_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError): return False if not isinstance(evaluation, dict) or evaluation_is_retryable_failure(evaluation): return False return True def write_profile( model_id: str, *, environment: dict[str, str], cached: bool, artifact_model_id: str | None = None, ) -> None: display_name, prior_provenance = existing_profile_metadata(model_id) if cached: raw_provider = str(prior_provenance.get("raw_responses_collected_via", "unspecified")) if raw_provider == "unspecified": raw_provider = provider_label(parse_model_reference(model_id).provider) evaluator_model = str(prior_provenance.get("evaluation_model", "unspecified")) report_provider = str(prior_provenance.get("narrative_report_generated_via", "unspecified")) else: raw_provider = provider_label(parse_model_reference(model_id).provider) evaluator_model = environment["PROFILE_EVALUATOR_MODEL"] report_provider = provider_label( parse_model_reference(environment["PROFILE_REPORT_MODEL"]).provider ) output_path, profile = build_profile( model_id, display_name=display_name, raw_provider=raw_provider, evaluator_model=evaluator_model, report_provider=report_provider, artifact_model_id=artifact_model_id, ) print(f"Wrote {profile['model']['profile_status']} profile to {output_path}") for prompt_id in profile["validation"]["invalid_or_missing_prompt_ids"]: print(f"- Missing or invalid score: {prompt_id}") for error in profile["validation"]["evaluation_file_read_errors"]: print(f"- Could not read evaluation: {error}") def main(): args = parse_args() try: target_model = parse_model_reference(args.target_model).value except ValueError as error: raise SystemExit(f"error: {error}") from error environment = os.environ.copy() environment["PROFILE_TARGET_MODEL"] = target_model # One command-level model routes every external call in this pipeline. environment["PROFILE_EVALUATOR_MODEL"] = target_model environment["PROFILE_REPORT_MODEL"] = target_model evaluation_dir = EVALUATIONS_DIR / target_model has_complete_cache = ( evaluation_dir.is_dir() and evaluation_cache_is_complete(evaluation_dir) ) if has_complete_cache and not args.refresh: print( f"Found existing evaluations in {evaluation_dir}; " "rebuilding Profile only. Use --refresh to rerun live stages." ) write_profile( target_model, environment=environment, cached=True, artifact_model_id=target_model, ) return python = sys.executable run_step("1/4 Collecting target-model responses", [python, "-m", COLLECTION_MODULE], environment) run_step("2/4 Evaluating responses", [python, "-m", EVALUATION_MODULE], environment) run_step("3/4 Generating charts and narrative report", [python, "-m", VISUALIZATION_MODULE], environment) print(f"\n{'=' * 80}\n4/4 Building structured profile JSON\n{'=' * 80}") write_profile(target_model, environment=environment, cached=False) print(f"\nComplete. Profile: {profile_output_path(target_model)}") if __name__ == "__main__": main()