Initial commit
This commit is contained in:
@@ -0,0 +1,174 @@
|
||||
"""Run the complete behavioral-fingerprinting pipeline for one target model.
|
||||
|
||||
Usage:
|
||||
python src/run_profile.py opencode/deepseek-v4-flash
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from .paths import (
|
||||
EVALUATIONS_DIR,
|
||||
PROFILES_DIR,
|
||||
WORKSPACE_ROOT,
|
||||
model_profile_path,
|
||||
)
|
||||
from .profile_builder import DIMENSIONS, PERSONALITY_AXES, build_profile
|
||||
from .providers import parse_model_reference, provider_label
|
||||
from .retry_policy import evaluation_is_retryable_failure
|
||||
|
||||
|
||||
COLLECTION_MODULE = (
|
||||
"scripts.static_compile.profile_generation.model_preference.run_experiment"
|
||||
)
|
||||
EVALUATION_MODULE = (
|
||||
"scripts.static_compile.profile_generation.model_preference.run_evaluation"
|
||||
)
|
||||
VISUALIZATION_MODULE = (
|
||||
"scripts.static_compile.profile_generation.model_preference.visualize_results"
|
||||
)
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Collect responses, evaluate them, visualize results, and build one profile."
|
||||
)
|
||||
parser.add_argument(
|
||||
"target_model",
|
||||
help="Target model in provider/model-id format, for example: opencode/qwen3.6-plus",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--refresh",
|
||||
action="store_true",
|
||||
help="Run collection and evaluation stages even when evaluation JSON already exists; successful cached responses and evaluations are retained.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def run_step(name, command, environment):
|
||||
print(f"\n{'=' * 80}\n{name}\n{'=' * 80}", flush=True)
|
||||
subprocess.run(command, check=True, env=environment, cwd=WORKSPACE_ROOT)
|
||||
|
||||
|
||||
def profile_output_path(model_id: str):
|
||||
return model_profile_path(PROFILES_DIR, model_id)
|
||||
|
||||
|
||||
def existing_profile_metadata(model_id: str) -> tuple[str, dict[str, str]]:
|
||||
"""Preserve provenance when rebuilding a Profile from cached evaluations."""
|
||||
|
||||
output_path = profile_output_path(model_id)
|
||||
if not output_path.is_file():
|
||||
return model_id, {}
|
||||
try:
|
||||
existing = json.loads(output_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return model_id, {}
|
||||
model = existing.get("model")
|
||||
provenance = existing.get("provenance")
|
||||
display_name = (
|
||||
model.get("display_name")
|
||||
if isinstance(model, dict) and isinstance(model.get("display_name"), str)
|
||||
else model_id
|
||||
)
|
||||
return display_name, provenance if isinstance(provenance, dict) else {}
|
||||
|
||||
|
||||
def evaluation_cache_is_complete(evaluation_dir) -> bool:
|
||||
"""Return whether every expected evaluation exists and is reusable."""
|
||||
|
||||
expected_prompt_ids = {
|
||||
prompt_id
|
||||
for dimension in DIMENSIONS
|
||||
for prompt_id in dimension["prompt_ids"]
|
||||
} | set(PERSONALITY_AXES)
|
||||
for prompt_id in expected_prompt_ids:
|
||||
evaluation_path = evaluation_dir / f"{prompt_id}.json"
|
||||
try:
|
||||
evaluation = json.loads(evaluation_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return False
|
||||
if not isinstance(evaluation, dict) or evaluation_is_retryable_failure(evaluation):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def write_profile(
|
||||
model_id: str,
|
||||
*,
|
||||
environment: dict[str, str],
|
||||
cached: bool,
|
||||
artifact_model_id: str | None = None,
|
||||
) -> None:
|
||||
display_name, prior_provenance = existing_profile_metadata(model_id)
|
||||
if cached:
|
||||
raw_provider = str(prior_provenance.get("raw_responses_collected_via", "unspecified"))
|
||||
if raw_provider == "unspecified":
|
||||
raw_provider = provider_label(parse_model_reference(model_id).provider)
|
||||
evaluator_model = str(prior_provenance.get("evaluation_model", "unspecified"))
|
||||
report_provider = str(prior_provenance.get("narrative_report_generated_via", "unspecified"))
|
||||
else:
|
||||
raw_provider = provider_label(parse_model_reference(model_id).provider)
|
||||
evaluator_model = environment["PROFILE_EVALUATOR_MODEL"]
|
||||
report_provider = provider_label(
|
||||
parse_model_reference(environment["PROFILE_REPORT_MODEL"]).provider
|
||||
)
|
||||
output_path, profile = build_profile(
|
||||
model_id,
|
||||
display_name=display_name,
|
||||
raw_provider=raw_provider,
|
||||
evaluator_model=evaluator_model,
|
||||
report_provider=report_provider,
|
||||
artifact_model_id=artifact_model_id,
|
||||
)
|
||||
print(f"Wrote {profile['model']['profile_status']} profile to {output_path}")
|
||||
for prompt_id in profile["validation"]["invalid_or_missing_prompt_ids"]:
|
||||
print(f"- Missing or invalid score: {prompt_id}")
|
||||
for error in profile["validation"]["evaluation_file_read_errors"]:
|
||||
print(f"- Could not read evaluation: {error}")
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
try:
|
||||
target_model = parse_model_reference(args.target_model).value
|
||||
except ValueError as error:
|
||||
raise SystemExit(f"error: {error}") from error
|
||||
|
||||
environment = os.environ.copy()
|
||||
environment["PROFILE_TARGET_MODEL"] = target_model
|
||||
# One command-level model routes every external call in this pipeline.
|
||||
environment["PROFILE_EVALUATOR_MODEL"] = target_model
|
||||
environment["PROFILE_REPORT_MODEL"] = target_model
|
||||
|
||||
evaluation_dir = EVALUATIONS_DIR / target_model
|
||||
has_complete_cache = (
|
||||
evaluation_dir.is_dir() and evaluation_cache_is_complete(evaluation_dir)
|
||||
)
|
||||
if has_complete_cache and not args.refresh:
|
||||
print(
|
||||
f"Found existing evaluations in {evaluation_dir}; "
|
||||
"rebuilding Profile only. Use --refresh to rerun live stages."
|
||||
)
|
||||
write_profile(
|
||||
target_model,
|
||||
environment=environment,
|
||||
cached=True,
|
||||
artifact_model_id=target_model,
|
||||
)
|
||||
return
|
||||
|
||||
python = sys.executable
|
||||
run_step("1/4 Collecting target-model responses", [python, "-m", COLLECTION_MODULE], environment)
|
||||
run_step("2/4 Evaluating responses", [python, "-m", EVALUATION_MODULE], environment)
|
||||
run_step("3/4 Generating charts and narrative report", [python, "-m", VISUALIZATION_MODULE], environment)
|
||||
print(f"\n{'=' * 80}\n4/4 Building structured profile JSON\n{'=' * 80}")
|
||||
write_profile(target_model, environment=environment, cached=False)
|
||||
print(f"\nComplete. Profile: {profile_output_path(target_model)}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user