Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
@@ -0,0 +1,174 @@
"""Run the complete behavioral-fingerprinting pipeline for one target model.
Usage:
python src/run_profile.py opencode/deepseek-v4-flash
"""
import argparse
import json
import os
import subprocess
import sys
from .paths import (
EVALUATIONS_DIR,
PROFILES_DIR,
WORKSPACE_ROOT,
model_profile_path,
)
from .profile_builder import DIMENSIONS, PERSONALITY_AXES, build_profile
from .providers import parse_model_reference, provider_label
from .retry_policy import evaluation_is_retryable_failure
COLLECTION_MODULE = (
"scripts.static_compile.profile_generation.model_preference.run_experiment"
)
EVALUATION_MODULE = (
"scripts.static_compile.profile_generation.model_preference.run_evaluation"
)
VISUALIZATION_MODULE = (
"scripts.static_compile.profile_generation.model_preference.visualize_results"
)
def parse_args():
parser = argparse.ArgumentParser(
description="Collect responses, evaluate them, visualize results, and build one profile."
)
parser.add_argument(
"target_model",
help="Target model in provider/model-id format, for example: opencode/qwen3.6-plus",
)
parser.add_argument(
"--refresh",
action="store_true",
help="Run collection and evaluation stages even when evaluation JSON already exists; successful cached responses and evaluations are retained.",
)
return parser.parse_args()
def run_step(name, command, environment):
print(f"\n{'=' * 80}\n{name}\n{'=' * 80}", flush=True)
subprocess.run(command, check=True, env=environment, cwd=WORKSPACE_ROOT)
def profile_output_path(model_id: str):
return model_profile_path(PROFILES_DIR, model_id)
def existing_profile_metadata(model_id: str) -> tuple[str, dict[str, str]]:
"""Preserve provenance when rebuilding a Profile from cached evaluations."""
output_path = profile_output_path(model_id)
if not output_path.is_file():
return model_id, {}
try:
existing = json.loads(output_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return model_id, {}
model = existing.get("model")
provenance = existing.get("provenance")
display_name = (
model.get("display_name")
if isinstance(model, dict) and isinstance(model.get("display_name"), str)
else model_id
)
return display_name, provenance if isinstance(provenance, dict) else {}
def evaluation_cache_is_complete(evaluation_dir) -> bool:
"""Return whether every expected evaluation exists and is reusable."""
expected_prompt_ids = {
prompt_id
for dimension in DIMENSIONS
for prompt_id in dimension["prompt_ids"]
} | set(PERSONALITY_AXES)
for prompt_id in expected_prompt_ids:
evaluation_path = evaluation_dir / f"{prompt_id}.json"
try:
evaluation = json.loads(evaluation_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return False
if not isinstance(evaluation, dict) or evaluation_is_retryable_failure(evaluation):
return False
return True
def write_profile(
model_id: str,
*,
environment: dict[str, str],
cached: bool,
artifact_model_id: str | None = None,
) -> None:
display_name, prior_provenance = existing_profile_metadata(model_id)
if cached:
raw_provider = str(prior_provenance.get("raw_responses_collected_via", "unspecified"))
if raw_provider == "unspecified":
raw_provider = provider_label(parse_model_reference(model_id).provider)
evaluator_model = str(prior_provenance.get("evaluation_model", "unspecified"))
report_provider = str(prior_provenance.get("narrative_report_generated_via", "unspecified"))
else:
raw_provider = provider_label(parse_model_reference(model_id).provider)
evaluator_model = environment["PROFILE_EVALUATOR_MODEL"]
report_provider = provider_label(
parse_model_reference(environment["PROFILE_REPORT_MODEL"]).provider
)
output_path, profile = build_profile(
model_id,
display_name=display_name,
raw_provider=raw_provider,
evaluator_model=evaluator_model,
report_provider=report_provider,
artifact_model_id=artifact_model_id,
)
print(f"Wrote {profile['model']['profile_status']} profile to {output_path}")
for prompt_id in profile["validation"]["invalid_or_missing_prompt_ids"]:
print(f"- Missing or invalid score: {prompt_id}")
for error in profile["validation"]["evaluation_file_read_errors"]:
print(f"- Could not read evaluation: {error}")
def main():
args = parse_args()
try:
target_model = parse_model_reference(args.target_model).value
except ValueError as error:
raise SystemExit(f"error: {error}") from error
environment = os.environ.copy()
environment["PROFILE_TARGET_MODEL"] = target_model
# One command-level model routes every external call in this pipeline.
environment["PROFILE_EVALUATOR_MODEL"] = target_model
environment["PROFILE_REPORT_MODEL"] = target_model
evaluation_dir = EVALUATIONS_DIR / target_model
has_complete_cache = (
evaluation_dir.is_dir() and evaluation_cache_is_complete(evaluation_dir)
)
if has_complete_cache and not args.refresh:
print(
f"Found existing evaluations in {evaluation_dir}; "
"rebuilding Profile only. Use --refresh to rerun live stages."
)
write_profile(
target_model,
environment=environment,
cached=True,
artifact_model_id=target_model,
)
return
python = sys.executable
run_step("1/4 Collecting target-model responses", [python, "-m", COLLECTION_MODULE], environment)
run_step("2/4 Evaluating responses", [python, "-m", EVALUATION_MODULE], environment)
run_step("3/4 Generating charts and narrative report", [python, "-m", VISUALIZATION_MODULE], environment)
print(f"\n{'=' * 80}\n4/4 Building structured profile JSON\n{'=' * 80}")
write_profile(target_model, environment=environment, cached=False)
print(f"\nComplete. Profile: {profile_output_path(target_model)}")
if __name__ == "__main__":
main()