Files
llm/scripts/run_live_model_evals.py
T

151 lines
5.2 KiB
Python

from __future__ import annotations
import argparse
import datetime as dt
import json
import subprocess
import sys
from pathlib import Path
from check_inference_endpoint import check_endpoint
from common import read_json
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_REPORT = ROOT / "reports" / "evals" / "live-model-evals.json"
GPU_PROFILES = ROOT / "config" / "gpu_profiles.json"
DEFAULT_1C_SYSTEM_PROMPT = ROOT / "plugins" / "1c" / "prompts" / "system.md"
SERVICE_TARGETS = {
"vllm-text": ["vllm-qwen3-4b", "vllm-qwen3-4b-1c-lora"],
"llama-gguf": ["llama-devstral-1c-q4"],
}
TARGETS = [
{
"id": "vllm-qwen3-4b",
"kind": "vllm",
"base_url": "http://docker-gpu.cin.su:8000",
"model": "qwen3-4b-instruct",
"eval_report": ROOT / "reports" / "evals" / "1c-smoke.vllm-qwen3-4b.json",
},
{
"id": "vllm-qwen3-4b-1c-lora",
"kind": "vllm-lora",
"base_url": "http://docker-gpu.cin.su:8000",
"model": "qwen3-4b-1c",
"eval_report": ROOT / "reports" / "evals" / "1c-smoke.vllm-qwen3-4b-1c-lora.json",
},
{
"id": "llama-devstral-1c-q4",
"kind": "llama.cpp",
"base_url": "http://docker-gpu.cin.su:8080",
"model": "devstral-1c-q4",
"eval_report": ROOT / "reports" / "evals" / "1c-smoke.llama-devstral-1c-q4.json",
},
]
def targets_for_profile(profile_id: str) -> list[dict]:
profiles = read_json(GPU_PROFILES)
profile = profiles.get(profile_id)
if not isinstance(profile, dict):
available = ", ".join(sorted(str(name) for name in profiles))
raise ValueError(f"Unknown GPU profile `{profile_id}`. Available: {available}")
target_ids = set()
for service in profile.get("starts") or []:
target_ids.update(SERVICE_TARGETS.get(service, []))
return [target for target in TARGETS if target["id"] in target_ids]
def run_eval(base_url: str, model: str, report_path: Path, *, system_prompt: Path | None = None) -> dict:
command = [
sys.executable,
"scripts/run_1c_smoke_eval.py",
"--base-url",
base_url,
"--model",
model,
"--report",
str(report_path),
]
if system_prompt:
command.extend(["--system-prompt", str(system_prompt)])
result = subprocess.run(
command,
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
return {
"command": command,
"status": "ok" if result.returncode == 0 else "failed",
"returncode": result.returncode,
"stdout": result.stdout.strip(),
"stderr": result.stderr.strip(),
"report": str(report_path.relative_to(ROOT)),
}
def main() -> int:
parser = argparse.ArgumentParser(description="Run live model evals against available inference endpoints.")
parser.add_argument("--profile", default="text", help="GPU profile from config/gpu_profiles.json to evaluate.")
parser.add_argument("--all-targets", action="store_true", help="Evaluate all known targets, including mutually exclusive GPU profiles.")
parser.add_argument("--timeout", type=int, default=10)
parser.add_argument("--report", type=Path, default=DEFAULT_REPORT)
parser.add_argument("--system-prompt", type=Path, default=DEFAULT_1C_SYSTEM_PROMPT)
parser.add_argument("--print", action="store_true")
args = parser.parse_args()
selected_targets = TARGETS if args.all_targets else targets_for_profile(args.profile)
results = []
for target in selected_targets:
endpoint = check_endpoint(target["base_url"], target["model"], args.timeout)
item = {
"id": target["id"],
"kind": target["kind"],
"base_url": target["base_url"],
"model": target["model"],
"endpoint": endpoint,
"eval": None,
"status": "blocked",
}
if endpoint["status"] == "ok":
item["eval"] = run_eval(
target["base_url"],
target["model"],
target["eval_report"],
system_prompt=args.system_prompt,
)
item["status"] = item["eval"]["status"]
results.append(item)
failed = [item["id"] for item in results if item["status"] == "failed"]
blocked = [item["id"] for item in results if item["status"] == "blocked"]
report = {
"created_at": dt.datetime.now(dt.UTC).isoformat(),
"profile": args.profile,
"all_targets": args.all_targets,
"system_prompt": str(args.system_prompt.relative_to(ROOT)) if args.system_prompt else None,
"status": "failed" if failed else "blocked" if blocked else "ok",
"failed": failed,
"blocked": blocked,
"results": results,
}
args.report.parent.mkdir(parents=True, exist_ok=True)
args.report.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
if args.print:
print(json.dumps(report, ensure_ascii=False, indent=2))
else:
print(f"Live eval status: {report['status']}")
print(f"Wrote report to {args.report}")
return 1 if report["status"] in {"failed", "blocked"} else 0
if __name__ == "__main__":
raise SystemExit(main())