Files
llm/scripts/collect_platform_status.py
T

157 lines
5.1 KiB
Python

from __future__ import annotations
import argparse
import datetime as dt
import json
import socket
import subprocess
import sys
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
DEFAULT_REPORT = ROOT / "reports" / "platform-status.json"
DEFAULT_DOCKER_HOST = "ssh://docker-gpu"
MODEL_CHAT_CONTAINER = "llm-model-chat-ui"
def run_json(command: list[str], timeout: int = 60) -> dict[str, Any]:
result = subprocess.run(
command,
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
timeout=timeout,
check=False,
)
payload = None
if result.stdout.strip().startswith("{"):
try:
payload = json.loads(result.stdout)
except json.JSONDecodeError:
payload = None
return {
"command": command,
"status": "ok" if result.returncode == 0 else "failed",
"returncode": result.returncode,
"stdout": result.stdout.strip(),
"stderr": result.stderr.strip(),
"json": payload,
}
def read_report(path: Path) -> dict[str, Any] | None:
if not path.exists():
return None
try:
with path.open("r", encoding="utf-8") as handle:
data = json.load(handle)
return data if isinstance(data, dict) else None
except json.JSONDecodeError:
return None
def is_port_open(host: str, port: int, timeout: float = 1.0) -> bool:
try:
with socket.create_connection((host, port), timeout=timeout):
return True
except OSError:
return False
def deployed_model_storage(docker_host: str) -> dict[str, Any]:
return run_json(
[
"docker",
"-H",
docker_host,
"exec",
MODEL_CHAT_CONTAINER,
"python3",
"scripts/check_model_storage.py",
"--models-root",
"/models",
"--print",
"--no-report",
"--warn-only",
],
timeout=90,
)
def main() -> int:
parser = argparse.ArgumentParser(description="Collect local platform status into one report.")
parser.add_argument("--report", type=Path, default=DEFAULT_REPORT)
parser.add_argument("--print", action="store_true")
parser.add_argument("--check-endpoints", action="store_true", help="Run GPU readiness check with network timeouts.")
parser.add_argument("--docker-host", default=DEFAULT_DOCKER_HOST, help="Docker endpoint for deployed GPU checks.")
parser.add_argument(
"--skip-deployed-storage",
action="store_true",
help="Skip model storage check inside the deployed model-chat container.",
)
args = parser.parse_args()
model_storage = run_json(
[sys.executable, "scripts/check_model_storage.py", "--print", "--no-report", "--warn-only"],
timeout=60,
)
if model_storage.get("json", {}).get("status") == "failed":
model_storage["status"] = "blocked"
checks = {
"model_cards": run_json([sys.executable, "scripts/validate_model_cards.py"], timeout=30),
"eval_files": run_json([sys.executable, "scripts/validate_evals.py"], timeout=30),
"model_storage": model_storage,
"deployed_model_storage": None if args.skip_deployed_storage else deployed_model_storage(args.docker_host),
"1c_plugin": run_json([sys.executable, "scripts/check_1c_plugin.py", "--no-report"], timeout=60),
"model_chat": {
"status": "ok" if is_port_open("127.0.0.1", 8765) else "stopped",
"url": "http://127.0.0.1:8765",
},
"gpu_readiness": read_report(ROOT / "reports" / "gpu-readiness.json"),
"live_model_evals": read_report(ROOT / "reports" / "evals" / "live-model-evals.json"),
}
if args.check_endpoints:
checks["gpu_readiness"] = run_json(
[sys.executable, "scripts/check_gpu_readiness.py", "--print", "--timeout", "8"],
timeout=45,
).get("json")
if isinstance(checks["gpu_readiness"], dict) and checks["gpu_readiness"].get("status") == "failed":
checks["gpu_readiness"]["status"] = "blocked"
failed = []
blocked = []
for name, check in checks.items():
if not check:
if name != "deployed_model_storage":
blocked.append(name)
continue
status = check.get("status")
if status == "failed":
failed.append(name)
elif status in {"blocked", "stopped"}:
blocked.append(name)
report = {
"created_at": dt.datetime.now(dt.UTC).isoformat(),
"status": "failed" if failed else "blocked" if blocked else "ok",
"failed": failed,
"blocked": blocked,
"checks": checks,
}
args.report.parent.mkdir(parents=True, exist_ok=True)
args.report.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
if args.print:
print(json.dumps(report, ensure_ascii=False, indent=2))
else:
print(f"Platform status: {report['status']}")
print(f"Wrote report to {args.report}")
return 1 if failed else 0
if __name__ == "__main__":
raise SystemExit(main())