from __future__ import annotations import argparse import json import sys import time import urllib.error import urllib.request from typing import Any def request(method: str, url: str, payload: dict[str, Any] | None = None, timeout: int = 20) -> tuple[int, Any]: data = None headers = {"Content-Type": "application/json"} if payload is not None: data = json.dumps(payload).encode("utf-8") req = urllib.request.Request(url, data=data, method=method, headers=headers) with urllib.request.urlopen(req, timeout=timeout) as resp: raw = resp.read().decode("utf-8") try: body: Any = json.loads(raw) except json.JSONDecodeError: body = raw return resp.status, body def check(name: str, fn) -> dict[str, Any]: started = time.perf_counter() try: data = fn() return { "name": name, "ok": True, "duration_ms": int((time.perf_counter() - started) * 1000), "data": data, } except Exception as exc: # noqa: BLE001 - verification should report every failure shape. error = str(exc) if isinstance(exc, urllib.error.HTTPError): body = exc.read().decode("utf-8", errors="replace") error = f"HTTP Error {exc.code}: {body or exc.reason}" return { "name": name, "ok": False, "duration_ms": int((time.perf_counter() - started) * 1000), "error": error, } def unwrap(payload: Any) -> Any: if isinstance(payload, dict) and payload.get("code") == 0: return payload.get("data") return payload def post_json(url: str, payload: dict[str, Any], timeout: int = 20) -> Any: return unwrap(request("POST", url, payload, timeout=timeout)[1]) def get_json(url: str, timeout: int = 20) -> Any: return unwrap(request("GET", url, timeout=timeout)[1]) def run_training_smoke(backend_url: str) -> dict[str, Any]: suffix = str(int(time.time())) nodes = get_json(f"{backend_url}/compute/nodes", timeout=10) or [] if not nodes: raise RuntimeError("no compute node found") node = next((item for item in nodes if item.get("enabled") and item.get("scheduler_status") == "online"), nodes[0]) gpus = get_json(f"{backend_url}/compute/gpus", timeout=10) or [] gpu_id = int((gpus[0] or {}).get("id", 0)) if gpus else 0 model = post_json( f"{backend_url}/model-manage", { "name": f"smoke-base-{suffix}", "type": "LLM", "purpose": "training", "model_source": "local", "path": "/data/yg-ft/models/smoke-base", "description": "Smoke verification model path.", }, timeout=10, ) dataset_id = post_json( f"{backend_url}/dataset-manage", { "name": f"smoke-dataset-{suffix}", "type": "train", "storage_type": "local", "source": "smoke", "count": 2, "description": "Smoke verification dataset.", }, timeout=10, )["id"] task_id = post_json( f"{backend_url}/fine-tune", { "name": f"smoke-train-{suffix}", "description": "Automated smoke fine-tune job.", "engine": "smoke", "train_type": "SFT", "train_method": "lora", "template": "qwen", "base_model": model["id"], "train_dataset_id": dataset_id, "requested_node_id": node["id"], "gpus": [gpu_id], "batch_size": 1, "learning_rate": 0.0002, "n_epochs": 1, "save_steps": 2, "quantization_bit": 0, }, timeout=10, )["id"] command_preview = post_json( f"{backend_url}/fine-tune/{task_id}/command-preview", {"requested_node_id": node["id"], "gpus": [gpu_id]}, timeout=30, ) if not command_preview.get("preview", {}).get("command"): raise RuntimeError(f"command preview is empty: {command_preview}") preflight = post_json( f"{backend_url}/fine-tune/{task_id}/preflight", {"requested_node_id": node["id"], "gpus": [gpu_id]}, timeout=30, ) if not preflight.get("valid"): raise RuntimeError(f"fine-tune preflight failed: {preflight}") started = post_json( f"{backend_url}/fine-tune/start", {"task_id": task_id, "requested_node_id": node["id"], "gpus": [gpu_id]}, timeout=30, ) compute_job_id = started.get("compute_job_id") if not compute_job_id: raise RuntimeError(f"task did not create compute_job_id: {started}") task = started poll_items: list[dict[str, Any]] = [] deadline = time.time() + 45 while time.time() < deadline: with_context = post_json(f"{backend_url}/internal/compute-sync/jobs/poll", {}, timeout=20) task = get_json(f"{backend_url}/fine-tune/{task_id}", timeout=10) poll_items.append( { "status": task.get("status"), "progress": task.get("progress"), "compute_job_id": task.get("compute_job_id"), "synced": with_context.get("synced") if isinstance(with_context, dict) else None, } ) if task.get("status") in {"completed", "failed", "stopped"}: break time.sleep(1) logs = get_json(f"{backend_url}/compute/jobs/{compute_job_id}/logs?tail_lines=80", timeout=20) if task.get("status") != "completed": raise RuntimeError({"task": task, "poll": poll_items[-8:], "logs": logs}) if not logs.get("content"): raise RuntimeError("compute job logs are empty") return { "task_id": task_id, "compute_job_id": compute_job_id, "node_id": node["id"], "gpu_id": gpu_id, "status": task.get("status"), "progress": task.get("progress"), "artifact_count": len(task.get("artifacts") or []), "log_lines": logs.get("total_lines"), "metric_count": len(logs.get("metrics") or []), "preflight_valid": preflight.get("valid"), "command": command_preview.get("preview", {}).get("command"), "poll": poll_items[-8:], } def main() -> int: parser = argparse.ArgumentParser(description="Verify YG Zhilian deployment.") parser.add_argument("--frontend-url", default="http://localhost:16801", help="Frontend base URL.") parser.add_argument("--backend-url", default="http://localhost:17861/modelTF", help="Backend API base URL with /modelTF.") parser.add_argument("--username", default="admin", help="Login username.") parser.add_argument("--password", default="admin123", help="Login password.") parser.add_argument("--skip-compute-test", action="store_true", help="Skip active compute node connection tests.") parser.add_argument("--run-training-smoke", action="store_true", help="Create and run a short smoke fine-tune job.") args = parser.parse_args() frontend_url = args.frontend_url.rstrip("/") backend_url = args.backend_url.rstrip("/") report: list[dict[str, Any]] = [] report.append( check( "frontend index", lambda: {"status": request("GET", f"{frontend_url}/index.html", timeout=5)[0]}, ) ) report.append( check( "backend health", lambda: unwrap(request("GET", f"{backend_url}/health", timeout=10)[1]), ) ) report.append( check( "admin login", lambda: unwrap( request( "POST", f"{backend_url}/login", {"username": args.username, "password": args.password}, timeout=10, )[1] ), ) ) nodes_result: dict[str, Any] = {"nodes": []} def load_nodes() -> dict[str, Any]: nodes = unwrap(request("GET", f"{backend_url}/compute/nodes", timeout=10)[1]) or [] nodes_result["nodes"] = nodes return {"count": len(nodes), "nodes": [{"id": n.get("id"), "code": n.get("code"), "api_base_url": n.get("api_base_url")} for n in nodes]} report.append(check("compute nodes", load_nodes)) if not args.skip_compute_test: for node in nodes_result["nodes"]: node_id = node.get("id") node_code = node.get("code") if not node_id: continue report.append( check( f"compute node test {node_code}", lambda node_id=node_id: unwrap( request("POST", f"{backend_url}/compute/nodes/{node_id}/test-connection", timeout=30)[1] ), ) ) report.append( check( "compute gpus", lambda: {"count": len(unwrap(request("GET", f"{backend_url}/compute/gpus", timeout=10)[1]) or [])}, ) ) report.append( check( "compute queue", lambda: {"count": len(unwrap(request("GET", f"{backend_url}/compute/queue", timeout=10)[1]) or [])}, ) ) if args.run_training_smoke: report.append(check("fine-tune smoke job", lambda: run_training_smoke(backend_url))) ok = all(item["ok"] for item in report) print(json.dumps({"ok": ok, "checks": report}, ensure_ascii=False, indent=2)) return 0 if ok else 1 if __name__ == "__main__": sys.exit(main())