feat: 平台治理与权限体系完善,存储进度/GPU预留/审批中心与日志整合
- 平台治理: 租户用户权限层次、资源ACL、审批中心与审批模板、访问申请 - 存储: MinIO 存储进度迁移、对象存储安全加固与测试 - 计算: GPU 资源预留、compute 轮询与同步增强 - 权限: permission v2 迁移、权限安全验收测试 - 日志: 后端运行日志中文说明、操作日志整合 - 数据处理/评测: 数据转换与模型评测优化 Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -14,6 +14,18 @@ from app.modules.storage.minio_store import get_object_storage
|
||||
MAX_STARTING_ATTEMPTS = 40
|
||||
|
||||
|
||||
def _extract_job_failure_reason(log_text: str, limit: int = 2000) -> str:
|
||||
"""Return a concise actionable reason from a failed Compute job log."""
|
||||
lines = [line.strip() for line in str(log_text or "").splitlines() if line.strip()]
|
||||
if not lines:
|
||||
return ""
|
||||
markers = ("[eval] FAILED", "Traceback", "RuntimeError", "Error:", "ERROR")
|
||||
for index in range(len(lines) - 1, -1, -1):
|
||||
if any(marker in lines[index] for marker in markers):
|
||||
return "\n".join(lines[index : index + 8])[-limit:]
|
||||
return "\n".join(lines[-8:])[-limit:]
|
||||
|
||||
|
||||
async def _archive_node_directory(
|
||||
store: Any,
|
||||
client: ComputeNodeClient,
|
||||
@@ -27,12 +39,15 @@ async def _archive_node_directory(
|
||||
"""Archive a completed node directory to MinIO, preserving subdirectories."""
|
||||
data_root = Path(str(node.get("data_root") or "/data/yg-ft")).resolve()
|
||||
source = Path(source_path).resolve()
|
||||
if source == data_root:
|
||||
raise RuntimeError("refuse to archive compute data root; output_dir must be a task subdirectory")
|
||||
try:
|
||||
relative_root = source.relative_to(data_root).as_posix()
|
||||
except ValueError as exc:
|
||||
raise RuntimeError(f"artifact path is outside compute data root: {source_path}") from exc
|
||||
queue = [relative_root]
|
||||
archived: list[dict[str, Any]] = []
|
||||
max_files = 10000
|
||||
while queue:
|
||||
relative = queue.pop(0)
|
||||
listing = await client.list_files(root="data", relative_path=relative)
|
||||
@@ -49,6 +64,8 @@ async def _archive_node_directory(
|
||||
except ValueError:
|
||||
relative_file = Path(str(item.get("name") or Path(path).name)).name
|
||||
object_key = f"{object_prefix}/{version_id}/{relative_file}"
|
||||
if len(archived) >= max_files:
|
||||
raise RuntimeError(f"archive file count exceeds limit {max_files}")
|
||||
upload_url = get_object_storage().presigned_put(object_key)
|
||||
result = await client.upload_file_to_url(path, upload_url, object_key)
|
||||
metadata = get_object_storage().stat(object_key)
|
||||
@@ -115,11 +132,13 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
|
||||
item["status"] = "error"
|
||||
item["error"] = "compute node deleted"
|
||||
store.mark_inference_unloaded(item.get("node_id") or "")
|
||||
store.release_external_gpus("inference", str(task["id"]), item.get("node_id"))
|
||||
continue
|
||||
if not node.get("enabled") or node.get("scheduler_status") != "online":
|
||||
item["status"] = "error"
|
||||
item["error"] = "compute node offline"
|
||||
store.mark_inference_unloaded(node["id"])
|
||||
store.release_external_gpus("inference", str(task["id"]), node["id"])
|
||||
continue
|
||||
try:
|
||||
status = await ComputeNodeClient(node["api_base_url"]).inference_status()
|
||||
@@ -128,6 +147,7 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
|
||||
item["status"] = "error"
|
||||
item["error"] = f"compute node unreachable: {exc}"
|
||||
store.mark_inference_unloaded(node["id"])
|
||||
store.release_external_gpus("inference", str(task["id"]), node["id"])
|
||||
continue
|
||||
node_status = status.get("status")
|
||||
if node_status == "ready":
|
||||
@@ -139,11 +159,13 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
|
||||
item["status"] = "error"
|
||||
item["error"] = status.get("error") or "model load failed on compute node"
|
||||
store.mark_inference_unloaded(node["id"])
|
||||
store.release_external_gpus("inference", str(task["id"]), node["id"])
|
||||
elif node_status == "idle":
|
||||
# 节点重启导致已加载模型丢失
|
||||
item["status"] = "error"
|
||||
item["error"] = "model disappeared from compute node (node may have restarted)"
|
||||
store.mark_inference_unloaded(node["id"])
|
||||
store.release_external_gpus("inference", str(task["id"]), node["id"])
|
||||
# node_status == "loading" -> 保持 starting,下轮再查
|
||||
if dirty:
|
||||
if any(i.get("status") in {"ready", "running"} for i in items):
|
||||
@@ -157,11 +179,16 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
|
||||
return reconciled
|
||||
|
||||
|
||||
async def fetch_eval_result_content(client: ComputeNodeClient, node: dict[str, Any], job: dict[str, Any]) -> dict[str, Any] | None:
|
||||
async def fetch_eval_result_content(
|
||||
client: ComputeNodeClient,
|
||||
node: dict[str, Any],
|
||||
job: dict[str, Any],
|
||||
file_name: str = "eval_results.json",
|
||||
) -> dict[str, Any] | None:
|
||||
output_dir = job.get("output_dir")
|
||||
if not output_dir:
|
||||
return None
|
||||
full_path = f"{str(output_dir).rstrip('/')}/eval_results.json"
|
||||
full_path = f"{str(output_dir).rstrip('/')}/{file_name}"
|
||||
data_root = "/data/yg-ft/"
|
||||
if full_path.startswith(data_root):
|
||||
full_path = full_path[len(data_root):]
|
||||
@@ -175,11 +202,36 @@ async def fetch_eval_result_content(client: ComputeNodeClient, node: dict[str, A
|
||||
return payload if isinstance(payload, dict) else None
|
||||
|
||||
|
||||
async def poll_compute_jobs_once() -> dict[str, Any]:
|
||||
store = get_platform_store()
|
||||
async def fetch_eval_progress_content(
|
||||
client: ComputeNodeClient,
|
||||
node: dict[str, Any],
|
||||
job: dict[str, Any],
|
||||
) -> dict[str, Any] | None:
|
||||
return await fetch_eval_result_content(client, node, job, "eval_progress.json")
|
||||
|
||||
|
||||
async def poll_compute_jobs_once(store: Any | None = None) -> dict[str, Any]:
|
||||
store = store or get_platform_store()
|
||||
synced: list[dict[str, Any]] = []
|
||||
failed: list[dict[str, str]] = []
|
||||
for task in store.running_compute_tasks():
|
||||
online_nodes = {
|
||||
str(node.get("id"))
|
||||
for node in store.compute_nodes()
|
||||
if node.get("enabled") and node.get("scheduler_status") in {"online", "draining"}
|
||||
}
|
||||
training_tasks = {str(task["id"]): task for task in store.running_compute_tasks()}
|
||||
# Completed tasks whose MinIO archive was interrupted remain eligible for
|
||||
# reconciliation after a Backend restart or a transient node failure.
|
||||
if get_settings().minio_enabled:
|
||||
for task in store.tasks():
|
||||
if task.get("status") != "completed" or not task.get("compute_job_id"):
|
||||
continue
|
||||
if (
|
||||
str(task.get("archive_status") or "") != "completed"
|
||||
and str(task.get("compute_node_id")) in online_nodes
|
||||
):
|
||||
training_tasks.setdefault(str(task["id"]), task)
|
||||
for task in training_tasks.values():
|
||||
node = _node_for_task(task)
|
||||
if not node:
|
||||
failed.append({"task_id": task["id"], "error": "compute node not found"})
|
||||
@@ -215,26 +267,47 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
|
||||
None,
|
||||
)
|
||||
if trained_model:
|
||||
archived = await _archive_node_directory(
|
||||
store,
|
||||
client,
|
||||
node,
|
||||
str(job["output_dir"]),
|
||||
"trained_model",
|
||||
str(trained_model["id"]),
|
||||
str(job.get("id") or task.get("compute_job_id") or task["id"]),
|
||||
f"trained_models/{trained_model['id']}",
|
||||
)
|
||||
artifacts = store.model_artifacts(str(trained_model["id"]))
|
||||
if archived and artifacts:
|
||||
store.link_model_artifact_storage_object(
|
||||
str(artifacts[0]["id"]), str(archived[0]["id"])
|
||||
try:
|
||||
archived = await _archive_node_directory(
|
||||
store,
|
||||
client,
|
||||
node,
|
||||
str(job["output_dir"]),
|
||||
"trained_model",
|
||||
str(trained_model["id"]),
|
||||
str(job.get("id") or task.get("compute_job_id") or task["id"]),
|
||||
f"trained_models/{trained_model['id']}",
|
||||
)
|
||||
artifacts = store.model_artifacts(str(trained_model["id"]))
|
||||
if archived and artifacts:
|
||||
store.link_model_artifact_storage_object(
|
||||
str(artifacts[0]["id"]), str(archived[0]["id"])
|
||||
)
|
||||
store.update_task(task["id"], {
|
||||
"archive_status": "completed",
|
||||
"archive_object_ids": [str(item["id"]) for item in archived],
|
||||
"archive_error": "",
|
||||
})
|
||||
except Exception as archive_exc:
|
||||
store.update_task(task["id"], {
|
||||
"archive_status": "pending",
|
||||
"archive_error": str(archive_exc)[:2000],
|
||||
})
|
||||
raise
|
||||
synced.append(updated_task)
|
||||
except Exception as exc: # noqa: BLE001 - keep polling other jobs
|
||||
failed.append({"task_id": task["id"], "error": str(exc)})
|
||||
standalone_synced: list[dict[str, Any]] = []
|
||||
for record in store.active_standalone_compute_jobs():
|
||||
standalone_jobs = {
|
||||
str(record["id"]): record
|
||||
for record in store.active_standalone_compute_jobs()
|
||||
if str(record.get("node_id")) in online_nodes
|
||||
}
|
||||
if get_settings().minio_enabled:
|
||||
for record in store.standalone_compute_jobs_pending_archive():
|
||||
if str(record.get("node_id")) in online_nodes:
|
||||
standalone_jobs.setdefault(str(record["id"]), record)
|
||||
for record in standalone_jobs.values():
|
||||
node = next((item for item in store.compute_nodes() if item["id"] == record.get("node_id")), None)
|
||||
if not node:
|
||||
failed.append({"job_id": record["id"], "error": "compute node not found"})
|
||||
@@ -266,12 +339,32 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
|
||||
store.link_model_artifact_storage_object(
|
||||
str(artifacts[0]["id"]), str(archived[0]["id"])
|
||||
)
|
||||
store.update_compute_job_archive(record["id"], "completed", [str(item["id"]) for item in archived])
|
||||
except Exception as exc: # noqa: BLE001 - keep polling other jobs
|
||||
try:
|
||||
store.update_compute_job_archive(record["id"], "pending", [], str(exc)[:2000])
|
||||
except Exception:
|
||||
pass
|
||||
failed.append({"job_id": record["id"], "error": str(exc)})
|
||||
|
||||
# ── Eval job sync ────────────────────────────────────────────────
|
||||
eval_synced = 0
|
||||
for eval_task in store.running_eval_tasks():
|
||||
eval_tasks = {str(task["id"]): task for task in store.running_eval_tasks()}
|
||||
if get_settings().minio_enabled:
|
||||
# A completed evaluation can win the race with the poller: its status
|
||||
# is persisted before the report archive finishes. Keep such tasks in
|
||||
# the reconciliation set until the report object is available.
|
||||
for task in store.eval_tasks():
|
||||
if (
|
||||
task.get("status") == "completed"
|
||||
and task.get("compute_job_id")
|
||||
and str(task.get("archive_status") or "") != "completed"
|
||||
and str(task.get("compute_node_id")) in online_nodes
|
||||
):
|
||||
eval_tasks.setdefault(str(task["id"]), task)
|
||||
for eval_task in eval_tasks.values():
|
||||
if str(eval_task.get("compute_node_id")) not in online_nodes:
|
||||
continue
|
||||
node = next(
|
||||
(item for item in store.compute_nodes() if item["id"] == eval_task.get("compute_node_id")),
|
||||
None,
|
||||
@@ -283,15 +376,39 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
|
||||
client = ComputeNodeClient(node["api_base_url"])
|
||||
job = await client.get_job(eval_task["compute_job_id"])
|
||||
result_content = None
|
||||
# Try to read eval_results.json from the job output directory
|
||||
# Read live progress and partial results while the evaluator is running.
|
||||
if job.get("status") in {"queued", "running"} and job.get("output_dir"):
|
||||
try:
|
||||
progress_content = await fetch_eval_progress_content(client, node, job)
|
||||
if progress_content:
|
||||
store.update_eval_task(
|
||||
eval_task["id"],
|
||||
{
|
||||
"progress_detail": progress_content,
|
||||
"progress": progress_content.get("percentage", eval_task.get("progress", 0)),
|
||||
},
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
result_content = await fetch_eval_result_content(client, node, job)
|
||||
except Exception:
|
||||
result_content = None
|
||||
# Try to read eval_results.json from the job output directory on completion.
|
||||
if job.get("status") == "completed" and job.get("output_dir"):
|
||||
try:
|
||||
result_content = await fetch_eval_result_content(client, node, job)
|
||||
except Exception:
|
||||
pass
|
||||
if job.get("status") in {"failed", "stopped"} and not job.get("error"):
|
||||
try:
|
||||
failure_logs = await client.job_logs(eval_task["compute_job_id"], tail_lines=120)
|
||||
job["error"] = _extract_job_failure_reason(str(failure_logs.get("content") or ""))
|
||||
except Exception:
|
||||
pass
|
||||
store.apply_eval_job_result(eval_task["id"], job, result_content)
|
||||
if get_settings().minio_enabled and job.get("status") == "completed" and job.get("output_dir"):
|
||||
await _archive_node_directory(
|
||||
archived = await _archive_node_directory(
|
||||
store,
|
||||
client,
|
||||
node,
|
||||
@@ -301,9 +418,28 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
|
||||
str(job.get("id") or eval_task.get("compute_job_id") or eval_task["id"]),
|
||||
f"evaluations/{eval_task['id']}",
|
||||
)
|
||||
report_object = next(
|
||||
(item for item in archived if Path(str(item.get("file_name") or "")).name == "eval_results.json"),
|
||||
archived[0] if archived else None,
|
||||
)
|
||||
store.update_eval_task(eval_task["id"], {
|
||||
"report_storage_object_id": str(report_object["id"]) if report_object else "",
|
||||
"archive_status": "completed",
|
||||
"archive_object_ids": [str(item["id"]) for item in archived],
|
||||
"archive_error": "",
|
||||
})
|
||||
# 评测 GPU 占用由 eval_tasks 状态派生,无需维护推理内存标记
|
||||
eval_synced += 1
|
||||
except Exception as exc: # noqa: BLE001
|
||||
try:
|
||||
current_eval = store.eval_task(eval_task["id"])
|
||||
except Exception:
|
||||
current_eval = eval_task
|
||||
if current_eval.get("status") == "completed":
|
||||
try:
|
||||
store.update_eval_task(eval_task["id"], {"archive_status": "pending", "archive_error": str(exc)[:2000]})
|
||||
except Exception:
|
||||
pass
|
||||
failed.append({"eval_task_id": eval_task["id"], "error": str(exc)})
|
||||
|
||||
# ── Inference load reconciliation ─────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user