feat: 平台治理与权限体系完善,存储进度/GPU预留/审批中心与日志整合

- 平台治理: 租户用户权限层次、资源ACL、审批中心与审批模板、访问申请
- 存储: MinIO 存储进度迁移、对象存储安全加固与测试
- 计算: GPU 资源预留、compute 轮询与同步增强
- 权限: permission v2 迁移、权限安全验收测试
- 日志: 后端运行日志中文说明、操作日志整合
- 数据处理/评测: 数据转换与模型评测优化

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
wuyongtao
2026-08-21 09:49:48 +08:00
parent 080ef6ab00
commit 6f0e82f351
94 changed files with 9547 additions and 1045 deletions

View File

@@ -14,6 +14,18 @@ from app.modules.storage.minio_store import get_object_storage
MAX_STARTING_ATTEMPTS = 40
def _extract_job_failure_reason(log_text: str, limit: int = 2000) -> str:
"""Return a concise actionable reason from a failed Compute job log."""
lines = [line.strip() for line in str(log_text or "").splitlines() if line.strip()]
if not lines:
return ""
markers = ("[eval] FAILED", "Traceback", "RuntimeError", "Error:", "ERROR")
for index in range(len(lines) - 1, -1, -1):
if any(marker in lines[index] for marker in markers):
return "\n".join(lines[index : index + 8])[-limit:]
return "\n".join(lines[-8:])[-limit:]
async def _archive_node_directory(
store: Any,
client: ComputeNodeClient,
@@ -27,12 +39,15 @@ async def _archive_node_directory(
"""Archive a completed node directory to MinIO, preserving subdirectories."""
data_root = Path(str(node.get("data_root") or "/data/yg-ft")).resolve()
source = Path(source_path).resolve()
if source == data_root:
raise RuntimeError("refuse to archive compute data root; output_dir must be a task subdirectory")
try:
relative_root = source.relative_to(data_root).as_posix()
except ValueError as exc:
raise RuntimeError(f"artifact path is outside compute data root: {source_path}") from exc
queue = [relative_root]
archived: list[dict[str, Any]] = []
max_files = 10000
while queue:
relative = queue.pop(0)
listing = await client.list_files(root="data", relative_path=relative)
@@ -49,6 +64,8 @@ async def _archive_node_directory(
except ValueError:
relative_file = Path(str(item.get("name") or Path(path).name)).name
object_key = f"{object_prefix}/{version_id}/{relative_file}"
if len(archived) >= max_files:
raise RuntimeError(f"archive file count exceeds limit {max_files}")
upload_url = get_object_storage().presigned_put(object_key)
result = await client.upload_file_to_url(path, upload_url, object_key)
metadata = get_object_storage().stat(object_key)
@@ -115,11 +132,13 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
item["status"] = "error"
item["error"] = "compute node deleted"
store.mark_inference_unloaded(item.get("node_id") or "")
store.release_external_gpus("inference", str(task["id"]), item.get("node_id"))
continue
if not node.get("enabled") or node.get("scheduler_status") != "online":
item["status"] = "error"
item["error"] = "compute node offline"
store.mark_inference_unloaded(node["id"])
store.release_external_gpus("inference", str(task["id"]), node["id"])
continue
try:
status = await ComputeNodeClient(node["api_base_url"]).inference_status()
@@ -128,6 +147,7 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
item["status"] = "error"
item["error"] = f"compute node unreachable: {exc}"
store.mark_inference_unloaded(node["id"])
store.release_external_gpus("inference", str(task["id"]), node["id"])
continue
node_status = status.get("status")
if node_status == "ready":
@@ -139,11 +159,13 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
item["status"] = "error"
item["error"] = status.get("error") or "model load failed on compute node"
store.mark_inference_unloaded(node["id"])
store.release_external_gpus("inference", str(task["id"]), node["id"])
elif node_status == "idle":
# 节点重启导致已加载模型丢失
item["status"] = "error"
item["error"] = "model disappeared from compute node (node may have restarted)"
store.mark_inference_unloaded(node["id"])
store.release_external_gpus("inference", str(task["id"]), node["id"])
# node_status == "loading" -> 保持 starting下轮再查
if dirty:
if any(i.get("status") in {"ready", "running"} for i in items):
@@ -157,11 +179,16 @@ async def reconcile_inference_loads(store: Any) -> list[dict[str, Any]]:
return reconciled
async def fetch_eval_result_content(client: ComputeNodeClient, node: dict[str, Any], job: dict[str, Any]) -> dict[str, Any] | None:
async def fetch_eval_result_content(
client: ComputeNodeClient,
node: dict[str, Any],
job: dict[str, Any],
file_name: str = "eval_results.json",
) -> dict[str, Any] | None:
output_dir = job.get("output_dir")
if not output_dir:
return None
full_path = f"{str(output_dir).rstrip('/')}/eval_results.json"
full_path = f"{str(output_dir).rstrip('/')}/{file_name}"
data_root = "/data/yg-ft/"
if full_path.startswith(data_root):
full_path = full_path[len(data_root):]
@@ -175,11 +202,36 @@ async def fetch_eval_result_content(client: ComputeNodeClient, node: dict[str, A
return payload if isinstance(payload, dict) else None
async def poll_compute_jobs_once() -> dict[str, Any]:
store = get_platform_store()
async def fetch_eval_progress_content(
client: ComputeNodeClient,
node: dict[str, Any],
job: dict[str, Any],
) -> dict[str, Any] | None:
return await fetch_eval_result_content(client, node, job, "eval_progress.json")
async def poll_compute_jobs_once(store: Any | None = None) -> dict[str, Any]:
store = store or get_platform_store()
synced: list[dict[str, Any]] = []
failed: list[dict[str, str]] = []
for task in store.running_compute_tasks():
online_nodes = {
str(node.get("id"))
for node in store.compute_nodes()
if node.get("enabled") and node.get("scheduler_status") in {"online", "draining"}
}
training_tasks = {str(task["id"]): task for task in store.running_compute_tasks()}
# Completed tasks whose MinIO archive was interrupted remain eligible for
# reconciliation after a Backend restart or a transient node failure.
if get_settings().minio_enabled:
for task in store.tasks():
if task.get("status") != "completed" or not task.get("compute_job_id"):
continue
if (
str(task.get("archive_status") or "") != "completed"
and str(task.get("compute_node_id")) in online_nodes
):
training_tasks.setdefault(str(task["id"]), task)
for task in training_tasks.values():
node = _node_for_task(task)
if not node:
failed.append({"task_id": task["id"], "error": "compute node not found"})
@@ -215,26 +267,47 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
None,
)
if trained_model:
archived = await _archive_node_directory(
store,
client,
node,
str(job["output_dir"]),
"trained_model",
str(trained_model["id"]),
str(job.get("id") or task.get("compute_job_id") or task["id"]),
f"trained_models/{trained_model['id']}",
)
artifacts = store.model_artifacts(str(trained_model["id"]))
if archived and artifacts:
store.link_model_artifact_storage_object(
str(artifacts[0]["id"]), str(archived[0]["id"])
try:
archived = await _archive_node_directory(
store,
client,
node,
str(job["output_dir"]),
"trained_model",
str(trained_model["id"]),
str(job.get("id") or task.get("compute_job_id") or task["id"]),
f"trained_models/{trained_model['id']}",
)
artifacts = store.model_artifacts(str(trained_model["id"]))
if archived and artifacts:
store.link_model_artifact_storage_object(
str(artifacts[0]["id"]), str(archived[0]["id"])
)
store.update_task(task["id"], {
"archive_status": "completed",
"archive_object_ids": [str(item["id"]) for item in archived],
"archive_error": "",
})
except Exception as archive_exc:
store.update_task(task["id"], {
"archive_status": "pending",
"archive_error": str(archive_exc)[:2000],
})
raise
synced.append(updated_task)
except Exception as exc: # noqa: BLE001 - keep polling other jobs
failed.append({"task_id": task["id"], "error": str(exc)})
standalone_synced: list[dict[str, Any]] = []
for record in store.active_standalone_compute_jobs():
standalone_jobs = {
str(record["id"]): record
for record in store.active_standalone_compute_jobs()
if str(record.get("node_id")) in online_nodes
}
if get_settings().minio_enabled:
for record in store.standalone_compute_jobs_pending_archive():
if str(record.get("node_id")) in online_nodes:
standalone_jobs.setdefault(str(record["id"]), record)
for record in standalone_jobs.values():
node = next((item for item in store.compute_nodes() if item["id"] == record.get("node_id")), None)
if not node:
failed.append({"job_id": record["id"], "error": "compute node not found"})
@@ -266,12 +339,32 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
store.link_model_artifact_storage_object(
str(artifacts[0]["id"]), str(archived[0]["id"])
)
store.update_compute_job_archive(record["id"], "completed", [str(item["id"]) for item in archived])
except Exception as exc: # noqa: BLE001 - keep polling other jobs
try:
store.update_compute_job_archive(record["id"], "pending", [], str(exc)[:2000])
except Exception:
pass
failed.append({"job_id": record["id"], "error": str(exc)})
# ── Eval job sync ────────────────────────────────────────────────
eval_synced = 0
for eval_task in store.running_eval_tasks():
eval_tasks = {str(task["id"]): task for task in store.running_eval_tasks()}
if get_settings().minio_enabled:
# A completed evaluation can win the race with the poller: its status
# is persisted before the report archive finishes. Keep such tasks in
# the reconciliation set until the report object is available.
for task in store.eval_tasks():
if (
task.get("status") == "completed"
and task.get("compute_job_id")
and str(task.get("archive_status") or "") != "completed"
and str(task.get("compute_node_id")) in online_nodes
):
eval_tasks.setdefault(str(task["id"]), task)
for eval_task in eval_tasks.values():
if str(eval_task.get("compute_node_id")) not in online_nodes:
continue
node = next(
(item for item in store.compute_nodes() if item["id"] == eval_task.get("compute_node_id")),
None,
@@ -283,15 +376,39 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
client = ComputeNodeClient(node["api_base_url"])
job = await client.get_job(eval_task["compute_job_id"])
result_content = None
# Try to read eval_results.json from the job output directory
# Read live progress and partial results while the evaluator is running.
if job.get("status") in {"queued", "running"} and job.get("output_dir"):
try:
progress_content = await fetch_eval_progress_content(client, node, job)
if progress_content:
store.update_eval_task(
eval_task["id"],
{
"progress_detail": progress_content,
"progress": progress_content.get("percentage", eval_task.get("progress", 0)),
},
)
except Exception:
pass
try:
result_content = await fetch_eval_result_content(client, node, job)
except Exception:
result_content = None
# Try to read eval_results.json from the job output directory on completion.
if job.get("status") == "completed" and job.get("output_dir"):
try:
result_content = await fetch_eval_result_content(client, node, job)
except Exception:
pass
if job.get("status") in {"failed", "stopped"} and not job.get("error"):
try:
failure_logs = await client.job_logs(eval_task["compute_job_id"], tail_lines=120)
job["error"] = _extract_job_failure_reason(str(failure_logs.get("content") or ""))
except Exception:
pass
store.apply_eval_job_result(eval_task["id"], job, result_content)
if get_settings().minio_enabled and job.get("status") == "completed" and job.get("output_dir"):
await _archive_node_directory(
archived = await _archive_node_directory(
store,
client,
node,
@@ -301,9 +418,28 @@ async def poll_compute_jobs_once() -> dict[str, Any]:
str(job.get("id") or eval_task.get("compute_job_id") or eval_task["id"]),
f"evaluations/{eval_task['id']}",
)
report_object = next(
(item for item in archived if Path(str(item.get("file_name") or "")).name == "eval_results.json"),
archived[0] if archived else None,
)
store.update_eval_task(eval_task["id"], {
"report_storage_object_id": str(report_object["id"]) if report_object else "",
"archive_status": "completed",
"archive_object_ids": [str(item["id"]) for item in archived],
"archive_error": "",
})
# 评测 GPU 占用由 eval_tasks 状态派生,无需维护推理内存标记
eval_synced += 1
except Exception as exc: # noqa: BLE001
try:
current_eval = store.eval_task(eval_task["id"])
except Exception:
current_eval = eval_task
if current_eval.get("status") == "completed":
try:
store.update_eval_task(eval_task["id"], {"archive_status": "pending", "archive_error": str(exc)[:2000]})
except Exception:
pass
failed.append({"eval_task_id": eval_task["id"], "error": str(exc)})
# ── Inference load reconciliation ─────────────────────────────────────