# Conflicts:
#	backend/app/api/v1/endpoints/platform.py
#	compute/requirements.txt
This commit is contained in:
wuyongtao
2026-08-03 09:42:49 +08:00
67 changed files with 7775 additions and 653 deletions

View File

@@ -8,12 +8,14 @@ import os
import re import re
import socket import socket
import time import time
from collections.abc import Iterator, Mapping
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from contextlib import contextmanager from contextlib import contextmanager
from copy import deepcopy
from dataclasses import asdict from dataclasses import asdict
from pathlib import Path from pathlib import Path
from threading import BoundedSemaphore, Lock from threading import BoundedSemaphore, Lock
from typing import Any, Iterator, Literal from typing import Any, Literal
from urllib.parse import quote, urlsplit from urllib.parse import quote, urlsplit
import httpx import httpx
@@ -32,6 +34,7 @@ from fastapi import (
from fastapi.responses import StreamingResponse from fastapi.responses import StreamingResponse
from psycopg.rows import dict_row from psycopg.rows import dict_row
from app.core.auth import filter_accessible_resource_ids, get_current_user, is_admin
from app.modules.data_process.algorithms import ( from app.modules.data_process.algorithms import (
ParsedText, ParsedText,
canonical_record_json, canonical_record_json,
@@ -45,9 +48,10 @@ from app.modules.data_process.algorithms import (
is_near_duplicate, is_near_duplicate,
near_duplicate_fingerprint, near_duplicate_fingerprint,
parse_text_content, parse_text_content,
preprocess_structured_records, preprocess_structured_records_with_lineage,
remove_document_noise, remove_document_noise,
score_quality, score_quality,
structured_json_dumps,
) )
from app.modules.data_process.document_chunking import ( from app.modules.data_process.document_chunking import (
DocumentChunk, DocumentChunk,
@@ -75,9 +79,11 @@ from app.modules.data_process.store import (
NotFoundError, NotFoundError,
get_data_process_store, get_data_process_store,
new_id, new_id,
repeat_task_id,
) )
from app.schemas.data_process import ( from app.schemas.data_process import (
DataProcessRegenerateRequest, DataProcessRegenerateRequest,
DataProcessRepeatRequest,
DataProcessStatus, DataProcessStatus,
DataProcessTaskCreate, DataProcessTaskCreate,
DataProcessTaskUpdate, DataProcessTaskUpdate,
@@ -261,7 +267,14 @@ def _parse_stored_source(source: dict[str, Any]) -> ParsedText:
content = str(source.get("content") or "") content = str(source.get("content") or "")
file_format = str(source.get("file_format") or "").lower() file_format = str(source.get("file_format") or "").lower()
if file_format == "xlsx": if file_format == "xlsx":
# XLSX 上传阶段已安全解析为 JSONL 后入库。 raw_content = source.get("raw_content")
if isinstance(raw_content, bytes):
return parse_text_content(
raw_content,
filename=str(source.get("name") or "source.xlsx"),
file_format="xlsx",
)
# 兼容原始对象已缺失的历史文件:退化为上传阶段生成的 JSONL。
return parse_text_content(content, file_format="jsonl") return parse_text_content(content, file_format="jsonl")
if file_format in {"pdf", "docx", "pptx"}: if file_format in {"pdf", "docx", "pptx"}:
# 文档上传阶段已抽取文本,预览阶段只需要对正文切片。 # 文档上传阶段已抽取文本,预览阶段只需要对正文切片。
@@ -381,11 +394,12 @@ def _build_preview_items(
seen_near_duplicate_bands: dict[tuple[int, int], list[str]] = {} seen_near_duplicate_bands: dict[tuple[int, int], list[str]] = {}
items: list[dict[str, Any]] = [] items: list[dict[str, Any]] = []
def append_item(item: dict[str, Any]) -> None: def append_item(item: dict[str, Any], *, dedup_content: str) -> None:
content = str(item.get("edited_content") or "").strip() content = str(item.get("edited_content") or "").strip()
if should_clean_invalid and not content: if should_clean_invalid and not content:
return return
content_hash = hashlib.sha256(content.encode("utf-8")).hexdigest() # 去重必须基于脱敏前内容,否则不同原文可能在替换 PII 后被错误合并。
content_hash = hashlib.sha256(dedup_content.strip().encode("utf-8")).hexdigest()
if should_deduplicate and content_hash in seen_content_hashes: if should_deduplicate and content_hash in seen_content_hashes:
return return
seen_content_hashes.add(content_hash) seen_content_hashes.add(content_hash)
@@ -447,6 +461,7 @@ def _build_preview_items(
continue continue
for key in band_keys: for key in band_keys:
seen_near_duplicate_bands.setdefault(key, []).append(content) seen_near_duplicate_bands.setdefault(key, []).append(content)
dedup_content = content
pii_counts: dict[str, int] = {} pii_counts: dict[str, int] = {}
if should_desensitize: if should_desensitize:
content, pii_counts = desensitize_pii(content) content, pii_counts = desensitize_pii(content)
@@ -476,7 +491,8 @@ def _build_preview_items(
else "original" else "original"
), ),
"quality_score": quality, "quality_score": quality,
} },
dedup_content=dedup_content,
) )
continue continue
@@ -488,48 +504,76 @@ def _build_preview_items(
"filter_anomaly", "filter_anomaly",
} }
source_records = list(parsed.records) source_records = list(parsed.records)
processed_records = preprocess_structured_records( processed_records = preprocess_structured_records_with_lineage(
source_records, source_records,
structured_options, structured_options,
) )
if not processed_records and parsed.text and not source_records: for processed in processed_records:
processed_records = [{"value": parsed.text}] source_index = processed.source_index
same_cardinality = len(processed_records) == len(source_records) record = processed.record
for index, record in enumerate(processed_records): original_record = (
original_record = source_records[index] if same_cardinality else record source_records[source_index]
original_content = json.dumps( if source_index < len(source_records)
original_record, else record
ensure_ascii=False,
separators=(",", ":"),
) )
source_locator = (
deepcopy(parsed.record_locators[source_index])
if source_index < len(parsed.record_locators)
else None
)
original_content = structured_json_dumps(original_record)
pii_counts: dict[str, int] = {} pii_counts: dict[str, int] = {}
edited_record = record edited_record = record
dedup_content = (
canonical_record_json(record)
if "normalize_format" in preprocess_options
else structured_json_dumps(record)
)
if should_desensitize: if should_desensitize:
edited_record, pii_counts = desensitize_structured_record(record) edited_record, pii_counts = desensitize_structured_record(record)
content = ( content = (
canonical_record_json(edited_record) canonical_record_json(edited_record)
if "normalize_format" in preprocess_options if "normalize_format" in preprocess_options
else json.dumps( else structured_json_dumps(edited_record)
edited_record,
ensure_ascii=False,
separators=(",", ":"),
)
) )
quality = _preview_quality(content, config) quality = _preview_quality(content, config)
quality["pii_replacements"] = pii_counts quality["pii_replacements"] = pii_counts
if source_locator is not None:
quality["source_locator"] = source_locator
source_start = (
source_locator.get("source_start")
if source_locator is not None
else None
)
source_end = (
source_locator.get("source_end")
if source_locator is not None
else None
)
source_start_line = (
source_locator.get("start_line")
if source_locator is not None
else None
)
source_end_line = (
source_locator.get("end_line")
if source_locator is not None
else None
)
append_item( append_item(
{ {
"source_file_id": source["id"], "source_file_id": source["id"],
"original_content": original_content, "original_content": original_content,
"edited_content": content, "edited_content": content,
"source_start": None, "source_start": source_start,
"source_end": None, "source_end": source_end,
"source_start_line": None, "source_start_line": source_start_line,
"source_end_line": None, "source_end_line": source_end_line,
"token_count": estimate_token_count(content), "token_count": estimate_token_count(content),
"status": "modified" if content != original_content else "original", "status": "modified" if content != original_content else "original",
"quality_score": quality, "quality_score": quality,
} },
dedup_content=dedup_content,
) )
return items return items
@@ -772,18 +816,33 @@ def list_tasks(
keyword: str | None = Query(default=None), keyword: str | None = Query(default=None),
status: DataProcessStatus | None = Query(default=None), status: DataProcessStatus | None = Query(default=None),
process_type: ProcessType | None = Query(default=None), process_type: ProcessType | None = Query(default=None),
tenant_id: str | None = Query(default=None),
project_id: str | None = Query(default=None),
store: DataProcessStore = Depends(get_data_process_store), store: DataProcessStore = Depends(get_data_process_store),
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]: ) -> dict[str, Any]:
with api_errors(): with api_errors():
return ok( tasks = store.list_tasks(
store.list_tasks(
page=page, page=page,
page_size=page_size, page_size=page_size,
keyword=keyword, keyword=keyword,
status=status, status=status,
process_type=process_type, process_type=process_type,
tenant_id=tenant_id,
project_id=project_id,
)
# #4 资源 ACL 过滤admin 放行,普通用户只看到自己被授权的数据处理任务
items = tasks.get("items", [])
if not is_admin(current_user) and items:
accessible_ids = set(
filter_accessible_resource_ids(
"data-process", [t["id"] for t in items], current_user
) )
) )
items = [t for t in items if t["id"] in accessible_ids]
tasks["items"] = items
tasks["total"] = len(items)
return ok(tasks)
@router.post("") @router.post("")
@@ -850,6 +909,135 @@ def prepare_regeneration(
) )
def _repeat_file_copies(
store: DataProcessStore,
storage: LocalDataProcessStorage,
source_task_id: str,
request_id: str,
) -> tuple[dict[str, dict[str, str]], list[StagedSourceObject]]:
"""为新任务创建独立的源文件引用,避免删除任一任务时互相影响。"""
target_task_id = repeat_task_id(source_task_id, request_id)
copies: dict[str, dict[str, str]] = {}
staged: list[StagedSourceObject] = []
batch_id = storage.new_batch_id()
for summary in store.list_source_files(source_task_id):
old_file_id = str(summary["id"])
source = store.get_source_file(source_task_id, old_file_id, include_content=True)
new_file_id = new_id("dpsf")
old_reference = str(source.get("storage_object_id") or "")
if old_reference.startswith("local://data-process/"):
staged_object = storage.stage_copy(
batch_id=batch_id,
source_reference=old_reference,
expected_source_task_id=source_task_id,
expected_source_file_id=old_file_id,
task_id=target_task_id,
source_file_id=new_file_id,
version=1,
name=str(source["name"]),
)
staged.append(staged_object)
new_reference = staged_object.reference
elif old_reference.startswith("db://data-process/") or not old_reference:
new_reference = f"db://data-process/{target_task_id}/{new_file_id}/v1"
else:
raise ValueError("源任务包含不受支持的文件存储引用")
copies[old_file_id] = {
"id": new_file_id,
"storage_object_id": new_reference,
}
return copies, staged
def _remove_repeated_storage_objects(
storage: LocalDataProcessStorage,
task_id: str,
staged: list[StagedSourceObject],
copies: dict[str, dict[str, str]],
) -> None:
source_file_ids = {
str(copy["storage_object_id"]): str(copy["id"])
for copy in copies.values()
}
for item in staged:
try:
storage.delete(
item.reference,
expected_task_id=task_id,
expected_source_file_id=source_file_ids[item.reference],
)
except Exception:
logger.exception(
"failed to roll back repeated data process source object task_id=%s",
task_id,
)
@router.post("/{task_id}/repeat", status_code=202)
def repeat_generation(
task_id: str,
payload: DataProcessRepeatRequest,
background_tasks: BackgroundTasks,
store: DataProcessStore = Depends(get_data_process_store),
storage: LocalDataProcessStorage = Depends(get_data_process_storage),
) -> dict[str, Any]:
"""按原任务快照创建独立任务,并立即在后台开始新一批生成。"""
with api_errors():
repeated = store.find_repeated_task(task_id, payload.request_id)
staged: list[StagedSourceObject] = []
target_task_id = repeat_task_id(task_id, payload.request_id)
if repeated is None:
copies, staged = _repeat_file_copies(
store,
storage,
task_id,
payload.request_id,
)
storage.publish(staged)
try:
repeated = store.repeat_task(
task_id,
expected_updated_at=payload.expected_updated_at,
request_id=payload.request_id,
file_copies=copies,
)
except Exception:
_remove_repeated_storage_objects(
storage,
target_task_id,
staged,
copies,
)
raise
if not repeated["created"]:
_remove_repeated_storage_objects(
storage,
target_task_id,
staged,
copies,
)
repeated_task = repeated["task"]
if repeated_task.get("status") == "pending":
try:
started = store.start_generation(target_task_id, replace_existing=True)
background_tasks.add_task(
_run_generation,
store,
target_task_id,
str(started["generation_run_id"]),
)
except ConflictError:
latest = store.get_task(target_task_id)
if latest.get("status") != "running":
raise
repeated["task"] = store.get_task(target_task_id)
repeated["progress"] = store.progress(target_task_id)
return ok(repeated, "已按原配置创建新任务并开始后台生成")
@router.delete("/{task_id}") @router.delete("/{task_id}")
def delete_task( def delete_task(
task_id: str, task_id: str,
@@ -919,7 +1107,7 @@ async def upload_source_files(
f"{suffix} is not supported for {process_type} data processing", f"{suffix} is not supported for {process_type} data processing",
) )
parsed = parse_text_content(raw, filename=name) parsed = parse_text_content(raw, filename=name)
if not parsed.text: if not parsed.text.strip():
raise fail(400, f"source file is empty: {name}") raise fail(400, f"source file is empty: {name}")
batch_size += len(raw) batch_size += len(raw)
if batch_size > MAX_SOURCE_BATCH_BYTES: if batch_size > MAX_SOURCE_BATCH_BYTES:
@@ -934,7 +1122,11 @@ async def upload_source_files(
content=raw, content=raw,
) )
staged.append(staged_object) staged.append(staged_object)
record_count = len(parsed.records) or (1 if parsed.text else 0) record_count = (
len(parsed.records)
if process_type == "structured"
else (1 if parsed.text else 0)
)
prepared.append( prepared.append(
{ {
"id": source_file_id, "id": source_file_id,
@@ -1362,15 +1554,28 @@ def _prepare_preview_items(
_value(config, "chunk_method", "chunkMethod", "layout_hybrid") _value(config, "chunk_method", "chunkMethod", "layout_hybrid")
) )
is_unstructured = task.get("process_type") == "unstructured" is_unstructured = task.get("process_type") == "unstructured"
if is_unstructured and ( needs_unstructured_raw = is_unstructured and (
chunk_method == "layout_hybrid" chunk_method == "layout_hybrid"
or preprocess_options & {"clean_invalid", "clean_invalid_content"} or preprocess_options & {"clean_invalid", "clean_invalid_content"}
): )
has_structured_xlsx = not is_unstructured and any(
str(source.get("file_format") or "").lower() == "xlsx"
for source in sources
)
if needs_unstructured_raw or has_structured_xlsx:
for index, source in enumerate(sources): for index, source in enumerate(sources):
if ( source_format = str(source.get("file_format") or "").lower()
chunk_method != "layout_hybrid" needs_structured_xlsx = not is_unstructured and source_format == "xlsx"
and str(source.get("file_format") or "").lower() != "pdf" needs_layout_raw = is_unstructured and chunk_method == "layout_hybrid"
): needs_pdf_noise = (
is_unstructured
and not needs_layout_raw
and source_format == "pdf"
and bool(
preprocess_options & {"clean_invalid", "clean_invalid_content"}
)
)
if not (needs_structured_xlsx or needs_layout_raw or needs_pdf_noise):
continue continue
storage_object_id = str(source.get("storage_object_id") or "") storage_object_id = str(source.get("storage_object_id") or "")
actual_size = storage.file_size( actual_size = storage.file_size(
@@ -1379,7 +1584,7 @@ def _prepare_preview_items(
expected_source_file_id=str(source["id"]), expected_source_file_id=str(source["id"]),
) )
if actual_size is None: if actual_size is None:
if chunk_method == "layout_hybrid": if needs_layout_raw:
raise InvalidStateError( raise InvalidStateError(
"版面结构混合切分无法读取原始文件,请重新上传后再处理" "版面结构混合切分无法读取原始文件,请重新上传后再处理"
) )
@@ -1396,12 +1601,10 @@ def _prepare_preview_items(
) )
) )
enriched = dict(source) enriched = dict(source)
if chunk_method == "layout_hybrid": if needs_structured_xlsx or needs_layout_raw:
enriched["raw_content"] = raw enriched["raw_content"] = raw
sources[index] = enriched sources[index] = enriched
continue continue
if str(source.get("file_format") or "").lower() != "pdf":
continue
pages = extract_pdf_page_texts(raw) pages = extract_pdf_page_texts(raw)
extracted_text = "\n\n".join(page.text for page in pages if page.text) extracted_text = "\n\n".join(page.text for page in pages if page.text)
if extracted_text != str(source.get("content") or ""): if extracted_text != str(source.get("content") or ""):
@@ -1413,7 +1616,7 @@ def _prepare_preview_items(
enriched["document_noise_spans"] = detect_pdf_document_noise(pages) enriched["document_noise_spans"] = detect_pdf_document_noise(pages)
sources[index] = enriched sources[index] = enriched
items = _build_preview_items(task, sources) items = _build_preview_items(task, sources)
if not items and source_file_ids is None: if not items and source_file_ids is None and is_unstructured:
raise InvalidStateError("source files did not produce preview items") raise InvalidStateError("source files did not produce preview items")
return items return items
@@ -1435,6 +1638,7 @@ def _run_preview(
len(source_file_ids), len(source_file_ids),
) )
try: try:
is_unstructured = store.get_task(task_id).get("process_type") == "unstructured"
if not store.mark_preview_running(task_id, preview_run_id): if not store.mark_preview_running(task_id, preview_run_id):
logger.info( logger.info(
"data process preview skipped inactive run task_id=%s preview_run_id=%s", "data process preview skipped inactive run task_id=%s preview_run_id=%s",
@@ -1461,7 +1665,7 @@ def _run_preview(
storage, storage,
[source_file_id], [source_file_id],
) )
if not items: if not items and is_unstructured:
raise InvalidStateError( raise InvalidStateError(
f"source file did not produce preview items: {source_file_id}" f"source file did not produce preview items: {source_file_id}"
) )
@@ -1657,8 +1861,13 @@ def update_preview_item(
) -> dict[str, Any]: ) -> dict[str, Any]:
with api_errors(): with api_errors():
task = store.get_task(task_id) task = store.get_task(task_id)
existing = store.get_preview_item(task_id, preview_id)
update = payload.model_dump(exclude_unset=True, mode="json") update = payload.model_dump(exclude_unset=True, mode="json")
update["quality_score"] = _preview_quality(payload.edited_content, task.get("config") or {}) quality = _preview_quality(payload.edited_content, task.get("config") or {})
source_locator = (existing.get("quality_score") or {}).get("source_locator")
if isinstance(source_locator, Mapping):
quality["source_locator"] = deepcopy(dict(source_locator))
update["quality_score"] = quality
item = store.update_preview_item( item = store.update_preview_item(
task_id, task_id,
preview_id, preview_id,
@@ -2089,13 +2298,11 @@ def regenerate_results_batch(
max_keepalive_connections=RESULT_REGENERATION_CONCURRENCY, max_keepalive_connections=RESULT_REGENERATION_CONCURRENCY,
) )
# httpx.Client 支持跨线程复用,批次内共享连接池可减少重复建连开销。 # httpx.Client 支持跨线程复用,批次内共享连接池可减少重复建连开销。
with ( with httpx.Client(timeout=model_timeout, limits=model_limits) as model_client, \
httpx.Client(timeout=model_timeout, limits=model_limits) as model_client,
ThreadPoolExecutor( ThreadPoolExecutor(
max_workers=min(RESULT_REGENERATION_CONCURRENCY, len(prepared)), max_workers=min(RESULT_REGENERATION_CONCURRENCY, len(prepared)),
thread_name_prefix="data-result-regeneration", thread_name_prefix="data-result-regeneration",
) as executor, ) as executor:
):
futures = { futures = {
executor.submit( executor.submit(
_regenerate_result_in_place, _regenerate_result_in_place,

View File

@@ -2,14 +2,16 @@
import json import json
import uuid import uuid
from datetime import datetime, timedelta, timezone
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from fastapi import APIRouter, BackgroundTasks, Body, File, HTTPException, Query, UploadFile from fastapi import APIRouter, BackgroundTasks, Body, Depends, File, HTTPException, Query, Request, UploadFile
from fastapi.responses import PlainTextResponse, StreamingResponse from fastapi.responses import PlainTextResponse, StreamingResponse
import httpx import httpx
from app.core.auth import filter_accessible_resource_ids, get_current_user, has_resource_access, is_admin
from app.core.config import get_settings from app.core.config import get_settings
from app.db.platform_store import get_platform_store from app.db.platform_store import get_platform_store
from app.modules.compute_gateway.client import ComputeNodeClient from app.modules.compute_gateway.client import ComputeNodeClient
@@ -91,6 +93,35 @@ def fail(status_code: int, message: str) -> HTTPException:
return HTTPException(status_code=status_code, detail={"code": status_code, "message": message, "data": None}) return HTTPException(status_code=status_code, detail={"code": status_code, "message": message, "data": None})
def _require_approval_or_admin(
resource_type: str,
resource_id: str,
current_user: dict[str, Any],
action_desc: str = "",
) -> dict[str, Any] | None:
"""
高风险操作审批旁路:
- admin 用户直接放行(返回 None
- 普通用户创建审批实例返回审批待定响应code=202非 None
code=202 使前端响应拦截器走业务错误分支,弹提示并 reject
避免前端误认为删除成功。
"""
if is_admin(current_user):
return None
store = get_platform_store()
instance = store.create_approval_instance({
"resource_type": resource_type,
"resource_id": resource_id,
"applicant_id": current_user.get("id"),
"template_id": None,
})
return {
"code": 202,
"message": f"操作已提交审批,等待管理员批准:{action_desc}",
"data": {"approval_required": True, "approval_id": instance["id"]},
}
def _node_for_task(task: dict[str, Any]) -> dict[str, Any] | None: def _node_for_task(task: dict[str, Any]) -> dict[str, Any] | None:
return next((node for node in get_platform_store().compute_nodes() if node["id"] == task.get("compute_node_id")), None) return next((node for node in get_platform_store().compute_nodes() if node["id"] == task.get("compute_node_id")), None)
@@ -287,8 +318,18 @@ async def login(payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
@router.get("/me") @router.get("/me")
async def me() -> dict[str, Any]: async def me(request: Request) -> dict[str, Any]:
return ok(get_platform_store().users()[0]) """根据 Authorization header 中的 token 返回当前登录用户信息"""
store = get_platform_store()
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
# token 格式: platform-token-{user_id}
if token.startswith("platform-token-"):
user_id = token[len("platform-token-"):]
for u in store.users():
if u.get("id") == user_id:
return ok(u)
raise fail(401, "invalid or missing token")
@router.get("/dashboard/overview") @router.get("/dashboard/overview")
@@ -307,6 +348,183 @@ async def dashboard_overview() -> dict[str, Any]:
) )
@router.get("/dashboard/stats")
async def dashboard_stats() -> dict[str, Any]:
"""看板聚合数据:基于平台真实数据;缺项做合理近似。"""
store = get_platform_store()
tasks = store.tasks()
users = store.users()
nodes = store.compute_nodes()
datasets = store.datasets()
eval_tasks = store.eval_tasks()
# 数据处理任务总数(来自 data_process 模块)
try:
from app.modules.data_process.store import get_data_process_store
dp_store = get_data_process_store()
dp_result = dp_store.list_tasks(page=1, page_size=1)
dp_count = int(dp_result.get("total", 0))
except Exception:
dp_count = 0
running_statuses = {"syncing", "queued", "running"}
running_ft = [t for t in tasks if t.get("status") in running_statuses]
failed_ft = [t for t in tasks if t.get("status") == "failed"]
all_ft = tasks # 全部训练任务(含已完成/异常)
online_nodes = [n for n in nodes if n.get("scheduler_status") == "online"]
# 近 7 天训练统计(按创建日期分桶)
now = datetime.now(timezone.utc)
train_by_day: dict[str, int] = {}
for t in tasks:
ct = t.get("create_time")
if ct:
train_by_day[ct[:10]] = train_by_day.get(ct[:10], 0) + 1
training_7d = []
for i in range(6, -1, -1):
day = (now - timedelta(days=i)).strftime("%Y-%m-%d")
training_7d.append(
{
"date": day[5:],
"train": train_by_day.get(day, 0),
"gpu": sum(len(t.get("gpus") or []) for t in running_ft),
"accuracy": None,
}
)
# 服务状态 —— 与界面实际数据对齐
service_status = [
{
"type": "模型推理",
"status": "error" if (nodes and not online_nodes) else ("busy" if (nodes and len(online_nodes) < len(nodes)) else "normal"),
"count": len(online_nodes),
},
{
"type": "模型训练",
"status": "error" if failed_ft else ("busy" if running_ft else "normal"),
"count": len(all_ft),
},
{
"type": "模型评测",
"status": "normal",
"count": len(eval_tasks),
},
{
"type": "数据处理",
"status": "normal" if not failed_ft else "busy",
"count": dp_count,
},
]
# 训练任务状态归一化
status_map = {
"syncing": "running",
"queued": "running",
"running": "running",
"pending": "pending",
"paused": "pending",
"completed": "completed",
"failed": "failed",
"error": "failed",
"cancelled": "failed",
"stopped": "failed",
}
training_tasks = [
{
"id": t.get("id"),
"name": t.get("name"),
"status": status_map.get(t.get("status"), "pending"),
"train_type": t.get("train_type") or t.get("trainType") or "",
"train_method": t.get("train_method") or t.get("trainMethod") or "",
"base_model": t.get("base_model") or t.get("baseModel") or "",
"progress": t.get("progress", 0),
"accuracy": t.get("accuracy"),
"started_at": (t.get("create_time") or "")[:16],
}
for t in tasks[:8]
]
# 用户操作分布:统计平台全部操作(含治理模块)
MODULE_LABELS = [
("data-process", "数据处理"),
("data_process", "数据处理"),
("dataset", "数据集管理"),
("fine-tune", "模型训练"),
("fine_tune", "模型训练"),
("model-eval", "模型评测"),
("eval", "模型评测"),
("model-inference", "模型推理"),
("inference", "模型推理"),
("model-manage", "模型管理"),
("model", "模型管理"),
("trained", "模型管理"),
# 治理模块操作
("tenant", "租户与项目"),
("project", "租户与项目"),
("approval", "租户与项目"),
("acl", "租户与项目"),
("user", "用户管理"),
("role", "用户管理"),
]
OP_ORDER = [
"数据集管理",
"数据处理",
"模型训练",
"模型评测",
"模型推理",
"模型管理",
"租户与项目",
"用户管理",
]
def _op_module(action: str) -> str | None:
a = (action or "").lower()
for prefix, label in MODULE_LABELS:
if a.startswith(prefix):
return label
return None
audit = store.audit_logs(limit=1000)
op_counter: dict[str, int] = {label: 0 for label in OP_ORDER}
for log in audit.get("items", []):
label = _op_module(log.get("action") or "")
if label:
op_counter[label] += 1
operation_distribution = [{"name": k, "value": v} for k, v in op_counter.items()]
# 最近登录用户
recent = sorted(
[u for u in users if u.get("last_login")],
key=lambda u: u["last_login"],
reverse=True,
)[:5]
recent_login_users = [
{
"user": u.get("display_name") or u.get("username"),
"role": u.get("role"),
"last_login": (u.get("last_login") or "")[:16],
}
for u in recent
]
# 登录时长排行(本月)
login_duration_rank = store.login_duration_rank()
return ok(
{
"online_services": sum(s["count"] for s in service_status),
"running_tasks": len(running_ft),
"pending_alerts": 0,
"training_7d": training_7d,
"service_status": service_status,
"training_tasks": training_tasks,
"operation_distribution": operation_distribution,
"login_duration_rank": login_duration_rank,
"recent_login_users": recent_login_users,
}
)
@router.get("/system-info") @router.get("/system-info")
async def system_info() -> dict[str, Any]: async def system_info() -> dict[str, Any]:
return ok(get_platform_store().system_info()) return ok(get_platform_store().system_info())
@@ -341,6 +559,21 @@ async def delete_user(user_id: str, current_username: str | None = Query(default
raise fail(400, str(exc)) raise fail(400, str(exc))
@router.post("/users/{user_id}/reset-password")
async def reset_user_password(
user_id: str,
payload: dict[str, Any] = Body(default={}),
) -> dict[str, Any]:
new_password = payload.get("password") or "Platform@123"
try:
get_platform_store().reset_password(user_id, new_password)
return ok({"reset": user_id})
except KeyError:
raise fail(404, "user not found")
except ValueError as exc:
raise fail(400, str(exc))
@router.get("/model-manage/local-models") @router.get("/model-manage/local-models")
async def local_models() -> dict[str, Any]: async def local_models() -> dict[str, Any]:
store = get_platform_store() store = get_platform_store()
@@ -404,8 +637,13 @@ async def model_by_name(name: str) -> dict[str, Any]:
@router.get("/model-manage") @router.get("/model-manage")
async def model_list() -> dict[str, Any]: async def model_list(current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
return ok(get_platform_store().models()) models = get_platform_store().models()
if current_user.get("role") == "admin" or current_user.get("protected"):
return ok(models)
# 普通用户只返回有 ACL 授权的模型
accessible = set(filter_accessible_resource_ids("model", [m["id"] for m in models], current_user))
return ok([m for m in models if m["id"] in accessible])
@router.post("/model-manage") @router.post("/model-manage")
@@ -421,11 +659,14 @@ async def create_model(payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
@router.get("/model-manage/{model_id}") @router.get("/model-manage/{model_id}")
async def model_detail(model_id: str) -> dict[str, Any]: async def model_detail(model_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
try: try:
return ok(get_platform_store().model(model_id)) model = get_platform_store().model(model_id)
except KeyError: except KeyError:
raise fail(404, "model not found") raise fail(404, "model not found")
if not has_resource_access("model", model_id, current_user, "read"):
raise fail(403, "no permission to access this model")
return ok(model)
@router.put("/model-manage/{model_id}") @router.put("/model-manage/{model_id}")
@@ -445,7 +686,12 @@ async def update_model_purpose(model_id: str, payload: dict[str, Any] = Body(...
@router.delete("/model-manage/{model_id}") @router.delete("/model-manage/{model_id}")
async def delete_model(model_id: str) -> dict[str, Any]: async def delete_model(model_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
if not has_resource_access("model", model_id, current_user, "delete"):
raise fail(403, "no permission to delete this model")
pending = _require_approval_or_admin("model", model_id, current_user, f"删除模型 {model_id}")
if pending:
return pending
get_platform_store().delete_model(model_id) get_platform_store().delete_model(model_id)
return ok({"deleted": model_id}) return ok({"deleted": model_id})
@@ -707,8 +953,12 @@ async def download_dataset_file(dataset_id: str, file_id: str, version_id: str |
@router.get("/dataset-manage") @router.get("/dataset-manage")
async def dataset_list() -> dict[str, Any]: async def dataset_list(current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
return ok(get_platform_store().datasets()) datasets = get_platform_store().datasets()
if current_user.get("role") == "admin" or current_user.get("protected"):
return ok(datasets)
accessible = set(filter_accessible_resource_ids("dataset", [d["id"] for d in datasets], current_user))
return ok([d for d in datasets if d["id"] in accessible])
@router.post("/dataset-manage") @router.post("/dataset-manage")
@@ -718,11 +968,14 @@ async def create_dataset(payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
@router.get("/dataset-manage/{dataset_id}") @router.get("/dataset-manage/{dataset_id}")
async def dataset_detail(dataset_id: str) -> dict[str, Any]: async def dataset_detail(dataset_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
try: try:
return ok(get_platform_store().dataset(dataset_id)) dataset = get_platform_store().dataset(dataset_id)
except KeyError: except KeyError:
raise fail(404, "dataset not found") raise fail(404, "dataset not found")
if not has_resource_access("dataset", dataset_id, current_user, "read"):
raise fail(403, "no permission to access this dataset")
return ok(dataset)
@router.put("/dataset-manage/{dataset_id}") @router.put("/dataset-manage/{dataset_id}")
@@ -734,7 +987,12 @@ async def update_dataset(dataset_id: str, payload: dict[str, Any] = Body(...)) -
@router.delete("/dataset-manage/{dataset_id}") @router.delete("/dataset-manage/{dataset_id}")
async def delete_dataset(dataset_id: str) -> dict[str, Any]: async def delete_dataset(dataset_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
if not has_resource_access("dataset", dataset_id, current_user, "delete"):
raise fail(403, "no permission to delete this dataset")
pending = _require_approval_or_admin("dataset", dataset_id, current_user, f"删除数据集 {dataset_id}")
if pending:
return pending
get_platform_store().delete_dataset(dataset_id) get_platform_store().delete_dataset(dataset_id)
return ok({"deleted": dataset_id}) return ok({"deleted": dataset_id})
@@ -759,8 +1017,12 @@ async def tensorboard_start() -> dict[str, Any]:
@router.get("/fine-tune") @router.get("/fine-tune")
async def fine_tune_list() -> dict[str, Any]: async def fine_tune_list(current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
return ok(get_platform_store().tasks()) tasks = get_platform_store().tasks()
if current_user.get("role") == "admin" or current_user.get("protected"):
return ok(tasks)
accessible = set(filter_accessible_resource_ids("fine-tune", [t["id"] for t in tasks], current_user))
return ok([t for t in tasks if t["id"] in accessible])
@router.post("/fine-tune") @router.post("/fine-tune")
@@ -953,7 +1215,12 @@ async def retry_fine_tune(task_id: str, payload: dict[str, Any] | None = Body(de
@router.delete("/fine-tune/{task_id}") @router.delete("/fine-tune/{task_id}")
async def delete_fine_tune(task_id: str) -> dict[str, Any]: async def delete_fine_tune(task_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
if not has_resource_access("fine-tune", task_id, current_user, "delete"):
raise fail(403, "no permission to delete this task")
pending = _require_approval_or_admin("fine-tune", task_id, current_user, f"删除训练任务 {task_id}")
if pending:
return pending
get_platform_store().delete_task(task_id) get_platform_store().delete_task(task_id)
return ok({"deleted": task_id}) return ok({"deleted": task_id})
@@ -993,12 +1260,16 @@ async def fine_tune_metrics(task_id: str) -> dict[str, Any]:
@router.get("/model-eval") @router.get("/model-eval")
async def model_eval_list() -> dict[str, Any]: async def model_eval_list(current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
return ok(get_platform_store().eval_tasks()) tasks = get_platform_store().eval_tasks()
if current_user.get("role") == "admin" or current_user.get("protected"):
return ok(tasks)
accessible = set(filter_accessible_resource_ids("eval", [t["id"] for t in tasks], current_user))
return ok([t for t in tasks if t["id"] in accessible])
@router.get("/model-eval/{task_id}") @router.get("/model-eval/{task_id}")
async def model_eval_detail(task_id: str) -> dict[str, Any]: async def model_eval_detail(task_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
try: try:
store = get_platform_store() store = get_platform_store()
task = store.eval_task(task_id) task = store.eval_task(task_id)
@@ -1019,9 +1290,11 @@ async def model_eval_detail(task_id: str) -> dict[str, Any]:
break break
except Exception: except Exception:
pass pass
return ok(task)
except KeyError: except KeyError:
raise fail(404, "eval task not found") raise fail(404, "eval task not found")
if not has_resource_access("eval", task_id, current_user, "read"):
raise fail(403, "no permission to access this eval task")
return ok(task)
@router.post("/model-eval/start") @router.post("/model-eval/start")
@@ -1172,7 +1445,12 @@ async def model_eval_start(payload: dict[str, Any] = Body(...)) -> dict[str, Any
@router.delete("/model-eval/{task_id}") @router.delete("/model-eval/{task_id}")
async def model_eval_delete(task_id: str) -> dict[str, Any]: async def model_eval_delete(task_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
if not has_resource_access("eval", task_id, current_user, "delete"):
raise fail(403, "no permission to delete this eval task")
pending = _require_approval_or_admin("eval", task_id, current_user, f"删除评测任务 {task_id}")
if pending:
return pending
get_platform_store().delete_eval_task(task_id) get_platform_store().delete_eval_task(task_id)
return ok({"deleted": task_id}) return ok({"deleted": task_id})

View File

@@ -3,8 +3,20 @@
from app.api.v1.endpoints.data_process import router as data_process_router from app.api.v1.endpoints.data_process import router as data_process_router
from app.api.v1.endpoints.platform import router as platform_router from app.api.v1.endpoints.platform import router as platform_router
from app.api.v1.endpoints.health import router as health_router from app.api.v1.endpoints.health import router as health_router
from app.modules.tenant.router import router as tenant_router
from app.modules.project.router import router as project_router
from app.modules.approval.router import router as approval_router
from app.modules.system.router import router as system_router
from app.modules.retention.router import router as retention_router
from app.modules.resource.router import router as resource_router
api_router = APIRouter() api_router = APIRouter()
api_router.include_router(health_router, tags=["health"]) api_router.include_router(health_router, tags=["health"])
api_router.include_router(data_process_router, tags=["data-process"]) api_router.include_router(data_process_router, tags=["data-process"])
api_router.include_router(platform_router, tags=["platform"]) api_router.include_router(platform_router, tags=["platform"])
api_router.include_router(system_router, tags=["system"])
api_router.include_router(tenant_router, tags=["tenant"])
api_router.include_router(project_router, tags=["project"])
api_router.include_router(approval_router, tags=["approval"])
api_router.include_router(retention_router, tags=["retention"])
api_router.include_router(resource_router, tags=["resource"])

138
backend/app/core/auth.py Normal file
View File

@@ -0,0 +1,138 @@
"""鉴权依赖:从 Authorization header 解析当前用户,提供权限校验。"""
from __future__ import annotations
from typing import Any
from fastapi import Depends, HTTPException, Query, Request, status
from app.db.platform_store import get_platform_store
# 无需鉴权的路径前缀(健康检查、登录等)
PUBLIC_PATHS = ("/health", "/login", "/system-info")
def _extract_token(request: Request) -> str | None:
"""从 Authorization header 提取 token格式: Bearer platform-token-{user_id})。"""
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
if token.startswith("platform-token-"):
return token[len("platform-token-"):]
return None
def get_current_user(request: Request) -> dict[str, Any]:
"""
FastAPI 依赖:解析当前登录用户。
- 公开路径(/health, /login 等)直接放行,返回匿名用户。
- 无 token 或 token 无效时抛 401。
- admin 用户标记为超级管理员,拥有全部权限。
"""
path = request.url.path
# 去掉路由前缀后判断
for prefix in PUBLIC_PATHS:
if path.endswith(prefix):
return {"id": None, "username": "anonymous", "role": "viewer", "permissions": [], "protected": False}
user_id = _extract_token(request)
if not user_id:
raise HTTPException(status_code=status.HTTP_401_UNAUTHORIZED, detail="missing or invalid token")
store = get_platform_store()
for u in store.users():
if u.get("id") == user_id:
return u
raise HTTPException(status_code=status.HTTP_401_UNAUTHORIZED, detail="user not found")
def require_admin(current_user: dict[str, Any] = Depends(get_current_user)) -> dict[str, Any]:
"""FastAPI 依赖要求当前用户是管理员role=admin 或 protected"""
if current_user.get("role") == "admin" or current_user.get("protected"):
return current_user
raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail="admin permission required")
def is_admin(user: dict[str, Any]) -> bool:
"""判断用户是否为管理员admin 角色或 protected 标记)。"""
return user.get("role") == "admin" or user.get("protected", False)
def has_resource_access(
resource_type: str,
resource_id: str,
user: dict[str, Any],
permission: str = "read",
) -> bool:
"""
检查用户对某资源是否有指定权限。
- admin/protected 用户直接放行(旁路)。
- 其他用户检查 acls 表中是否有对应授权。
"""
if user.get("role") == "admin" or user.get("protected"):
return True
store = get_platform_store()
acls = store.get_acl(resource_type, resource_id)
user_id = user.get("id")
user_role = user.get("role")
for entry in acls:
# 按 user 授权
if entry.get("principal_type") == "user" and entry.get("principal_id") == user_id:
if _permission_covers(entry.get("permission"), permission):
return True
# 按 role 授权
if entry.get("principal_type") == "role" and entry.get("principal_id") == user_role:
if _permission_covers(entry.get("permission"), permission):
return True
return False
def _permission_covers(granted: str | None, required: str) -> bool:
"""权限覆盖判断write/execute 覆盖 readadmin 覆盖一切。"""
if not granted:
return False
if granted == "admin":
return True
if granted == required:
return True
# write 覆盖 read
if required == "read" and granted in ("write", "execute"):
return True
return False
def filter_accessible_resource_ids(
resource_type: str,
all_ids: list[str],
user: dict[str, Any],
) -> list[str]:
"""
从全部资源 ID 中过滤出当前用户可访问的 ID 列表。
- admin 直接返回全部。
- 普通用户查 acls 表取交集。
"""
if user.get("role") == "admin" or user.get("protected"):
return all_ids
if not all_ids:
return []
store = get_platform_store()
user_id = user.get("id")
user_role = user.get("role")
# 查询该用户在该资源类型下有 read 权限的所有 resource_id
with store.connect() as conn:
rows = conn.execute(
"""
SELECT DISTINCT resource_id FROM acls
WHERE resource_type=? AND (
(principal_type='user' AND principal_id=?)
OR (principal_type='role' AND principal_id=?)
)
""",
(resource_type, user_id, user_role),
).fetchall()
accessible = {r["resource_id"] for r in rows}
return [rid for rid in all_ids if rid in accessible]

View File

@@ -2,6 +2,18 @@
from functools import lru_cache from functools import lru_cache
import os import os
try:
from pathlib import Path as _Path
from dotenv import load_dotenv
# 显式指定 backend 目录下的 .env并强制覆盖已有环境变量
# 确保远程数据库配置生效,不被本地默认值或残留环境变量影响。
_env_path = _Path(__file__).resolve().parent.parent.parent / ".env"
load_dotenv(dotenv_path=_env_path, override=True)
except ImportError:
pass
def _int_env(name: str, default: int) -> int: def _int_env(name: str, default: int) -> int:
raw = os.getenv(name) raw = os.getenv(name)

View File

@@ -14,6 +14,7 @@ from pathlib import Path
from typing import Any, Iterator from typing import Any, Iterator
import psycopg import psycopg
from psycopg_pool import ConnectionPool
from app.core.config import get_settings from app.core.config import get_settings
@@ -291,6 +292,32 @@ class PlatformStore:
def __init__(self, database_url: str | None = None) -> None: def __init__(self, database_url: str | None = None) -> None:
settings = get_settings() settings = get_settings()
self.database_url = _psycopg_url(database_url or settings.database_url) self.database_url = _psycopg_url(database_url or settings.database_url)
# Reuse connections via a pool to avoid the TCP+auth handshake on every
# request (notably expensive against the remote PostgreSQL instance).
# TCP keepalive 让操作系统持续保活连接,抵抗远程库空闲静默断连。
pool_kwargs = {
"keepalives": 1,
"keepalives_idle": 30,
"keepalives_interval": 10,
"keepalives_count": 5,
}
self._pool = ConnectionPool(
conninfo=self.database_url,
kwargs=pool_kwargs,
min_size=2,
max_size=10,
# 借出前校验连接可用性,避免执行 SQL 时才发现 [BAD] 再重建。
check=ConnectionPool.check_connection,
# 不主动回收空闲连接(远程库约 10s 断,由 keepalive 维持),
# 减少无谓的重建握手。
max_idle=0,
# 请求最多排队等待 5s避免雪崩时无限堆积。
max_waiting=16,
open=False,
)
# 注意:不要在此调用 pool.wait(),它会阻塞等待 min_size 个连接就绪,
# 在远程库响应慢/超时时会卡死 uvicorn worker 进程,导致所有请求无响应。
self._pool.open()
self.ensure_schema() self.ensure_schema()
self.ensure_seed_data() self.ensure_seed_data()
# Track which compute nodes have an active inference model loaded # Track which compute nodes have an active inference model loaded
@@ -309,7 +336,7 @@ class PlatformStore:
@contextmanager @contextmanager
def connect(self) -> Iterator["PgConnection"]: def connect(self) -> Iterator["PgConnection"]:
raw_conn = psycopg.connect(self.database_url) with self._pool.connection() as raw_conn:
conn = PgConnection(raw_conn) conn = PgConnection(raw_conn)
try: try:
yield conn yield conn
@@ -320,6 +347,13 @@ class PlatformStore:
finally: finally:
conn.close() conn.close()
def close_pool(self) -> None:
"""Release pooled connections. Safe to call multiple times."""
try:
self._pool.close()
except Exception:
pass
def ensure_schema(self) -> None: def ensure_schema(self) -> None:
schema_path = Path(__file__).with_name("sql") / "001_platform_runtime.sql" schema_path = Path(__file__).with_name("sql") / "001_platform_runtime.sql"
with self.connect() as conn: with self.connect() as conn:
@@ -348,6 +382,11 @@ class PlatformStore:
"last_error": "TEXT", "last_error": "TEXT",
}, },
) )
schema_dir = Path(__file__).with_name("sql")
for extra in ("002_governance.sql", "003_tenant_quota.sql"):
extra_path = schema_dir / extra
if extra_path.exists():
conn.executescript(extra_path.read_text(encoding="utf-8"))
def _column_names(self, conn: PgConnection, table_name: str) -> set[str]: def _column_names(self, conn: PgConnection, table_name: str) -> set[str]:
columns = conn.execute( columns = conn.execute(
@@ -1108,6 +1147,18 @@ class PlatformStore:
raise ValueError("protected user cannot be deleted") raise ValueError("protected user cannot be deleted")
conn.execute("DELETE FROM users WHERE id=?", (user_id,)) conn.execute("DELETE FROM users WHERE id=?", (user_id,))
def reset_password(self, user_id: str, new_password: str) -> None:
with self.connect() as conn:
row = conn.execute("SELECT protected FROM users WHERE id=?", (user_id,)).fetchone()
if not row:
raise KeyError(user_id)
if row["protected"]:
raise ValueError("protected user cannot reset password")
conn.execute(
"UPDATE users SET password_hash=? WHERE id=?",
(hash_password(new_password), user_id),
)
def _user(self, row: PgRow) -> dict[str, Any]: def _user(self, row: PgRow) -> dict[str, Any]:
return { return {
"id": row["id"], "id": row["id"],
@@ -2954,6 +3005,672 @@ class PlatformStore:
] ]
return {"file": file_name, "content": "\n".join(lines), "size": "1 KB"} return {"file": file_name, "content": "\n".join(lines), "size": "1 KB"}
# ===================== 平台治理:角色 =====================
def roles(self) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute("SELECT * FROM roles ORDER BY name").fetchall()
return [dict(r) for r in rows]
# ===================== 平台治理:审计日志 =====================
def audit_logs(
self,
*,
tenant_id: str | None = None,
project_id: str | None = None,
actor_id: str | None = None,
action: str | None = None,
target_type: str | None = None,
start_time: str | None = None,
end_time: str | None = None,
limit: int = 50,
offset: int = 0,
) -> dict[str, Any]:
clauses: list[str] = []
params: list[Any] = []
if tenant_id:
clauses.append("tenant_id=?")
params.append(tenant_id)
if project_id:
clauses.append("project_id=?")
params.append(project_id)
if actor_id:
clauses.append("actor_id=?")
params.append(actor_id)
if action:
clauses.append("action=?")
params.append(action)
if target_type:
clauses.append("target_type=?")
params.append(target_type)
if start_time:
clauses.append("time>=?")
params.append(start_time)
if end_time:
clauses.append("time<=?")
params.append(end_time)
where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
with self.connect() as conn:
total = conn.execute(f"SELECT COUNT(*) AS c FROM audit_logs{where}", tuple(params)).fetchone()["c"]
params_paged = list(params) + [limit, offset]
rows = conn.execute(
f"SELECT * FROM audit_logs{where} ORDER BY time DESC LIMIT ? OFFSET ?",
tuple(params_paged),
).fetchall()
return {"total": total, "items": [dict(r) for r in rows]}
def record_audit(
self,
*,
action: str,
actor_id: str | None = None,
target_type: str | None = None,
target_id: str | None = None,
tenant_id: str | None = None,
project_id: str | None = None,
detail: str | None = None,
ip: str | None = None,
) -> None:
with self.connect() as conn:
conn.execute(
"""
INSERT INTO audit_logs
(id, tenant_id, project_id, actor_id, action, target_type, target_id, detail, client_ip, time)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
new_id("log"),
tenant_id,
project_id,
actor_id,
action,
target_type,
target_id,
detail,
ip,
utcnow(),
),
)
# ===================== 平台治理:会话 =====================
def create_session(self, user_id: str, *, ip: str | None = None) -> dict[str, Any]:
sid = new_id("sess")
login_at = utcnow()
with self.connect() as conn:
conn.execute(
"INSERT INTO sessions (id, user_id, username, login_at, create_time) "
"VALUES (%s, %s, (SELECT username FROM users WHERE id=%s), %s, %s)",
(sid, user_id, user_id, login_at, login_at),
)
return {"session_id": sid, "user_id": user_id, "login_at": login_at}
def finish_session(self, session_id: str) -> None:
"""登出时记录 logout_at 与时长(秒)。"""
logout_at = utcnow()
with self.connect() as conn:
conn.execute(
"UPDATE sessions SET logout_at=%s, "
"duration_seconds=EXTRACT(EPOCH FROM (%s::timestamptz - login_at::timestamptz))::int "
"WHERE id=%s AND logout_at IS NULL",
(logout_at, logout_at, session_id),
)
def active_sessions(self, user_id: str) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute(
"SELECT * FROM sessions WHERE user_id=%s AND logout_at IS NULL "
"ORDER BY login_at DESC",
(user_id,),
).fetchall()
return [dict(r) for r in rows]
def destroy_session(self, session_id: str) -> None:
with self.connect() as conn:
conn.execute("DELETE FROM sessions WHERE id=%s", (session_id,))
def extend_session(self, session_id: str, *, expires_in_seconds: int = 3600 * 8) -> dict[str, Any] | None:
# 兼容旧调用,仅更新 login_at 之后延长的含义在此简化为 no-op 返回现有记录。
with self.connect() as conn:
row = conn.execute("SELECT * FROM sessions WHERE id=%s", (session_id,)).fetchone()
if not row:
return None
return {
"session_id": session_id,
"user_id": row["user_id"],
"login_at": row["login_at"],
}
def set_session_user(self, session_id: str, user_id: str) -> None:
with self.connect() as conn:
conn.execute("UPDATE sessions SET user_id=%s WHERE id=%s", (user_id, session_id))
def login_duration_rank(self, limit: int = 8, days: int = 30) -> list[dict[str, Any]]:
"""登录时长排行:按用户聚合近 N 天的会话时长(小时)。
sessions 表列login_at(TEXT), logout_at(TEXT), duration_seconds(INT)。
优先用 duration_seconds为空时回退计算 now-login_at未登出或 logout_at-login_at。
"""
with self.connect() as conn:
rows = conn.execute(
"SELECT s.user_id, s.login_at, s.logout_at, s.duration_seconds, "
"u.username, u.display_name, u.role "
"FROM sessions s LEFT JOIN users u ON s.user_id = u.id "
"WHERE s.login_at::timestamptz >= NOW() - make_interval(days => %s)",
(days,),
).fetchall()
now = datetime.now(timezone.utc)
agg: dict[str, dict[str, Any]] = {}
for r in rows:
uid = r["user_id"] or ""
bucket = agg.setdefault(
uid,
{
"user": r["display_name"] or r["username"] or uid,
"role": r["role"] or "",
"total": 0.0,
},
)
dur = r["duration_seconds"]
if dur is not None:
bucket["total"] += float(dur)
continue
start = parse_time(r["login_at"])
end = parse_time(r["logout_at"]) if r["logout_at"] else None
if start and end:
bucket["total"] += max(0, (end - start).total_seconds())
elif start:
bucket["total"] += max(0, (now - start).total_seconds())
result = [
{"user": b["user"], "role": b["role"], "duration": round(b["total"] / 3600, 1)}
for b in agg.values()
]
result.sort(key=lambda x: x["duration"], reverse=True)
return result[:limit]
# ===================== 平台治理:审批 =====================
def create_approval_template(self, payload: dict[str, Any]) -> dict[str, Any]:
with self.connect() as conn:
tid = new_id("tpl")
conn.execute(
"INSERT INTO approval_templates (id, name, steps, create_time) VALUES (?, ?, ?, ?)",
(tid, payload["name"], json_dumps(payload.get("steps", [])), utcnow()),
)
return self.approval_template(tid)
def approval_templates(self) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute("SELECT * FROM approval_templates ORDER BY create_time DESC").fetchall()
return [dict(r) for r in rows]
def approval_template(self, template_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM approval_templates WHERE id=?", (template_id,)).fetchone()
if not row:
raise KeyError(template_id)
return dict(row)
def update_approval_template(self, template_id: str, payload: dict[str, Any]) -> dict[str, Any]:
fields = {k: v for k, v in payload.items() if k in ("name", "steps")}
if "steps" in fields:
fields["steps"] = json_dumps(fields["steps"])
if not fields:
return self.approval_template(template_id)
set_clause = ", ".join(f"{k}=?" for k in fields)
params = list(fields.values()) + [template_id]
with self.connect() as conn:
conn.execute(f"UPDATE approval_templates SET {set_clause} WHERE id=?", tuple(params))
return self.approval_template(template_id)
def delete_approval_template(self, template_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM approval_templates WHERE id=?", (template_id,)).fetchone()
if not row:
raise KeyError(template_id)
conn.execute("DELETE FROM approval_templates WHERE id=?", (template_id,))
return dict(row)
def create_approval_instance(self, payload: dict[str, Any]) -> dict[str, Any]:
template_id = payload.get("template_id")
steps = []
if template_id:
tpl = self.approval_template(template_id)
steps = json_loads(tpl["steps"]) if tpl.get("steps") else []
with self.connect() as conn:
iid = new_id("appr")
conn.execute(
"""
INSERT INTO approval_instances
(id, template_id, resource_type, resource_id, applicant_id, status, current_step, create_time)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
""",
(
iid,
template_id,
payload["resource_type"],
payload["resource_id"],
payload["applicant_id"],
"pending",
0,
utcnow(),
),
)
for idx, step in enumerate(steps):
conn.execute(
"INSERT INTO approval_steps (id, instance_id, step_index, approver_id, status, time) VALUES (?, ?, ?, ?, ?, ?)",
(new_id("step"), iid, idx, step.get("approver_id"), "pending", None),
)
return self.approval_instance(iid)
def approval_instances(self, *, status: str | None = None) -> list[dict[str, Any]]:
with self.connect() as conn:
if status:
rows = conn.execute(
"SELECT * FROM approval_instances WHERE status=? ORDER BY create_time DESC", (status,)
).fetchall()
else:
rows = conn.execute("SELECT * FROM approval_instances ORDER BY create_time DESC").fetchall()
return [dict(r) for r in rows]
def approval_instance(self, instance_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM approval_instances WHERE id=?", (instance_id,)).fetchone()
if not row:
raise KeyError(instance_id)
steps = conn.execute(
"SELECT * FROM approval_steps WHERE instance_id=? ORDER BY step_index", (instance_id,)
).fetchall()
result = dict(row)
result["steps"] = [dict(s) for s in steps]
return result
def decide_approval_step(self, instance_id: str, step_index: int, *, approver_id: str, approved: bool, comment: str | None = None) -> dict[str, Any]:
with self.connect() as conn:
inst = conn.execute("SELECT * FROM approval_instances WHERE id=?", (instance_id,)).fetchone()
if not inst:
raise KeyError(instance_id)
if inst["status"] != "pending":
raise ValueError("instance not pending")
step = conn.execute(
"SELECT * FROM approval_steps WHERE instance_id=? AND step_index=?",
(instance_id, step_index),
).fetchone()
if not step:
raise KeyError("step not found")
if step["status"] != "pending":
raise ValueError("step already decided")
new_status = "approved" if approved else "rejected"
conn.execute(
"UPDATE approval_steps SET status=?, comment=?, time=? WHERE id=?",
(new_status, comment, utcnow(), step["id"]),
)
if approved:
conn.execute(
"UPDATE approval_instances SET current_step=? WHERE id=?",
(step_index + 1, instance_id),
)
step_rows = conn.execute(
"SELECT * FROM approval_steps WHERE instance_id=? ORDER BY step_index", (instance_id,)
).fetchall()
if all(s["status"] == "approved" for s in step_rows):
conn.execute("UPDATE approval_instances SET status='approved' WHERE id=?", (instance_id,))
else:
conn.execute("UPDATE approval_instances SET status='rejected' WHERE id=?", (instance_id,))
return self.approval_instance(instance_id)
# ===================== 平台治理:租户 =====================
def tenants(self) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute("SELECT * FROM tenants ORDER BY create_time DESC").fetchall()
return [dict(r) for r in rows]
def tenant(self, tenant_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM tenants WHERE id=?", (tenant_id,)).fetchone()
if not row:
raise KeyError(tenant_id)
return dict(row)
def create_tenant(self, payload: dict[str, Any]) -> dict[str, Any]:
with self.connect() as conn:
tid = new_id("tnt")
conn.execute(
"""
INSERT INTO tenants (id, name, code, status, owner_user_id, quota, retention_policy_id, create_time)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
""",
(
tid,
payload["name"],
payload.get("code"),
"active",
payload.get("owner_user_id"),
json_dumps(payload.get("quota", {})),
payload.get("retention_policy_id"),
utcnow(),
),
)
return self.tenant(tid)
def update_tenant(self, tenant_id: str, payload: dict[str, Any]) -> dict[str, Any]:
fields = {k: v for k, v in payload.items() if k in ("name", "code", "status", "owner_user_id", "quota", "retention_policy_id")}
if "quota" in fields:
fields["quota"] = json_dumps(fields["quota"])
if not fields:
return self.tenant(tenant_id)
set_clause = ", ".join(f"{k}=?" for k in fields)
params = list(fields.values()) + [tenant_id]
with self.connect() as conn:
conn.execute(f"UPDATE tenants SET {set_clause} WHERE id=?", tuple(params))
return self.tenant(tenant_id)
def set_tenant_quota(self, tenant_id: str, quota: dict[str, Any]) -> dict[str, Any]:
with self.connect() as conn:
conn.execute("UPDATE tenants SET quota=? WHERE id=?", (json_dumps(quota), tenant_id))
return self.tenant(tenant_id)
def set_tenant_retention(self, tenant_id: str, retention_policy_id: str | None) -> dict[str, Any]:
with self.connect() as conn:
conn.execute("UPDATE tenants SET retention_policy_id=? WHERE id=?", (retention_policy_id, tenant_id))
return self.tenant(tenant_id)
def delete_tenant(self, tenant_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM tenants WHERE id=?", (tenant_id,)).fetchone()
if not row:
raise KeyError(tenant_id)
conn.execute("DELETE FROM tenants WHERE id=?", (tenant_id,))
return dict(row)
def get_acl(self, resource_type: str, resource_id: str) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute(
"SELECT * FROM acls WHERE resource_type=? AND resource_id=?",
(resource_type, resource_id),
).fetchall()
return [dict(r) for r in rows]
def set_acl(self, resource_type: str, resource_id: str, entries: list[dict[str, Any]]) -> list[dict[str, Any]]:
with self.connect() as conn:
conn.execute(
"DELETE FROM acls WHERE resource_type=? AND resource_id=?",
(resource_type, resource_id),
)
for e in entries:
conn.execute(
"""
INSERT INTO acls (id, resource_type, resource_id, principal_type, principal_id, permission, create_time)
VALUES (?, ?, ?, ?, ?, ?, ?)
""",
(
new_id("acl"),
resource_type,
resource_id,
e.get("principal_type"),
e.get("principal_id"),
e.get("permission"),
utcnow(),
),
)
rows = conn.execute(
"SELECT * FROM acls WHERE resource_type=? AND resource_id=?",
(resource_type, resource_id),
).fetchall()
return [dict(r) for r in rows]
# ===================== 平台治理:项目空间 =====================
def projects(self, *, tenant_id: str = "default", status: str | None = None, keyword: str | None = None) -> list[dict[str, Any]]:
clauses = ["tenant_id=?"]
params: list[Any] = [tenant_id]
if status:
clauses.append("status=?")
params.append(status)
if keyword:
clauses.append("(name LIKE ? OR code LIKE ?)")
params.extend([f"%{keyword}%", f"%{keyword}%"])
with self.connect() as conn:
rows = conn.execute(
f"SELECT * FROM projects WHERE {' AND '.join(clauses)} ORDER BY create_time DESC",
tuple(params),
).fetchall()
return [dict(r) for r in rows]
def project(self, project_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM projects WHERE id=?", (project_id,)).fetchone()
if not row:
raise KeyError(project_id)
return dict(row)
def create_project(self, payload: dict[str, Any]) -> dict[str, Any]:
with self.connect() as conn:
pid = new_id("prj")
conn.execute(
"""
INSERT INTO projects (id, tenant_id, name, code, description, quota, status, create_time, create_by, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
pid,
payload.get("tenant_id", "default"),
payload["name"],
payload["code"],
payload.get("description"),
json_dumps(payload.get("quota", {})),
"active",
utcnow(),
payload.get("create_by"),
utcnow(),
),
)
return self.project(pid)
def update_project(self, project_id: str, payload: dict[str, Any]) -> dict[str, Any]:
fields = {k: v for k, v in payload.items() if k in ("name", "code", "description", "quota", "status")}
if "quota" in fields:
fields["quota"] = json_dumps(fields["quota"])
if not fields:
return self.project(project_id)
set_clause = ", ".join(f"{k}=?" for k in fields)
params = list(fields.values()) + [project_id]
with self.connect() as conn:
conn.execute(f"UPDATE projects SET {set_clause} WHERE id=?", tuple(params))
return self.project(project_id)
def archive_project(self, project_id: str) -> dict[str, Any]:
with self.connect() as conn:
conn.execute("UPDATE projects SET status='archived' WHERE id=?", (project_id,))
return self.project(project_id)
def activate_project(self, project_id: str) -> dict[str, Any]:
with self.connect() as conn:
conn.execute("UPDATE projects SET status='active' WHERE id=?", (project_id,))
return self.project(project_id)
def delete_project(self, project_id: str) -> None:
with self.connect() as conn:
conn.execute("DELETE FROM projects WHERE id=?", (project_id,))
def project_members(self, project_id: str) -> list[dict[str, Any]]:
with self.connect() as conn:
row = conn.execute("SELECT * FROM projects WHERE id=?", (project_id,)).fetchone()
if not row:
raise KeyError(project_id)
rows = conn.execute(
"""
SELECT pm.*, u.username, u.display_name
FROM project_members pm JOIN users u ON u.id = pm.user_id
WHERE pm.project_id=?
""",
(project_id,),
).fetchall()
return [dict(r) for r in rows]
def add_project_member(self, project_id: str, payload: dict[str, Any]) -> dict[str, Any]:
user_id = payload["user_id"]
role = payload.get("role", "member")
with self.connect() as conn:
conn.execute(
"INSERT INTO project_members (project_id, user_id, role, create_time) VALUES (?, ?, ?, ?)",
(project_id, user_id, role, utcnow()),
)
row = conn.execute(
"""
SELECT pm.*, u.username, u.display_name
FROM project_members pm JOIN users u ON u.id = pm.user_id
WHERE pm.project_id=? AND pm.user_id=?
""",
(project_id, user_id),
).fetchone()
return {
"project_id": row["project_id"],
"user_id": row["user_id"],
"username": row["username"],
"display_name": row["display_name"],
"role": row["role"],
"create_time": row["create_time"],
}
def update_project_member_role(self, project_id: str, user_id: str, role: str) -> dict[str, Any]:
with self.connect() as conn:
conn.execute(
"UPDATE project_members SET role=? WHERE project_id=? AND user_id=?",
(role, project_id, user_id),
)
row = conn.execute(
"""
SELECT pm.*, u.username, u.display_name
FROM project_members pm JOIN users u ON u.id = pm.user_id
WHERE pm.project_id=? AND pm.user_id=?
""",
(project_id, user_id),
).fetchone()
if not row:
raise KeyError(user_id)
return {
"project_id": row["project_id"],
"user_id": row["user_id"],
"username": row["username"],
"display_name": row["display_name"],
"role": row["role"],
"create_time": row["create_time"],
}
def remove_project_member(self, project_id: str, user_id: str) -> None:
with self.connect() as conn:
conn.execute(
"DELETE FROM project_members WHERE project_id=? AND user_id=?",
(project_id, user_id),
)
# ===================== 平台治理:资源 ACL =====================
def resource_acl(self, resource_type: str, resource_id: str) -> list[dict[str, Any]]:
"""返回资源 ACL按主体分组permissions 为数组。"""
rows = self.get_acl(resource_type, resource_id)
grouped: dict[str, dict[str, Any]] = {}
for r in rows:
key = f"{r.get('principal_type')}:{r.get('principal_id')}"
bucket = grouped.setdefault(
key,
{
"subject_type": r.get("principal_type"),
"subject_id": r.get("principal_id"),
"permissions": [],
},
)
perm = r.get("permission")
if perm and perm not in bucket["permissions"]:
bucket["permissions"].append(perm)
return list(grouped.values())
def set_resource_acl(
self, resource_type: str, resource_id: str, entries: list[dict[str, Any]]
) -> list[dict[str, Any]]:
"""按前端格式设置资源 ACLentries 为 [{subject_type, subject_id, permissions: []}]。"""
flat: list[dict[str, Any]] = []
for e in entries:
for perm in e.get("permissions") or []:
flat.append(
{
"principal_type": e.get("subject_type"),
"principal_id": e.get("subject_id"),
"permission": perm,
}
)
self.set_acl(resource_type, resource_id, flat)
return self.resource_acl(resource_type, resource_id)
# ===================== 平台治理:留存策略 =====================
def retention_policies(self) -> list[dict[str, Any]]:
with self.connect() as conn:
rows = conn.execute(
"SELECT * FROM retention_policies ORDER BY create_time DESC"
).fetchall()
return [dict(r) for r in rows]
def retention_policy(self, policy_id: str) -> dict[str, Any]:
with self.connect() as conn:
row = conn.execute(
"SELECT * FROM retention_policies WHERE id=?", (policy_id,)
).fetchone()
if not row:
raise KeyError(policy_id)
return dict(row)
def create_retention_policy(self, payload: dict[str, Any]) -> dict[str, Any]:
pid = payload.get("id") or new_id("rpol")
with self.connect() as conn:
conn.execute(
"""
INSERT INTO retention_policies
(id, name, scope, rule, status, create_time, create_by, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
""",
(
pid,
payload["name"],
payload.get("scope"),
payload.get("rule"),
payload.get("status", "active"),
utcnow(),
payload.get("create_by"),
utcnow(),
),
)
return self.retention_policy(pid)
def update_retention_policy(
self, policy_id: str, payload: dict[str, Any]
) -> dict[str, Any]:
fields = {
k: v
for k, v in payload.items()
if k in ("name", "scope", "rule", "status")
}
if not fields:
return self.retention_policy(policy_id)
fields["updated_at"] = utcnow()
set_clause = ", ".join(f"{k}=?" for k in fields)
params = list(fields.values()) + [policy_id]
with self.connect() as conn:
conn.execute(
f"UPDATE retention_policies SET {set_clause} WHERE id=?",
tuple(params),
)
return self.retention_policy(policy_id)
def delete_retention_policy(self, policy_id: str) -> None:
with self.connect() as conn:
conn.execute(
"DELETE FROM retention_policies WHERE id=?", (policy_id,)
)
_store: PlatformStore | None = None _store: PlatformStore | None = None
@@ -2963,3 +3680,16 @@ def get_platform_store() -> PlatformStore:
if _store is None: if _store is None:
_store = PlatformStore() _store = PlatformStore()
return _store return _store
import atexit as _atexit
def _close_store_pool() -> None:
global _store
if _store is not None:
_store.close_pool()
_store = None
_atexit.register(_close_store_pool)

View File

@@ -282,3 +282,51 @@ CREATE INDEX IF NOT EXISTS idx_sync_jobs_node_status ON resource_sync_jobs(targe
CREATE INDEX IF NOT EXISTS idx_eval_tasks_status ON eval_tasks(status); CREATE INDEX IF NOT EXISTS idx_eval_tasks_status ON eval_tasks(status);
CREATE INDEX IF NOT EXISTS idx_eval_dimensions_active ON eval_dimensions(is_active); CREATE INDEX IF NOT EXISTS idx_eval_dimensions_active ON eval_dimensions(is_active);
CREATE INDEX IF NOT EXISTS idx_compare_tasks_status ON compare_tasks(status); CREATE INDEX IF NOT EXISTS idx_compare_tasks_status ON compare_tasks(status);
-- ===================== Project / Tenant =====================
CREATE TABLE IF NOT EXISTS projects (
id TEXT PRIMARY KEY,
tenant_id TEXT NOT NULL DEFAULT 'default',
name TEXT NOT NULL,
code TEXT NOT NULL,
description TEXT,
quota TEXT,
status TEXT NOT NULL DEFAULT 'active',
create_time TEXT NOT NULL,
create_by TEXT,
updated_at TEXT
);
CREATE TABLE IF NOT EXISTS project_members (
project_id TEXT NOT NULL REFERENCES projects(id) ON DELETE CASCADE,
user_id TEXT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
role TEXT NOT NULL DEFAULT 'member',
create_time TEXT NOT NULL,
PRIMARY KEY (project_id, user_id)
);
CREATE TABLE IF NOT EXISTS roles (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
permissions TEXT NOT NULL DEFAULT '[]',
create_time TEXT
);
CREATE TABLE IF NOT EXISTS sessions (
id TEXT PRIMARY KEY,
user_id TEXT NOT NULL,
issued_at TEXT NOT NULL,
expires_at TEXT NOT NULL,
ip TEXT
);
CREATE TABLE IF NOT EXISTS acls (
id TEXT PRIMARY KEY,
resource_type TEXT NOT NULL,
resource_id TEXT NOT NULL,
principal_type TEXT NOT NULL,
principal_id TEXT NOT NULL,
permission TEXT NOT NULL,
create_time TEXT
);

View File

@@ -0,0 +1,68 @@
-- 平台治理:租户 / 审批 / 审计(字段以 platform_store 实际写入为准)
CREATE TABLE IF NOT EXISTS tenants (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
code TEXT,
status TEXT DEFAULT 'active',
owner_user_id TEXT,
quota TEXT,
retention_policy_id TEXT,
create_time TEXT
);
CREATE TABLE IF NOT EXISTS approval_templates (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
steps TEXT,
create_time TEXT
);
CREATE TABLE IF NOT EXISTS approval_instances (
id TEXT PRIMARY KEY,
template_id TEXT,
resource_type TEXT,
resource_id TEXT,
applicant_id TEXT,
status TEXT DEFAULT 'pending',
current_step INTEGER DEFAULT 0,
create_time TEXT
);
CREATE TABLE IF NOT EXISTS approval_steps (
id TEXT PRIMARY KEY,
instance_id TEXT,
step_index INTEGER,
approver_id TEXT,
status TEXT DEFAULT 'pending',
comment TEXT,
time TEXT
);
CREATE TABLE IF NOT EXISTS audit_logs (
id TEXT PRIMARY KEY,
tenant_id TEXT,
project_id TEXT,
actor_id TEXT,
action TEXT,
target_type TEXT,
target_id TEXT,
detail TEXT,
client_ip TEXT,
time TEXT
);
CREATE INDEX IF NOT EXISTS idx_audit_tenant ON audit_logs(tenant_id);
CREATE INDEX IF NOT EXISTS idx_audit_project ON audit_logs(project_id);
CREATE INDEX IF NOT EXISTS idx_audit_action ON audit_logs(action);
CREATE INDEX IF NOT EXISTS idx_audit_time ON audit_logs(time);
CREATE TABLE IF NOT EXISTS retention_policies (
id TEXT PRIMARY KEY,
name TEXT NOT NULL,
scope TEXT,
rule TEXT,
status TEXT DEFAULT 'active',
create_time TEXT,
create_by TEXT,
updated_at TEXT
);

View File

@@ -0,0 +1,3 @@
-- 租户配额与保留策略扩展(如后续治理表需补列,可在此追加)
ALTER TABLE tenants ADD COLUMN IF NOT EXISTS gpu_quota TEXT;
ALTER TABLE tenants ADD COLUMN IF NOT EXISTS storage_quota TEXT;

View File

@@ -0,0 +1,91 @@
from __future__ import annotations
from fastapi import APIRouter, Body
from typing import Any
from app.api.v1.endpoints.platform import ok, fail
from app.db.platform_store import get_platform_store
router = APIRouter(prefix="/approvals", tags=["approval"])
@router.get("/templates")
def list_templates() -> dict[str, Any]:
return ok(get_platform_store().approval_templates())
@router.post("/templates")
def create_template(payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
if not payload.get("name"):
raise fail(400, "name 必填")
return ok(get_platform_store().create_approval_template(payload))
@router.get("/templates/{template_id}")
def get_template(template_id: str) -> dict[str, Any]:
try:
return ok(get_platform_store().approval_template(template_id))
except KeyError:
raise fail(404, "template not found")
@router.put("/templates/{template_id}")
def update_template(template_id: str, payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
try:
return ok(get_platform_store().update_approval_template(template_id, payload))
except KeyError:
raise fail(404, "template not found")
@router.delete("/templates/{template_id}")
def delete_template(template_id: str) -> dict[str, Any]:
try:
return ok(get_platform_store().delete_approval_template(template_id))
except KeyError:
raise fail(404, "template not found")
@router.get("")
def list_instances(status: str | None = None) -> dict[str, Any]:
return ok(get_platform_store().approval_instances(status=status))
@router.post("")
def create_instance(payload: dict[str, Any] = Body(...)) -> dict[str, Any]:
for field in ("resource_type", "resource_id", "applicant_id"):
if not payload.get(field):
raise fail(400, f"{field} 必填")
try:
return ok(get_platform_store().create_approval_instance(payload))
except KeyError:
raise fail(404, "template not found")
@router.get("/{instance_id}")
def get_instance(instance_id: str) -> dict[str, Any]:
try:
return ok(get_platform_store().approval_instance(instance_id))
except KeyError:
raise fail(404, "instance not found")
@router.post("/{instance_id}/steps/{step_index}/decision")
def decide(
instance_id: str,
step_index: int,
payload: dict[str, Any] = Body(...),
) -> dict[str, Any]:
if not payload.get("approver_id"):
raise fail(400, "approver_id 必填")
try:
return ok(
get_platform_store().decide_approval_step(
instance_id,
step_index,
approver_id=payload["approver_id"],
approved=bool(payload.get("approved", False)),
comment=payload.get("comment"),
)
)
except (KeyError, ValueError) as e:
raise fail(400, str(e))

File diff suppressed because it is too large Load Diff

View File

@@ -137,6 +137,67 @@ class LocalDataProcessStorage:
self._issued_staged_objects[temporary_path] = staged self._issued_staged_objects[temporary_path] = staged
return staged return staged
def stage_copy(
self,
*,
batch_id: str,
source_reference: str,
expected_source_task_id: str,
expected_source_file_id: str,
task_id: str,
source_file_id: str,
version: int,
name: str,
) -> StagedSourceObject:
"""为不可变源对象创建独立目录项,不把大文件重新读入内存。"""
batch_id = _safe_component(batch_id, "batch id")
task_id = _safe_component(task_id, "task id")
source_file_id = _safe_component(source_file_id, "source file id")
if isinstance(version, bool) or not isinstance(version, int) or version < 1:
raise DataProcessStorageError("invalid source file version")
basename = _safe_basename(name)
source_relative = self._relative_from_reference(source_reference)
if source_relative is None:
raise DataProcessStorageError("original source object is not available")
self._assert_expected_owner(
source_relative,
expected_task_id=expected_source_task_id,
expected_source_file_id=expected_source_file_id,
)
descriptor, source_info = self._open_read_descriptor(source_relative)
os.close(descriptor)
batch_directory = self._ensure_directory(self._root / ".staging" / batch_id)
temporary_path = batch_directory / f"{source_file_id}-{uuid.uuid4().hex}.tmp"
source_path = self._path_for_relative(source_relative)
try:
os.link(source_path, temporary_path, follow_symlinks=False)
copy_info = temporary_path.lstat()
if (
not stat.S_ISREG(copy_info.st_mode)
or source_info.st_dev != copy_info.st_dev
or source_info.st_ino != copy_info.st_ino
):
raise DataProcessStorageError("source storage object changed while copying")
except Exception:
temporary_path.unlink(missing_ok=True)
raise
relative_path = PurePosixPath(
task_id,
source_file_id,
f"v{version}",
basename,
)
reference = (
"local://data-process/"
f"{task_id}/{source_file_id}/v{version}/{quote(basename, safe='')}"
)
staged = StagedSourceObject(reference, temporary_path, relative_path)
self._issued_staged_objects[temporary_path] = staged
return staged
def publish(self, objects: Iterable[StagedSourceObject]) -> None: def publish(self, objects: Iterable[StagedSourceObject]) -> None:
staged = list(objects) staged = list(objects)
published: list[StagedSourceObject] = [] published: list[StagedSourceObject] = []

View File

@@ -45,6 +45,13 @@ _UNSTRUCTURED_PREVIEW_DEFAULTS: dict[str, Any] = {
"preserve_lists": True, "preserve_lists": True,
} }
_REGENERATION_MARKER_KEY = "_regeneration_prepared" _REGENERATION_MARKER_KEY = "_regeneration_prepared"
_REPEAT_SOURCE_TASK_KEY = "_repeat_source_task_id"
_REPEAT_REQUEST_KEY = "_repeat_request_id"
_INTERNAL_CONFIG_KEYS = {
_REGENERATION_MARKER_KEY,
_REPEAT_SOURCE_TASK_KEY,
_REPEAT_REQUEST_KEY,
}
class DataProcessStoreError(RuntimeError): class DataProcessStoreError(RuntimeError):
@@ -71,6 +78,13 @@ def new_id(prefix: str) -> str:
return f"{prefix}_{uuid.uuid4().hex[:20]}" return f"{prefix}_{uuid.uuid4().hex[:20]}"
def repeat_task_id(source_task_id: str, request_id: str) -> str:
"""按源任务和请求幂等键生成稳定的新任务 ID。"""
digest = hashlib.sha256(f"{source_task_id}:{request_id}".encode()).hexdigest()
return f"dpt_{digest[:20]}"
def json_dumps(value: Any) -> str: def json_dumps(value: Any) -> str:
return json.dumps(value, ensure_ascii=False, separators=(",", ":")) return json.dumps(value, ensure_ascii=False, separators=(",", ":"))
@@ -183,17 +197,25 @@ def _is_regeneration_prepared(task: dict[str, Any]) -> bool:
return _regeneration_marker(task) is not None return _regeneration_marker(task) is not None
def _business_config(config: dict[str, Any] | None) -> dict[str, Any]:
"""过滤只供服务端维护的工作流标记。"""
return {
key: value
for key, value in (config or {}).items()
if key not in _INTERNAL_CONFIG_KEYS
}
def _public_task(item: dict[str, Any] | None) -> dict[str, Any] | None: def _public_task(item: dict[str, Any] | None) -> dict[str, Any] | None:
"""从 API 任务快照中移除服务端内部重新生成标记。""" """从 API 任务快照中移除服务端内部工作流标记。"""
if item is None: if item is None:
return None return None
public = dict(item) public = dict(item)
config = public.get("config") config = public.get("config")
if isinstance(config, dict) and _REGENERATION_MARKER_KEY in config: if isinstance(config, dict):
public["config"] = { public["config"] = _business_config(config)
key: value for key, value in config.items() if key != _REGENERATION_MARKER_KEY
}
return public return public
@@ -353,13 +375,7 @@ class DataProcessStore:
payload.get("description") or "", payload.get("description") or "",
payload["process_type"], payload["process_type"],
payload.get("source_dataset_id"), payload.get("source_dataset_id"),
json_dumps( json_dumps(_business_config(payload.get("config"))),
{
key: value
for key, value in (payload.get("config") or {}).items()
if key != _REGENERATION_MARKER_KEY
}
),
payload.get("tenant_id"), payload.get("tenant_id"),
payload.get("project_id"), payload.get("project_id"),
payload.get("owner_id"), payload.get("owner_id"),
@@ -373,6 +389,268 @@ class DataProcessStore:
raise ConflictError("data process task name already exists") from exc raise ConflictError("data process task name already exists") from exc
return _public_task(_decode_row(row)) or {} return _public_task(_decode_row(row)) or {}
@staticmethod
def _repeat_response(
conn: psycopg.Connection[dict[str, Any]],
row: dict[str, Any],
*,
source_task_id: str,
created: bool,
) -> dict[str, Any]:
task_id = str(row["id"])
counts = conn.execute(
"""
SELECT
(SELECT COUNT(*) FROM data_process_source_files
WHERE task_id=%s AND deleted_at IS NULL) AS source_file_count,
(SELECT COUNT(*) FROM data_process_preview_items
WHERE task_id=%s) AS preview_count
""",
(task_id, task_id),
).fetchone() or {}
task = _public_task(_decode_row(row)) or {}
task["source_file_count"] = int(counts.get("source_file_count") or 0)
task["preview_count"] = int(counts.get("preview_count") or 0)
return {
"task": task,
"source_task_id": source_task_id,
"created": created,
"copied_source_file_count": task["source_file_count"],
"copied_preview_count": task["preview_count"],
}
def find_repeated_task(
self,
source_task_id: str,
request_id: str,
) -> dict[str, Any] | None:
"""查找同一幂等请求已创建的新任务。"""
task_id = repeat_task_id(source_task_id, request_id)
with self.connect() as conn:
row = conn.execute(
"SELECT * FROM data_process_tasks WHERE id=%s",
(task_id,),
).fetchone()
if row is None:
return None
decoded = _decode_row(row) or {}
config = decoded.get("config") or {}
if (
config.get(_REPEAT_SOURCE_TASK_KEY) != source_task_id
or config.get(_REPEAT_REQUEST_KEY) != request_id
):
raise ConflictError("再次生成请求与现有任务冲突")
if decoded.get("deleted_at"):
raise ConflictError("此次再次生成创建的任务已被删除,请重新发起")
return self._repeat_response(
conn,
row,
source_task_id=source_task_id,
created=False,
)
def repeat_task(
self,
source_task_id: str,
*,
expected_updated_at: str,
request_id: str,
file_copies: dict[str, dict[str, str]],
) -> dict[str, Any]:
"""复制已确认任务的配置、源文件和预览,结果与发布数据保持独立。"""
task_id = repeat_task_id(source_task_id, request_id)
now = utcnow()
try:
with self.connect() as conn:
existing = conn.execute(
"SELECT * FROM data_process_tasks WHERE id=%s FOR UPDATE",
(task_id,),
).fetchone()
if existing is not None:
decoded = _decode_row(existing) or {}
config = decoded.get("config") or {}
if (
config.get(_REPEAT_SOURCE_TASK_KEY) != source_task_id
or config.get(_REPEAT_REQUEST_KEY) != request_id
):
raise ConflictError("再次生成请求与现有任务冲突")
if decoded.get("deleted_at"):
raise ConflictError("此次再次生成创建的任务已被删除,请重新发起")
return self._repeat_response(
conn,
existing,
source_task_id=source_task_id,
created=False,
)
source_task = self._task_in_connection(
conn,
source_task_id,
for_update=True,
)
if (
source_task.get("status") != "completed"
or source_task.get("results_confirmed") is False
):
raise InvalidStateError("只有已完成并确认结果的任务可以再次生成")
if source_task.get("preview_status") in ACTIVE_PREVIEW_STATUSES:
raise ConflictError("源任务仍在处理切分,暂时不能再次生成")
if expected_updated_at != _serialize_value(source_task.get("updated_at")):
raise ConflictError("源任务已被其他操作修改,请刷新后重试")
source_files = conn.execute(
"""
SELECT * FROM data_process_source_files
WHERE task_id=%s AND deleted_at IS NULL
ORDER BY created_at, id
""",
(source_task_id,),
).fetchall()
source_file_ids = {str(row["id"]) for row in source_files}
if source_file_ids != set(file_copies):
raise ConflictError("源文件快照已变化,请刷新后重试")
previews = conn.execute(
"""
SELECT * FROM data_process_preview_items
WHERE task_id=%s
ORDER BY source_file_id NULLS LAST, source_start NULLS LAST,
created_at, id
""",
(source_task_id,),
).fetchall()
if not previews:
raise InvalidStateError("源任务没有可用于再次生成的切分结果")
suffix = f"(再次生成-{task_id[-6:]}"
base_name = str(source_task.get("name") or "数据处理任务")
repeated_name = f"{base_name[: max(1, 150 - len(suffix))]}{suffix}"
repeated_config = _business_config(source_task.get("config") or {})
repeated_config[_REPEAT_SOURCE_TASK_KEY] = source_task_id
repeated_config[_REPEAT_REQUEST_KEY] = request_id
input_count = sum(int(row.get("record_count") or 0) for row in source_files)
task_row = conn.execute(
"""
INSERT INTO data_process_tasks
(id, name, description, status, process_type, source_dataset_id,
output_dataset_id, config, progress, input_count, output_count,
filtered_count, duplicate_count, error_count, failure_reason,
generation_run_id, results_confirmed, workflow_step,
preview_status, preview_progress, preview_run_id,
preview_failure_reason, preview_total_files,
preview_completed_files, tenant_id, project_id, owner_id,
approval_status, created_by, updated_by, created_at, updated_at)
VALUES
(%s, %s, %s, 'pending', %s, %s, NULL, %s, 20, %s, 0,
0, 0, 0, NULL, NULL, FALSE, 'preview', 'completed', 100,
NULL, NULL, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
RETURNING *
""",
(
task_id,
repeated_name,
source_task.get("description") or "",
source_task["process_type"],
source_task.get("source_dataset_id"),
json_dumps(repeated_config),
input_count,
len(source_files),
len(source_files),
source_task.get("tenant_id"),
source_task.get("project_id"),
source_task.get("owner_id"),
source_task.get("approval_status") or "not_required",
source_task.get("created_by"),
source_task.get("created_by"),
now,
now,
),
).fetchone()
file_id_map: dict[str, str] = {}
for source in source_files:
old_file_id = str(source["id"])
copy = file_copies[old_file_id]
new_file_id = str(copy["id"])
storage_object_id, metadata = _source_storage_descriptor(
{
"storage_object_id": copy["storage_object_id"],
"metadata": _json_value(source.get("metadata"), {}),
},
task_id,
new_file_id,
)
file_id_map[old_file_id] = new_file_id
conn.execute(
"""
INSERT INTO data_process_source_files
(id, task_id, storage_object_id, name, size_bytes, record_count,
file_format, checksum_sha256, version_no, content,
content_preview, metadata, tenant_id, project_id, created_by,
created_at, updated_at)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, 1, %s, %s, %s,
%s, %s, %s, %s, %s)
""",
(
new_file_id,
task_id,
storage_object_id,
source["name"],
source.get("size_bytes") or 0,
source.get("record_count") or 0,
source.get("file_format"),
source["checksum_sha256"],
source.get("content") or "",
source.get("content_preview"),
json_dumps(metadata),
source_task.get("tenant_id"),
source_task.get("project_id"),
source.get("created_by") or source_task.get("created_by"),
now,
now,
),
)
for preview in previews:
old_source_file_id = preview.get("source_file_id")
conn.execute(
"""
INSERT INTO data_process_preview_items
(id, task_id, source_file_id, original_content, edited_content,
source_start, source_end, source_start_line, source_end_line,
token_count, status, quality_score, created_at, updated_at)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s,
%s, %s)
""",
(
new_id("dpp"),
task_id,
file_id_map.get(str(old_source_file_id))
if old_source_file_id
else None,
preview.get("original_content") or "",
preview.get("edited_content") or "",
preview.get("source_start"),
preview.get("source_end"),
preview.get("source_start_line"),
preview.get("source_end_line"),
max(0, int(preview.get("token_count") or 0)),
preview.get("status") or "original",
json_dumps(_json_value(preview.get("quality_score"), {})),
now,
now,
),
)
return self._repeat_response(
conn,
task_row or {},
source_task_id=source_task_id,
created=True,
)
except psycopg.errors.UniqueViolation as exc:
raise ConflictError("再次生成任务名称或请求发生冲突,请重试") from exc
def get_task(self, task_id: str, *, for_update: bool = False) -> dict[str, Any]: def get_task(self, task_id: str, *, for_update: bool = False) -> dict[str, Any]:
lock = " FOR UPDATE" if for_update else "" lock = " FOR UPDATE" if for_update else ""
with self.connect() as conn: with self.connect() as conn:
@@ -654,14 +932,11 @@ class DataProcessStore:
"process type and source dataset cannot change during regeneration" "process type and source dataset cannot change during regeneration"
) )
if payload.get("config") is not None: if payload.get("config") is not None:
next_config = { next_config = _business_config(payload["config"])
key: value current_config = dict(task.get("config") or {})
for key, value in payload["config"].items() for key in _INTERNAL_CONFIG_KEYS:
if key != _REGENERATION_MARKER_KEY if key in current_config:
} next_config[key] = current_config[key]
current_marker = _regeneration_marker(task)
if current_marker:
next_config[_REGENERATION_MARKER_KEY] = current_marker
values["config"] = json_dumps(next_config) values["config"] = json_dumps(next_config)
invalidates_results = ( invalidates_results = (
("config" in payload and payload.get("config") != task.get("config")) ("config" in payload and payload.get("config") != task.get("config"))
@@ -753,8 +1028,10 @@ class DataProcessStore:
raise InvalidStateError("process_type cannot be changed during regeneration") raise InvalidStateError("process_type cannot be changed during regeneration")
current_config = dict(task.get("config") or {}) current_config = dict(task.get("config") or {})
next_config = dict(payload.get("config") or {}) next_config = _business_config(payload.get("config"))
next_config.pop(_REGENERATION_MARKER_KEY, None) for key in (_REPEAT_SOURCE_TASK_KEY, _REPEAT_REQUEST_KEY):
if key in current_config:
next_config[key] = current_config[key]
preview_invalidated = _preview_config_changed( preview_invalidated = _preview_config_changed(
process_type, process_type,
current_config, current_config,

View File

@@ -0,0 +1,244 @@
from __future__ import annotations
from fastapi import APIRouter, Body, Depends, Request
from typing import Any
from app.api.v1.endpoints.platform import ok, fail
from app.core.auth import filter_accessible_resource_ids, get_current_user, has_resource_access, is_admin
from app.db.platform_store import get_platform_store
router = APIRouter(prefix="/projects", tags=["project"])
def _actor(request: Request) -> str | None:
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
return token or None
def _require_no_pending_approval(resource_type: str, resource_id: str) -> None:
"""第 4 周:写操作审批拦截——存在待审批实例时拒绝执行。"""
store = get_platform_store()
pending = [
i for i in store.approval_instances(status="pending")
if i["resource_type"] == resource_type and i["resource_id"] == resource_id
]
if pending:
raise fail(409, "存在待审批的变更,请先完成审批")
def _require_approval_or_admin(
resource_type: str,
resource_id: str,
current_user: dict[str, Any],
action_desc: str = "",
) -> dict[str, Any] | None:
"""高风险操作审批旁路admin 直接放行普通用户创建审批实例code=202"""
if is_admin(current_user):
return None
store = get_platform_store()
instance = store.create_approval_instance({
"resource_type": resource_type,
"resource_id": resource_id,
"applicant_id": current_user.get("id"),
"template_id": None,
})
return {
"code": 202,
"message": f"操作已提交审批,等待管理员批准:{action_desc}",
"data": {"approval_required": True, "approval_id": instance["id"]},
}
@router.get("")
def list_projects(
tenant_id: str = "default",
status: str | None = None,
keyword: str | None = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
store = get_platform_store()
projects = store.projects(tenant_id=tenant_id, status=status, keyword=keyword)
# #1 ACL 过滤admin 直接放行,普通用户只能看到自己被授权的项目
accessible_ids = set(
filter_accessible_resource_ids("project", [p["id"] for p in projects], current_user)
)
filtered = [p for p in projects if p["id"] in accessible_ids]
return ok(filtered)
@router.post("")
def create_project(payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
store = get_platform_store()
proj = store.create_project(payload)
store.record_audit(
action="project.create",
actor_id=_actor(request) if request else None,
target_type="project",
target_id=proj["id"],
tenant_id=proj.get("tenant_id"),
detail=f"name={proj.get('name')}",
)
return ok(proj)
@router.get("/{project_id}")
def get_project(project_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
# #2 访问控制:普通用户无 read 权限则拒绝
if not has_resource_access("project", project_id, current_user, "read"):
raise fail(403, "no permission to access this project")
try:
return ok(get_platform_store().project(project_id))
except KeyError:
raise fail(404, "project not found")
@router.put("/{project_id}")
def update_project(
project_id: str,
payload: dict[str, Any] = Body(...),
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
if not has_resource_access("project", project_id, current_user, "write"):
raise fail(403, "no permission to update this project")
store = get_platform_store()
try:
proj = store.update_project(project_id, payload)
except KeyError:
raise fail(404, "project not found")
store.record_audit(
action="project.update",
actor_id=_actor(request) if request else None,
target_type="project",
target_id=project_id,
tenant_id=proj.get("tenant_id"),
detail=f"fields={','.join(payload.keys())}",
)
return ok(proj)
@router.post("/{project_id}/archive")
def archive_project(
project_id: str,
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
_require_no_pending_approval("project", project_id)
pending = _require_approval_or_admin("project", project_id, current_user, f"归档项目 {project_id}")
if pending:
return pending
store = get_platform_store()
try:
proj = store.archive_project(project_id)
except KeyError:
raise fail(404, "project not found")
store.record_audit(
action="project.archive",
actor_id=_actor(request) if request else None,
target_type="project",
target_id=project_id,
tenant_id=proj.get("tenant_id"),
)
return ok(proj)
@router.delete("/{project_id}")
def delete_project(
project_id: str,
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
_require_no_pending_approval("project", project_id)
pending = _require_approval_or_admin("project", project_id, current_user, f"删除项目 {project_id}")
if pending:
return pending
store = get_platform_store()
store.delete_project(project_id)
store.record_audit(
action="project.delete",
actor_id=_actor(request) if request else None,
target_type="project",
target_id=project_id,
)
return ok(None)
@router.get("/{project_id}/members")
def list_members(project_id: str, current_user: dict = Depends(get_current_user)) -> dict[str, Any]:
if not has_resource_access("project", project_id, current_user, "read"):
raise fail(403, "no permission to access this project")
try:
return ok(get_platform_store().project_members(project_id))
except KeyError:
raise fail(404, "project not found")
@router.post("/{project_id}/members")
def add_member(
project_id: str,
payload: dict[str, Any] = Body(...),
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
if not has_resource_access("project", project_id, current_user, "write"):
raise fail(403, "no permission to manage members of this project")
store = get_platform_store()
try:
member = store.add_project_member(project_id, payload)
except KeyError:
raise fail(404, "project not found")
store.record_audit(
action="project.member.add",
actor_id=_actor(request) if request else None,
target_type="project.member",
target_id=project_id,
detail=f"user_id={payload.get('user_id')},role={payload.get('role')}",
)
return ok(member)
@router.put("/{project_id}/members/{user_id}")
def update_member(
project_id: str,
user_id: str,
payload: dict[str, Any] = Body(...),
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
if not has_resource_access("project", project_id, current_user, "write"):
raise fail(403, "no permission to manage members of this project")
store = get_platform_store()
try:
member = store.update_project_member_role(project_id, user_id, payload)
except KeyError:
raise fail(404, "project or member not found")
store.record_audit(
action="project.member.update",
actor_id=_actor(request) if request else None,
target_type="project.member",
target_id=project_id,
detail=f"user_id={user_id},role={payload.get('role')}",
)
return ok(member)
@router.delete("/{project_id}/members/{user_id}")
def remove_member(
project_id: str,
user_id: str,
request: Request = None,
current_user: dict = Depends(get_current_user),
) -> dict[str, Any]:
if not has_resource_access("project", project_id, current_user, "write"):
raise fail(403, "no permission to manage members of this project")
store = get_platform_store()
store.remove_project_member(project_id, user_id)
store.record_audit(
action="project.member.remove",
actor_id=_actor(request) if request else None,
target_type="project.member",
target_id=project_id,
detail=f"user_id={user_id}",
)
return ok(None)

View File

@@ -0,0 +1 @@
"""Resource access control list (ACL) module."""

View File

@@ -0,0 +1,41 @@
from __future__ import annotations
from fastapi import APIRouter, Body, Request
from typing import Any
from app.api.v1.endpoints.platform import ok, fail
from app.db.platform_store import get_platform_store
router = APIRouter(prefix="/resources", tags=["resource"])
def _actor(request: Request) -> str | None:
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
return token or None
@router.get("/{resource_type}/{resource_id}/acl")
def get_acl(resource_type: str, resource_id: str) -> dict[str, Any]:
"""查询资源 ACL返回按主体分组的权限列表。"""
return ok(get_platform_store().resource_acl(resource_type, resource_id))
@router.put("/{resource_type}/{resource_id}/acl")
def set_acl(
resource_type: str,
resource_id: str,
payload: dict[str, Any] = Body(...),
request: Request = None,
) -> dict[str, Any]:
"""设置资源 ACLbody: { entries: [{ subject_type, subject_id, permissions: [] }] }"""
entries = payload.get("entries") or []
result = get_platform_store().set_resource_acl(resource_type, resource_id, entries)
get_platform_store().record_audit(
action="resource.acl.set",
actor_id=_actor(request) if request else None,
target_type=resource_type,
target_id=resource_id,
detail=f"entries={len(entries)}",
)
return ok(result)

View File

@@ -0,0 +1,75 @@
from __future__ import annotations
from fastapi import APIRouter, Body, Request
from typing import Any
from app.api.v1.endpoints.platform import ok, fail
from app.db.platform_store import get_platform_store
router = APIRouter(prefix="/retention-policies", tags=["retention"])
def _actor(request: Request) -> str | None:
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
return token or None
@router.get("")
def list_policies() -> dict[str, Any]:
return ok(get_platform_store().retention_policies())
@router.post("")
def create_policy(payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
if not payload.get("name"):
raise fail(400, "name 必填")
policy = get_platform_store().create_retention_policy(payload)
get_platform_store().record_audit(
action="retention.create",
actor_id=_actor(request) if request else None,
target_type="retention_policy",
target_id=policy["id"],
detail=f"name={policy.get('name')}",
)
return ok(policy)
@router.get("/{policy_id}")
def get_policy(policy_id: str) -> dict[str, Any]:
try:
return ok(get_platform_store().retention_policy(policy_id))
except KeyError:
raise fail(404, "retention policy not found")
@router.put("/{policy_id}")
def update_policy(
policy_id: str, payload: dict[str, Any] = Body(...), request: Request = None
) -> dict[str, Any]:
store = get_platform_store()
try:
policy = store.update_retention_policy(policy_id, payload)
except KeyError:
raise fail(404, "retention policy not found")
store.record_audit(
action="retention.update",
actor_id=_actor(request) if request else None,
target_type="retention_policy",
target_id=policy_id,
detail=f"fields={','.join(payload.keys())}",
)
return ok(policy)
@router.delete("/{policy_id}")
def delete_policy(policy_id: str, request: Request = None) -> dict[str, Any]:
store = get_platform_store()
store.delete_retention_policy(policy_id)
store.record_audit(
action="retention.delete",
actor_id=_actor(request) if request else None,
target_type="retention_policy",
target_id=policy_id,
)
return ok({"deleted": policy_id})

View File

@@ -0,0 +1,93 @@
from __future__ import annotations
from fastapi import APIRouter, Query
from fastapi.responses import StreamingResponse
from app.db.platform_store import ALL_PERMISSIONS, get_platform_store
router = APIRouter(prefix="/system", tags=["system"])
@router.get("/permissions/codes")
def permission_codes() -> dict:
"""返回平台权限码清单(权限码接口)。"""
return {"code": 0, "message": "ok", "data": {"codes": ALL_PERMISSIONS}}
@router.get("/permissions")
def permissions_overview() -> dict:
"""返回权限码清单与角色定义。"""
store = get_platform_store()
return {
"code": 0,
"message": "ok",
"data": {"codes": ALL_PERMISSIONS, "roles": store.roles()},
}
@router.get("/audit-logs")
def audit_logs(
tenant_id: str | None = Query(default=None, description="租户 ID"),
project_id: str | None = Query(default=None, description="项目 ID"),
actor_id: str | None = Query(default=None, description="操作人 ID"),
action: str | None = Query(default=None, description="动作类型"),
target_type: str | None = Query(default=None, description="目标类型"),
start_time: str | None = Query(default=None, description="ISO8601 起始时间"),
end_time: str | None = Query(default=None, description="ISO8601 结束时间"),
limit: int = Query(default=50, ge=1, le=200),
offset: int = Query(default=0, ge=0),
) -> dict:
"""审计日志查询:按租户/项目/操作人/动作/目标类型/时间范围分页过滤。"""
store = get_platform_store()
result = store.audit_logs(
tenant_id=tenant_id,
project_id=project_id,
actor_id=actor_id,
action=action,
target_type=target_type,
start_time=start_time,
end_time=end_time,
limit=limit,
offset=offset,
)
return {"code": 0, "message": "ok", "data": result}
@router.get("/audit-logs/export")
def audit_logs_export(
tenant_id: str | None = Query(default=None, description="租户 ID"),
project_id: str | None = Query(default=None, description="项目 ID"),
actor_id: str | None = Query(default=None, description="操作人 ID"),
action: str | None = Query(default=None, description="动作类型"),
target_type: str | None = Query(default=None, description="目标类型"),
start_time: str | None = Query(default=None, description="ISO8601 起始时间"),
end_time: str | None = Query(default=None, description="ISO8601 结束时间"),
) -> StreamingResponse:
"""审计日志导出:返回 CSV 流,与应用查询相同的过滤条件。"""
store = get_platform_store()
result = store.audit_logs(
tenant_id=tenant_id,
project_id=project_id,
actor_id=actor_id,
action=action,
target_type=target_type,
start_time=start_time,
end_time=end_time,
limit=10000,
offset=0,
)
items = result["items"]
columns = ["time", "tenant_id", "project_id", "actor_id", "action", "target_type", "target_id", "detail", "client_ip"]
header = ",".join(columns) + "\n"
def iter_rows():
yield header
for row in items:
yield ",".join(f'"{str(row.get(c, "") or "")}"' for c in columns) + "\n"
return StreamingResponse(
iter_rows(),
media_type="text/csv",
headers={"Content-Disposition": "attachment; filename=audit_logs.csv"},
)

View File

@@ -0,0 +1,116 @@
from __future__ import annotations
from fastapi import APIRouter, Body, Request
from typing import Any
from app.api.v1.endpoints.platform import ok, fail
from app.db.platform_store import get_platform_store
router = APIRouter(prefix="/tenants", tags=["tenant"])
def _actor(request: Request) -> str | None:
auth = request.headers.get("Authorization", "")
token = auth.replace("Bearer ", "").strip()
return token or None
@router.get("")
def list_tenants() -> dict[str, Any]:
return ok(get_platform_store().tenants())
@router.post("")
def create_tenant(payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
store = get_platform_store()
try:
tenant = store.create_tenant(payload)
except KeyError as e:
raise fail(400, f"missing field: {e}")
store.record_audit(
action="tenant.create",
actor_id=_actor(request) if request else None,
target_type="tenant",
target_id=tenant["id"],
tenant_id=tenant["id"],
detail=f"name={tenant.get('name')}",
)
return ok(tenant)
@router.get("/{tenant_id}")
def get_tenant(tenant_id: str) -> dict[str, Any]:
try:
return ok(get_platform_store().tenant(tenant_id))
except KeyError:
raise fail(404, "tenant not found")
@router.put("/{tenant_id}")
def update_tenant(tenant_id: str, payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
store = get_platform_store()
try:
tenant = store.update_tenant(tenant_id, payload)
except KeyError:
raise fail(404, "tenant not found")
store.record_audit(
action="tenant.update",
actor_id=_actor(request) if request else None,
target_type="tenant",
target_id=tenant_id,
tenant_id=tenant_id,
detail=f"fields={','.join(payload.keys())}",
)
return ok(tenant)
@router.put("/{tenant_id}/quota")
def set_quota(tenant_id: str, payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
store = get_platform_store()
try:
tenant = store.set_tenant_quota(tenant_id, payload.get("quota", {}))
except KeyError:
raise fail(404, "tenant not found")
store.record_audit(
action="tenant.quota.set",
actor_id=_actor(request) if request else None,
target_type="tenant",
target_id=tenant_id,
tenant_id=tenant_id,
)
return ok(tenant)
@router.put("/{tenant_id}/retention-policy")
def set_retention(tenant_id: str, payload: dict[str, Any] = Body(...), request: Request = None) -> dict[str, Any]:
store = get_platform_store()
try:
tenant = store.set_tenant_retention(tenant_id, payload.get("retention_policy_id"))
except KeyError:
raise fail(404, "tenant not found")
store.record_audit(
action="tenant.retention.set",
actor_id=_actor(request) if request else None,
target_type="tenant",
target_id=tenant_id,
tenant_id=tenant_id,
)
return ok(tenant)
@router.delete("/{tenant_id}")
def delete_tenant(tenant_id: str, request: Request = None) -> dict[str, Any]:
store = get_platform_store()
try:
tenant = store.delete_tenant(tenant_id)
except KeyError:
raise fail(404, "tenant not found")
store.record_audit(
action="tenant.delete",
actor_id=_actor(request) if request else None,
target_type="tenant",
target_id=tenant_id,
tenant_id=tenant_id,
detail=f"name={tenant.get('name')}",
)
return ok(tenant)

View File

@@ -220,6 +220,19 @@ class DataProcessRegenerateRequest(BaseModel):
return self return self
class DataProcessRepeatRequest(BaseModel):
"""按已确认任务的完整快照创建一批独立的新生成结果。"""
model_config = ConfigDict(extra="forbid")
expected_updated_at: str = Field(min_length=1)
request_id: str = Field(
min_length=8,
max_length=80,
pattern=r"^[A-Za-z0-9_-]+$",
)
class PreviewBuildRequest(BaseModel): class PreviewBuildRequest(BaseModel):
model_config = ConfigDict(extra="forbid") model_config = ConfigDict(extra="forbid")

View File

@@ -5,6 +5,7 @@ import json
import xml.etree.ElementTree as ET import xml.etree.ElementTree as ET
import zipfile import zipfile
from datetime import datetime from datetime import datetime
from decimal import Decimal
import pytest import pytest
from docx import Document from docx import Document
@@ -29,11 +30,13 @@ from app.modules.data_process.algorithms import (
normalize_text, normalize_text,
parse_text_content, parse_text_content,
preprocess_structured_records, preprocess_structured_records,
preprocess_structured_records_with_lineage,
record_fingerprint, record_fingerprint,
remove_document_noise, remove_document_noise,
score_quality, score_quality,
stable_split, stable_split,
stable_split_assignments, stable_split_assignments,
structured_json_dumps,
) )
@@ -194,6 +197,75 @@ def test_parse_utf8_json_jsonl_csv_markdown_and_txt() -> None:
assert parsed_txt.text == "普通文本" assert parsed_txt.text == "普通文本"
def test_structured_text_record_locators_preserve_logical_source_positions() -> None:
root_json = parse_text_content('{"id":1}', filename="root.json")
assert root_json.record_locators == (
{
"kind": "json",
"record_index": 1,
"json_pointer": "",
"source_start": 0,
"source_end": 8,
"start_line": 1,
"end_line": 1,
},
)
wrapped_json = parse_text_content(
'{"records":[{"id":1},{"id":1}]}',
filename="wrapped.json",
)
assert [locator["json_pointer"] for locator in wrapped_json.record_locators] == [
"/records/0",
"/records/1",
]
parsed_jsonl = parse_text_content(
'{"id":1}\r\n\r\n{"id":1}',
filename="records.jsonl",
)
assert [
(locator["record_index"], locator["start_line"], locator["end_line"])
for locator in parsed_jsonl.record_locators
] == [(1, 1, 1), (2, 3, 3)]
assert [
parsed_jsonl.text[locator["source_start"] : locator["source_end"]]
for locator in parsed_jsonl.record_locators
] == ['{"id":1}', '{"id":1}']
parsed_csv = parse_text_content(
'id,note\r\n1,"hello\r\nworld"\r\n\r\n2,plain',
filename="records.csv",
)
assert [
(locator["record_index"], locator["start_line"], locator["end_line"])
for locator in parsed_csv.record_locators
] == [(1, 2, 3), (2, 5, 5)]
assert [
parsed_csv.text[locator["source_start"] : locator["source_end"]]
for locator in parsed_csv.record_locators
] == ['1,"hello\nworld"', "2,plain"]
def test_structured_preprocess_lineage_survives_column_cleanup_and_row_removal() -> None:
processed = preprocess_structured_records_with_lineage(
[
{"id": "A", "value": "first", "empty": ""},
{"id": "", "value": "invalid", "empty": ""},
{"id": "A", "value": "duplicate identity", "empty": ""},
{"id": "B", "value": "second", "empty": ""},
],
["clean_invalid", "deduplicate"],
)
assert [entry.source_index for entry in processed] == [0, 1, 2, 3]
assert [entry.record for entry in processed] == [
{"id": "A", "value": "first"},
{"id": "", "value": "invalid"},
{"id": "A", "value": "duplicate identity"},
{"id": "B", "value": "second"},
]
def test_parse_pdf_docx_xlsx_and_pptx() -> None: def test_parse_pdf_docx_xlsx_and_pptx() -> None:
parsed_pdf = parse_text_content(_minimal_pdf(), filename="manual.pdf") parsed_pdf = parse_text_content(_minimal_pdf(), filename="manual.pdf")
assert parsed_pdf.format == "pdf" assert parsed_pdf.format == "pdf"
@@ -220,6 +292,24 @@ def test_parse_pdf_docx_xlsx_and_pptx() -> None:
{"name": "Alice", "score": 95, "created_at": "2026-07-23T10:30:00"}, {"name": "Alice", "score": 95, "created_at": "2026-07-23T10:30:00"},
{"name": "Bob", "score": 88, "created_at": "2026-07-24T09:00:00"}, {"name": "Bob", "score": 88, "created_at": "2026-07-24T09:00:00"},
) )
assert parsed_xlsx.record_locators == (
{
"kind": "xlsx",
"record_index": 1,
"sheet_index": 0,
"sheet_name": "数据",
"row_number": 2,
"sheet_record_index": 0,
},
{
"kind": "xlsx",
"record_index": 2,
"sheet_index": 0,
"sheet_name": "数据",
"row_number": 3,
"sheet_record_index": 1,
},
)
assert json.loads(parsed_xlsx.text.splitlines()[0]) == parsed_xlsx.records[0] assert json.loads(parsed_xlsx.text.splitlines()[0]) == parsed_xlsx.records[0]
parsed_pptx = parse_text_content(_pptx_bytes(), filename="slides.pptx") parsed_pptx = parse_text_content(_pptx_bytes(), filename="slides.pptx")
@@ -228,6 +318,44 @@ def test_parse_pdf_docx_xlsx_and_pptx() -> None:
assert parsed_pptx.records == () assert parsed_pptx.records == ()
def test_xlsx_record_locators_distinguish_sheets_rows_and_duplicate_records() -> None:
workbook = Workbook()
first = workbook.active
first.title = "甲表"
first.append(["说明"])
first.append([])
first.append(["id", "value"])
first.append([1, "same"])
first.append([1, "same"])
second = workbook.create_sheet("乙表")
second.append(["id", "value"])
second.append([1, "same"])
output = io.BytesIO()
workbook.save(output)
workbook.close()
parsed = parse_text_content(output.getvalue(), filename="duplicate.xlsx")
assert parsed.records == (
{"id": 1, "value": "same"},
{"id": 1, "value": "same"},
{"id": 1, "value": "same"},
)
assert [
(
locator["record_index"],
locator["sheet_index"],
locator["sheet_name"],
locator["row_number"],
locator["sheet_record_index"],
)
for locator in parsed.record_locators
] == [
(1, 0, "甲表", 4, 0),
(2, 0, "甲表", 5, 1),
(3, 1, "乙表", 2, 0),
]
def test_pdf_document_noise_removes_headers_page_numbers_and_toc_safely() -> None: def test_pdf_document_noise_removes_headers_page_numbers_and_toc_safely() -> None:
pages = _pdf_page_texts( pages = _pdf_page_texts(
""" """
@@ -568,7 +696,130 @@ def test_extract_json_scalar_and_nested_values_are_stable() -> None:
json.dumps({"items": [{"text": " 内容 "}], "ignored": 1}, ensure_ascii=False), json.dumps({"items": [{"text": " 内容 "}], "ignored": 1}, ensure_ascii=False),
"json", "json",
) )
assert result == [{"text": "内容"}] assert result == [{"items": [{"text": " 内容 "}], "ignored": 1}]
assert extract_structured_records(
'{"items":[{"text":" 内容 "}],"total":1}',
"json",
) == [{"text": " 内容 "}]
def test_json_parsing_is_strict_and_preserves_field_values() -> None:
source = '{"code":"","text":" 内容 ","quote":""}'
parsed = parse_text_content(source, filename="records.json")
assert parsed.text == source
assert parsed.records == (
{"code": "", "text": " 内容 ", "quote": ""},
)
invalid_values = (
'{"id":1,"id":2}',
'{"nested":{"id":1,"id":2}}',
'{"value":NaN}',
'{"value":Infinity}',
'{"value":-Infinity}',
'{"value":"bad\x00control"}',
)
for invalid in invalid_values:
with pytest.raises(ValueError):
parse_text_content(invalid, filename="invalid.json")
with pytest.raises(ValueError):
parse_text_content("\"id\":1", filename="invalid.json")
with pytest.raises(ValueError, match="nesting exceeds"):
parse_text_content("[" * 65 + "0" + "]" * 65, filename="deep.json")
def test_jsonl_uses_the_same_strict_lossless_number_and_text_contract() -> None:
source = (
' {"code":"","text":" 内容 ",'
'"value":0.123456789012345678901234567890}\r\n\r\n'
'{"id":2}\r\n'
)
parsed = parse_text_content(source, filename="records.jsonl")
assert parsed.text == source
assert parsed.records[0] == {
"code": "",
"text": " 内容 ",
"value": Decimal("0.123456789012345678901234567890"),
}
assert [
source[locator["source_start"] : locator["source_end"]]
for locator in parsed.record_locators
] == [
(
'{"code":"","text":" 内容 ",'
'"value":0.123456789012345678901234567890}'
),
'{"id":2}',
]
assert [locator["start_line"] for locator in parsed.record_locators] == [1, 3]
for invalid in ('{"id":1,"id":2}', '{"value":NaN}'):
with pytest.raises(ValueError, match="invalid JSONL at line 1"):
parse_text_content(invalid, filename="invalid.jsonl")
def test_json_record_contract_avoids_business_field_collisions() -> None:
assert extract_structured_records('[{"id":1},{"id":2}]', "json") == [
{"id": 1},
{"id": 2},
]
assert extract_structured_records('{"id":1,"data":[{"id":2}]}', "json") == [
{"id": 1, "data": [{"id": 2}]}
]
assert extract_structured_records(
'{"records":[{"id":1}],"data":[{"id":2}]}',
"json",
) == [{"records": [{"id": 1}], "data": [{"id": 2}]}]
assert extract_structured_records(
'{"response":{"data":[{"id":1}],"status":"ok"},"success":true,"code":0}',
"json",
) == [{"id": 1}]
assert extract_structured_records(
'{"payload":{"data":[{"id":2}],"total":1}}',
"json",
) == [{"id": 2}]
assert extract_structured_records('{"records":[],"total":0}', "json") == []
# 包装数组中的非对象不是记录集合,整体按一条业务对象保留。
assert extract_structured_records('{"data":[1,2]}', "json") == [
{"data": [1, 2]}
]
def test_json_record_locators_cover_pretty_and_minified_sources() -> None:
pretty = (
'{\n "records": [\n {"id": 1},\n'
' {\n "id": 2\n }\n ],\n "total": 2\n}'
)
parsed = parse_text_content(pretty, filename="pretty.json")
assert [
pretty[locator["source_start"] : locator["source_end"]]
for locator in parsed.record_locators
] == ['{"id": 1}', '{\n "id": 2\n }']
assert [
(locator["start_line"], locator["end_line"])
for locator in parsed.record_locators
] == [(3, 3), (4, 6)]
minified = '[{"id":1},{"id":2}]'
parsed = parse_text_content(minified, filename="minified.json")
assert [
minified[locator["source_start"] : locator["source_end"]]
for locator in parsed.record_locators
] == ['{"id":1}', '{"id":2}']
def test_high_precision_json_numbers_serialize_without_type_or_value_loss() -> None:
source = '[{"value":0.123456789012345678901234567890},{"value":1e400}]'
parsed = parse_text_content(source, filename="precise.json")
assert parsed.records[0]["value"] == Decimal("0.123456789012345678901234567890")
assert parsed.records[1]["value"] == Decimal("1e400")
assert structured_json_dumps(parsed.records[0]) == (
'{"value":0.123456789012345678901234567890}'
)
assert structured_json_dumps(parsed.records[1]) == '{"value":1E+400}'
assert isinstance(parsed.records[0]["value"], Decimal)
def test_desensitize_pii_returns_masked_text_and_counts() -> None: def test_desensitize_pii_returns_masked_text_and_counts() -> None:
@@ -587,9 +838,20 @@ def test_every_structured_preprocess_option_has_independent_behavior() -> None:
assert preprocess_structured_records(clean_source, []) == clean_source assert preprocess_structured_records(clean_source, []) == clean_source
assert preprocess_structured_records(clean_source, ["clean_invalid"]) == [ assert preprocess_structured_records(clean_source, ["clean_invalid"]) == [
{"id": "1", "name": "有效"}, {"id": "1", "name": "有效"},
{"id": "", "name": "缺少关键字段"},
{"id": "2", "name": "有效"}, {"id": "2", "name": "有效"},
] ]
hierarchy = [
{"id": "1", "parent_id": None, "name": "根节点", "empty": ""},
{"id": "2", "parent_id": "1", "name": "子节点", "empty": ""},
{"id": "", "parent_id": "", "name": "", "empty": ""},
]
assert preprocess_structured_records(hierarchy, ["clean_invalid"]) == [
{"id": "1", "parent_id": None, "name": "根节点"},
{"id": "2", "parent_id": "1", "name": "子节点"},
]
nested = [{"id": 1, "profile": {"name": "张三", "level": 2}}] nested = [{"id": 1, "profile": {"name": "张三", "level": 2}}]
assert "profile" in preprocess_structured_records(nested, [])[0] assert "profile" in preprocess_structured_records(nested, [])[0]
assert preprocess_structured_records(nested, ["detect_structure"])[0] == { assert preprocess_structured_records(nested, ["detect_structure"])[0] == {
@@ -601,13 +863,15 @@ def test_every_structured_preprocess_option_has_independent_behavior() -> None:
duplicates = [ duplicates = [
{"customer_id": "C-1", "value": "first"}, {"customer_id": "C-1", "value": "first"},
{"customer_id": "C-1", "value": "updated"}, {"customer_id": "C-1", "value": "updated"},
{"customer_id": "C-1", "value": "first"},
{"customer_id": "", "value": "blank-one"}, {"customer_id": "", "value": "blank-one"},
{"customer_id": "", "value": "blank-two"}, {"customer_id": "", "value": "blank-two"},
] ]
assert len(preprocess_structured_records(duplicates, [])) == 4 assert len(preprocess_structured_records(duplicates, [])) == 5
deduplicated = preprocess_structured_records(duplicates, ["deduplicate"]) deduplicated = preprocess_structured_records(duplicates, ["deduplicate"])
assert [record["value"] for record in deduplicated] == [ assert [record["value"] for record in deduplicated] == [
"first", "first",
"updated",
"blank-one", "blank-one",
"blank-two", "blank-two",
] ]
@@ -660,6 +924,41 @@ def test_structured_desensitization_counts_and_document_helpers() -> None:
) )
def test_structured_desensitization_only_masks_explicit_person_name_fields() -> None:
masked, counts = desensitize_structured_record(
{
"table_name": "customer_profile",
"chinese_name": "zh_CN",
"english_name": "en_US",
"product_name": "智能助手",
"metadata.table_name": "customer_archive",
"name": "张三",
"contact_name": "李四",
"姓名": "王五",
"profile.name": "赵六",
}
)
assert masked == {
"table_name": "customer_profile",
"chinese_name": "zh_CN",
"english_name": "en_US",
"product_name": "智能助手",
"metadata.table_name": "customer_archive",
"name": "[NAME]",
"contact_name": "[NAME]",
"姓名": "[NAME]",
"profile.name": "[NAME]",
}
assert counts == {
"email": 0,
"phone": 0,
"id_card": 0,
"name": 4,
"total": 4,
}
def test_quality_scoring_covers_all_dimensions_and_duplicates() -> None: def test_quality_scoring_covers_all_dimensions_and_duplicates() -> None:
valid = { valid = {
"instruction": "如何修改收货地址?", "instruction": "如何修改收货地址?",

View File

@@ -1,5 +1,6 @@
from __future__ import annotations from __future__ import annotations
import json
from copy import deepcopy from copy import deepcopy
from io import BytesIO from io import BytesIO
from pathlib import Path from pathlib import Path
@@ -21,7 +22,12 @@ from app.modules.data_process.storage import (
LocalDataProcessStorage, LocalDataProcessStorage,
get_data_process_storage, get_data_process_storage,
) )
from app.modules.data_process.store import InvalidStateError, NotFoundError, get_data_process_store from app.modules.data_process.store import (
InvalidStateError,
NotFoundError,
get_data_process_store,
repeat_task_id,
)
class FakeDataProcessStore: class FakeDataProcessStore:
@@ -35,6 +41,7 @@ class FakeDataProcessStore:
self.datasets: dict[str, dict[str, Any]] = {} self.datasets: dict[str, dict[str, Any]] = {}
self.models: dict[str, dict[str, Any]] = {} self.models: dict[str, dict[str, Any]] = {}
self.regeneration_prepared: set[str] = set() self.regeneration_prepared: set[str] = set()
self.repeat_requests: dict[tuple[str, str], str] = {}
self.sequence = 0 self.sequence = 0
def _id(self, prefix: str) -> str: def _id(self, prefix: str) -> str:
@@ -150,6 +157,121 @@ class FakeDataProcessStore:
"published_outputs_preserved": published_outputs_preserved, "published_outputs_preserved": published_outputs_preserved,
} }
def _repeat_response(
self,
source_task_id: str,
repeated_task_id: str,
*,
created: bool,
) -> dict[str, Any]:
task = self.get_task(repeated_task_id)
task["source_file_count"] = len(self.sources[repeated_task_id])
task["preview_count"] = len(self.previews[repeated_task_id])
return {
"task": task,
"source_task_id": source_task_id,
"created": created,
"copied_source_file_count": len(self.sources[repeated_task_id]),
"copied_preview_count": len(self.previews[repeated_task_id]),
}
def find_repeated_task(
self,
source_task_id: str,
request_id: str,
) -> dict[str, Any] | None:
repeated_task_id = self.repeat_requests.get((source_task_id, request_id))
if repeated_task_id is None:
return None
return self._repeat_response(
source_task_id,
repeated_task_id,
created=False,
)
def repeat_task(
self,
source_task_id: str,
*,
expected_updated_at: str,
request_id: str,
file_copies: dict[str, dict[str, str]],
) -> dict[str, Any]:
existing = self.find_repeated_task(source_task_id, request_id)
if existing is not None:
return existing
source_task = self.get_task(source_task_id)
if source_task["status"] != "completed" or source_task.get("results_confirmed") is False:
raise InvalidStateError("只有已完成并确认结果的任务可以再次生成")
if source_task.get("updated_at") != expected_updated_at:
raise InvalidStateError("源任务已被其他操作修改,请刷新后重试")
source_files = self.sources[source_task_id]
if set(file_copies) != {str(item["id"]) for item in source_files}:
raise InvalidStateError("源文件快照已变化,请刷新后重试")
if not self.previews[source_task_id]:
raise InvalidStateError("源任务没有可用于再次生成的切分结果")
repeated_task_id = repeat_task_id(source_task_id, request_id)
suffix = f"(再次生成-{repeated_task_id[-6:]}"
task = {
**deepcopy(source_task),
"id": repeated_task_id,
"name": f"{source_task['name'][: max(1, 150 - len(suffix))]}{suffix}",
"status": "pending",
"progress": 20,
"output_dataset_id": None,
"output_datasets": [],
"output_count": 0,
"filtered_count": 0,
"duplicate_count": 0,
"error_count": 0,
"failure_reason": None,
"generation_run_id": None,
"results_confirmed": False,
"workflow_step": "preview",
"preview_status": "completed",
"preview_progress": 100,
"preview_run_id": None,
"preview_failure_reason": None,
"preview_total_files": len(source_files),
"preview_completed_files": len(source_files),
"started_at": None,
"completed_at": None,
}
self.tasks[repeated_task_id] = task
self.sources[repeated_task_id] = []
file_id_map: dict[str, str] = {}
for source in source_files:
old_file_id = str(source["id"])
copy = file_copies[old_file_id]
file_id_map[old_file_id] = copy["id"]
self.sources[repeated_task_id].append(
{
**deepcopy(source),
"id": copy["id"],
"task_id": repeated_task_id,
"storage_object_id": copy["storage_object_id"],
}
)
self.previews[repeated_task_id] = [
{
**deepcopy(item),
"id": self._id("dpp"),
"task_id": repeated_task_id,
"source_file_id": file_id_map.get(str(item.get("source_file_id")))
if item.get("source_file_id")
else None,
}
for item in self.previews[source_task_id]
]
self.results[repeated_task_id] = []
self.repeat_requests[(source_task_id, request_id)] = repeated_task_id
return self._repeat_response(
source_task_id,
repeated_task_id,
created=True,
)
def delete_task(self, task_id: str, **_: Any) -> None: def delete_task(self, task_id: str, **_: Any) -> None:
self.get_task(task_id) self.get_task(task_id)
del self.tasks[task_id] del self.tasks[task_id]
@@ -899,6 +1021,15 @@ def test_data_process_full_contract_without_database(tmp_path: Path) -> None:
listed_preview = client.get(f"/modelTF/data-process/{task_id}/preview") listed_preview = client.get(f"/modelTF/data-process/{task_id}/preview")
assert listed_preview.json()["data"]["total"] == 2 assert listed_preview.json()["data"]["total"] == 2
preview_item = listed_preview.json()["data"]["items"][0] preview_item = listed_preview.json()["data"]["items"][0]
source_locator = preview_item["quality_score"]["source_locator"]
assert source_locator == {
"kind": "jsonl",
"record_index": 1,
"start_line": 1,
"end_line": 1,
"source_start": 0,
"source_end": len(preview_item["original_content"]),
}
updated_preview = client.put( updated_preview = client.put(
f"/modelTF/data-process/{task_id}/preview/{preview_item['id']}", f"/modelTF/data-process/{task_id}/preview/{preview_item['id']}",
json={ json={
@@ -907,6 +1038,7 @@ def test_data_process_full_contract_without_database(tmp_path: Path) -> None:
}, },
) )
assert "quality_score" in updated_preview.json()["data"] assert "quality_score" in updated_preview.json()["data"]
assert updated_preview.json()["data"]["quality_score"]["source_locator"] == source_locator
generated = client.post(f"/modelTF/data-process/{task_id}/generate") generated = client.post(f"/modelTF/data-process/{task_id}/generate")
assert generated.status_code == 200 assert generated.status_code == 200
@@ -1760,6 +1892,118 @@ def test_regenerate_endpoint_prepares_an_existing_published_task(tmp_path: Path)
assert [item["id"] for item in detail["output_datasets"]] == ["dataset_train"] assert [item["id"] for item in detail["output_datasets"]] == ["dataset_train"]
def test_completed_task_can_repeat_into_an_independent_background_task(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
client, store, storage = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={
"name": "原始生成任务",
"process_type": "structured",
"config": {"qa_pairs_per_row": 1, "temperature": 0.3},
},
).json()["data"]["id"]
uploaded = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files={"files": ("source.jsonl", b'{"name":"alpha"}\n', "application/jsonl")},
)
assert uploaded.status_code == 200
source_id = uploaded.json()["data"]["files"][0]["id"]
built = client.post(
f"/modelTF/data-process/{task_id}/preview/build",
json={"replace_existing": True},
)
assert built.status_code == 200
store.tasks[task_id].update(
status="completed",
progress=100,
results_confirmed=True,
workflow_step="results",
output_count=1,
output_dataset_id="dataset-original",
updated_at="2026-07-28T12:00:00Z",
)
store.results[task_id] = [{"id": "result-original", "output": "原结果"}]
store.datasets["dataset-original"] = {
"id": "dataset-original",
"name": "原数据集",
"type": "train",
"source_task_id": task_id,
"deleted_at": None,
}
original_task = deepcopy(store.tasks[task_id])
original_sources = deepcopy(store.sources[task_id])
original_previews = deepcopy(store.previews[task_id])
original_results = deepcopy(store.results[task_id])
original_datasets = deepcopy(store.datasets)
monkeypatch.setattr(data_process_endpoint, "_run_generation", lambda *_: None)
payload = {
"expected_updated_at": "2026-07-28T12:00:00Z",
"request_id": "repeat-request-0001",
}
response = client.post(f"/modelTF/data-process/{task_id}/repeat", json=payload)
assert response.status_code == 202
repeated = response.json()["data"]
repeated_task_id = repeated["task"]["id"]
assert repeated["created"] is True
assert repeated_task_id != task_id
assert repeated["task"]["status"] == "running"
assert repeated["task"]["workflow_step"] == "generate"
assert repeated["copied_source_file_count"] == 1
assert repeated["copied_preview_count"] == len(original_previews)
assert store.tasks[task_id] == original_task
assert store.sources[task_id] == original_sources
assert store.previews[task_id] == original_previews
assert store.results[task_id] == original_results
assert store.datasets == original_datasets
repeated_source = store.sources[repeated_task_id][0]
repeated_preview = store.previews[repeated_task_id][0]
assert repeated_source["id"] != source_id
assert repeated_source["storage_object_id"] != original_sources[0]["storage_object_id"]
assert repeated_preview["id"] != original_previews[0]["id"]
assert repeated_preview["source_file_id"] == repeated_source["id"]
assert storage.read(repeated_source["storage_object_id"]) == b'{"name":"alpha"}\n'
replay = client.post(f"/modelTF/data-process/{task_id}/repeat", json=payload)
assert replay.status_code == 202
assert replay.json()["data"]["created"] is False
assert replay.json()["data"]["task"]["id"] == repeated_task_id
assert len(store.tasks) == 2
assert len(store.sources[repeated_task_id]) == 1
def test_repeat_rejects_a_stale_source_snapshot_without_creating_a_task(
tmp_path: Path,
) -> None:
client, store, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={"name": "源任务", "process_type": "structured", "config": {}},
).json()["data"]["id"]
store.tasks[task_id].update(
status="completed",
results_confirmed=True,
updated_at="2026-07-28T12:00:00Z",
)
before = deepcopy(store.tasks)
response = client.post(
f"/modelTF/data-process/{task_id}/repeat",
json={
"expected_updated_at": "2026-07-28T11:59:59Z",
"request_id": "repeat-request-stale",
},
)
assert response.status_code == 409
assert store.tasks == before
def test_published_split_datasets_remain_in_detail_after_regeneration( def test_published_split_datasets_remain_in_detail_after_regeneration(
tmp_path: Path, tmp_path: Path,
monkeypatch: pytest.MonkeyPatch, monkeypatch: pytest.MonkeyPatch,
@@ -2112,6 +2356,91 @@ def test_preprocess_deduplicates_and_quality_filter_removes_short_results(
assert client.get(f"/modelTF/data-process/{task_id}/results").json()["data"]["total"] == 0 assert client.get(f"/modelTF/data-process/{task_id}/results").json()["data"]["total"] == 0
def test_structured_deduplication_preserves_distinct_rows_after_desensitization(
tmp_path: Path,
) -> None:
client, _, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={
"name": "先去重再脱敏",
"process_type": "structured",
"config": {"preprocess_options": ["deduplicate", "desensitize"]},
},
).json()["data"]["id"]
uploaded = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files={
"files": (
"names.jsonl",
(
'{"name":"张三","role":"开发"}\n'
'{"name":"李四","role":"开发"}\n'
),
"application/jsonl",
)
},
)
assert uploaded.status_code == 200
preview = client.post(f"/modelTF/data-process/{task_id}/preview/build")
assert preview.status_code == 200
items = preview.json()["data"]["items"]
assert len(items) == 2
assert len({item["original_content"] for item in items}) == 2
assert {item["edited_content"] for item in items} == {
'{"name":"[NAME]","role":"开发"}'
}
def test_structured_deduplication_removes_identical_rows_across_sources(
tmp_path: Path,
) -> None:
client, _, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={
"name": "跨源原文去重",
"process_type": "structured",
"config": {"preprocess_options": ["deduplicate", "desensitize"]},
},
).json()["data"]["id"]
uploaded = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files=[
(
"files",
(
"first.jsonl",
'{"name":"张三","role":"开发"}\n',
"application/jsonl",
),
),
(
"files",
(
"second.jsonl",
'\n{"name":"张三","role":"开发"}\n',
"application/jsonl",
),
),
],
)
assert uploaded.status_code == 200
first_source, second_source = uploaded.json()["data"]["files"]
preview = client.post(f"/modelTF/data-process/{task_id}/preview/build")
assert preview.status_code == 200
data = preview.json()["data"]
assert data["total"] == 1
assert data["file_counts"] == {
first_source["id"]: 1,
second_source["id"]: 0,
}
def test_stale_generation_worker_cannot_overwrite_new_run(monkeypatch: Any) -> None: def test_stale_generation_worker_cannot_overwrite_new_run(monkeypatch: Any) -> None:
store = FakeDataProcessStore() store = FakeDataProcessStore()
task = store.create_task( task = store.create_task(
@@ -2364,7 +2693,26 @@ def test_xlsx_upload_is_accepted_as_structured_records(tmp_path: Path) -> None:
) )
preview = client.post(f"/modelTF/data-process/{task_id}/preview/build") preview = client.post(f"/modelTF/data-process/{task_id}/preview/build")
assert preview.status_code == 200 assert preview.status_code == 200
assert preview.json()["data"]["total"] == 2 preview_items = preview.json()["data"]["items"]
assert len(preview_items) == 2
assert [item["quality_score"]["source_locator"] for item in preview_items] == [
{
"kind": "xlsx",
"record_index": 1,
"sheet_index": 0,
"sheet_name": "Sheet",
"row_number": 2,
"sheet_record_index": 0,
},
{
"kind": "xlsx",
"record_index": 2,
"sheet_index": 0,
"sheet_name": "Sheet",
"row_number": 3,
"sheet_record_index": 1,
},
]
def test_docx_preview_preserves_document_block_order_and_source_offsets( def test_docx_preview_preserves_document_block_order_and_source_offsets(
@@ -2757,6 +3105,218 @@ def _preview_task(
) )
def _structured_preview_task(
content: str,
*,
file_format: str,
options: list[str] | None = None,
) -> list[dict[str, Any]]:
return data_process_endpoint._build_preview_items(
{
"process_type": "structured",
"config": {"preprocess_options": options or []},
},
[
{
"id": "structured-source",
"name": f"records.{file_format}",
"file_format": file_format,
"content": content,
}
],
)
def test_structured_preview_exposes_json_jsonl_and_csv_source_locators() -> None:
json_source = '{"records":[{"id":1},{"id":2}]}'
json_items = _structured_preview_task(
json_source,
file_format="json",
)
assert [
item["quality_score"]["source_locator"]["json_pointer"]
for item in json_items
] == ["/records/0", "/records/1"]
assert [
json_source[item["source_start"] : item["source_end"]]
for item in json_items
] == ['{"id":1}', '{"id":2}']
assert [item["source_start_line"] for item in json_items] == [1, 1]
jsonl_source = '{"id":1}\n\n{"id":2}'
jsonl_items = _structured_preview_task(jsonl_source, file_format="jsonl")
assert [
item["quality_score"]["source_locator"]["record_index"]
for item in jsonl_items
] == [1, 2]
assert [item["source_start_line"] for item in jsonl_items] == [1, 3]
assert [
jsonl_source[item["source_start"] : item["source_end"]]
for item in jsonl_items
] == ['{"id":1}', '{"id":2}']
csv_source = 'id,note\n1,"hello\nworld"\n\n2,plain'
csv_items = _structured_preview_task(csv_source, file_format="csv")
assert [
(item["source_start_line"], item["source_end_line"])
for item in csv_items
] == [(2, 3), (5, 5)]
assert [
csv_source[item["source_start"] : item["source_end"]]
for item in csv_items
] == ['1,"hello\nworld"', "2,plain"]
def test_structured_empty_json_upload_and_preview_remain_empty(tmp_path: Path) -> None:
client, _, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={"name": "空 JSON", "process_type": "structured", "config": {}},
).json()["data"]["id"]
uploaded = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files=[
("files", ("empty-array.json", "[]", "application/json")),
(
"files",
("empty-wrapper.json", '{"records":[],"total":0}', "application/json"),
),
],
)
assert uploaded.status_code == 200
assert [item["record_count"] for item in uploaded.json()["data"]["files"]] == [0, 0]
preview = client.post(f"/modelTF/data-process/{task_id}/preview/build")
assert preview.status_code == 200
assert preview.json()["data"]["items"] == []
assert preview.json()["data"]["total"] == 0
assert set(preview.json()["data"]["file_counts"].values()) == {0}
def test_structured_json_upload_rejects_ambiguous_or_invalid_numbers(
tmp_path: Path,
) -> None:
client, _, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={"name": "严格 JSON", "process_type": "structured", "config": {}},
).json()["data"]["id"]
invalid_sources = (
("duplicate.json", '{"id":1,"id":2}'),
("duplicate.jsonl", '{"id":1,"id":2}\n'),
("nan.json", '{"value":NaN}'),
("infinity.json", '{"value":Infinity}'),
("control.json", '{"value":"bad\x00control"}'),
("deep.json", "[" * 10_000 + "0" + "]" * 10_000),
)
for filename, content in invalid_sources:
response = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files={"files": (filename, content, "application/json")},
)
assert response.status_code == 400, (filename, response.text)
def test_structured_json_preview_preserves_precision_and_business_data_field(
tmp_path: Path,
) -> None:
client, _, _ = make_client(tmp_path)
task_id = client.post(
"/modelTF/data-process",
json={"name": "无损 JSON", "process_type": "structured", "config": {}},
).json()["data"]["id"]
precise = '{"value":0.123456789012345678901234567890}'
business = '{"id":7,"data":[{"id":8}]}'
uploaded = client.post(
f"/modelTF/data-process/{task_id}/source-files",
files=[
("files", ("precise.json", precise, "application/json")),
("files", ("business.json", business, "application/json")),
],
)
assert uploaded.status_code == 200
assert [item["record_count"] for item in uploaded.json()["data"]["files"]] == [1, 1]
preview = client.post(f"/modelTF/data-process/{task_id}/preview/build")
assert preview.status_code == 200
items = preview.json()["data"]["items"]
assert [item["original_content"] for item in items] == [precise, business]
assert [
item["quality_score"]["source_locator"]["json_pointer"] for item in items
] == ["", ""]
assert [item["source_start"] for item in items] == [0, 0]
def test_structured_preview_lineage_survives_clean_deduplicate_and_filter() -> None:
source_records = [
{"id": "A", "amount": 10, "empty": ""},
{"id": "A", "amount": 10, "empty": ""},
{"id": "", "amount": 11, "empty": ""},
{"id": "B", "amount": 11, "empty": ""},
{"id": "C", "amount": 12, "empty": ""},
{"id": "D", "amount": 12, "empty": ""},
{"id": "E", "amount": 13, "empty": ""},
{"id": "F", "amount": 13, "empty": ""},
{"id": "G", "amount": 14, "empty": ""},
{"id": "H", "amount": 1000, "empty": ""},
]
source = "\n".join(
json.dumps(record, ensure_ascii=False, separators=(",", ":"))
for record in source_records
)
items = _structured_preview_task(
source,
file_format="jsonl",
options=["clean_invalid", "deduplicate", "filter_anomaly"],
)
assert [
item["quality_score"]["source_locator"]["record_index"]
for item in items
] == [1, 3, 4, 5, 6, 7, 8, 9]
assert [item["source_start_line"] for item in items] == [1, 3, 4, 5, 6, 7, 8, 9]
assert [json.loads(item["original_content"])["id"] for item in items] == [
"A",
"",
"B",
"C",
"D",
"E",
"F",
"G",
]
def test_structured_preview_deduplicates_exact_rows_not_matching_identifiers() -> None:
source_records = [
{"customer_id": "C-1", "status": "old"},
{"customer_id": "C-1", "status": "new"},
{"status": "old", "customer_id": "C-1"},
]
source = "\n".join(
json.dumps(record, ensure_ascii=False, separators=(",", ":"))
for record in source_records
)
items = _structured_preview_task(
source,
file_format="jsonl",
options=["clean_invalid", "deduplicate"],
)
assert [
item["quality_score"]["source_locator"]["record_index"]
for item in items
] == [1, 2]
assert [item["source_start_line"] for item in items] == [1, 2]
assert [json.loads(item["original_content"])["status"] for item in items] == [
"old",
"new",
]
def test_fixed_preview_preserves_source_offsets() -> None: def test_fixed_preview_preserves_source_offsets() -> None:
content = ( content = (
"# 第一章\n" "# 第一章\n"

View File

@@ -63,6 +63,40 @@ def test_stage_publish_read_delete_roundtrip_with_unicode_filename(tmp_path: Pat
_assert_staging_empty(storage) _assert_staging_empty(storage)
def test_stage_copy_creates_an_independently_deletable_source_object(
tmp_path: Path,
) -> None:
storage = LocalDataProcessStorage(tmp_path / "storage")
original = _stage(storage, content=b"immutable source")
storage.publish([original])
copied = storage.stage_copy(
batch_id="batch-copy",
source_reference=original.reference,
expected_source_task_id="task-1",
expected_source_file_id="source-1",
task_id="task-2",
source_file_id="source-2",
version=1,
name="source.txt",
)
storage.publish([copied])
assert storage.read(copied.reference) == b"immutable source"
assert storage.delete(
original.reference,
expected_task_id="task-1",
expected_source_file_id="source-1",
) is True
assert storage.read(copied.reference) == b"immutable source"
assert storage.delete(
copied.reference,
expected_task_id="task-2",
expected_source_file_id="source-2",
) is True
_assert_staging_empty(storage)
def test_db_reference_is_left_to_database_storage(tmp_path: Path) -> None: def test_db_reference_is_left_to_database_storage(tmp_path: Path) -> None:
storage = LocalDataProcessStorage(tmp_path / "storage") storage = LocalDataProcessStorage(tmp_path / "storage")

View File

@@ -18,6 +18,7 @@ from app.modules.data_process.store import (
_preview_config_changed, _preview_config_changed,
_reasoning_output_is_valid, _reasoning_output_is_valid,
_source_storage_descriptor, _source_storage_descriptor,
repeat_task_id,
) )
@@ -331,6 +332,140 @@ class _TaskDetailStore(DataProcessStore):
yield self._conn yield self._conn
class _RepeatConnection:
def __init__(self) -> None:
self.source_files = [
{
"id": "source-old",
"name": "source.jsonl",
"size_bytes": 12,
"record_count": 1,
"file_format": "jsonl",
"checksum_sha256": "a" * 64,
"content": '{"id":1}\n',
"content_preview": '{"id":1}',
"metadata": {"storage_backend": "local"},
"created_by": "user-1",
}
]
self.source_previews = [
{
"id": "preview-old",
"source_file_id": "source-old",
"original_content": '{"id":1}',
"edited_content": '{"id":1,"checked":true}',
"source_start": 0,
"source_end": 8,
"source_start_line": 1,
"source_end_line": 1,
"token_count": 5,
"status": "modified",
"quality_score": {"overall": 90},
}
]
self.created_task: dict[str, Any] | None = None
self.created_files: list[dict[str, Any]] = []
self.created_previews: list[dict[str, Any]] = []
def execute(self, sql: str, params: Any = None) -> _Result:
normalized = " ".join(sql.split())
if params is not None:
assert normalized.count("%s") == len(params)
if normalized.startswith("SELECT * FROM data_process_tasks WHERE id="):
return _Result(row=None)
if normalized.startswith("SELECT * FROM data_process_source_files"):
return _Result(rows=[dict(item) for item in self.source_files])
if normalized.startswith("SELECT * FROM data_process_preview_items"):
return _Result(rows=[dict(item) for item in self.source_previews])
if normalized.startswith("INSERT INTO data_process_tasks"):
self.created_task = {
"id": params[0],
"name": params[1],
"description": params[2],
"status": "pending",
"process_type": params[3],
"source_dataset_id": params[4],
"config": params[5],
"progress": 20,
"input_count": params[6],
"results_confirmed": False,
"workflow_step": "preview",
"preview_status": "completed",
"preview_progress": 100,
"preview_total_files": params[7],
"preview_completed_files": params[8],
"created_at": params[15],
"updated_at": params[16],
}
return _Result(row=dict(self.created_task))
if normalized.startswith("INSERT INTO data_process_source_files"):
self.created_files.append(
{
"id": params[0],
"task_id": params[1],
"storage_object_id": params[2],
"content": params[8],
}
)
return _Result()
if normalized.startswith("INSERT INTO data_process_preview_items"):
self.created_previews.append(
{
"id": params[0],
"task_id": params[1],
"source_file_id": params[2],
"edited_content": params[4],
}
)
return _Result()
if normalized.startswith("SELECT (SELECT COUNT(*) FROM data_process_source_files"):
return _Result(
row={
"source_file_count": len(self.created_files),
"preview_count": len(self.created_previews),
}
)
raise AssertionError(f"unexpected SQL: {normalized}")
class _RepeatStore(DataProcessStore):
def __init__(self, conn: _RepeatConnection) -> None:
self._conn = conn
@contextmanager
def connect(self) -> Iterator[_RepeatConnection]:
yield self._conn
def _task_in_connection(
self,
conn: Any,
task_id: str,
*,
for_update: bool = False,
) -> dict[str, Any]:
assert task_id == "task-source"
assert for_update is True
return {
"id": task_id,
"name": "原任务",
"description": "原描述",
"status": "completed",
"process_type": "structured",
"source_dataset_id": None,
"config": {
"temperature": 0.3,
"_regeneration_prepared": {"prepared": True},
},
"results_confirmed": True,
"preview_status": "completed",
"tenant_id": "tenant-1",
"project_id": "project-1",
"owner_id": "owner-1",
"created_by": "user-1",
"updated_at": "2026-07-28T12:00:00Z",
}
class _TaskListConnection: class _TaskListConnection:
def __init__(self) -> None: def __init__(self) -> None:
self.task = { self.task = {
@@ -574,6 +709,47 @@ def test_decode_row_serializes_postgres_numeric_values_as_json_numbers() -> None
assert decoded == {"progress": 100.0, "duration_seconds": 389.0} assert decoded == {"progress": 100.0, "duration_seconds": 389.0}
def test_repeat_task_copies_business_snapshot_with_new_resource_ids() -> None:
conn = _RepeatConnection()
store = _RepeatStore(conn)
request_id = "repeat-request-0001"
target_task_id = repeat_task_id("task-source", request_id)
repeated = store.repeat_task(
"task-source",
expected_updated_at="2026-07-28T12:00:00Z",
request_id=request_id,
file_copies={
"source-old": {
"id": "source-new",
"storage_object_id": (
f"local://data-process/{target_task_id}/source-new/v1/source.jsonl"
),
}
},
)
assert repeated["created"] is True
assert repeated["task"]["id"] == target_task_id
assert repeated["task"]["config"] == {"temperature": 0.3}
assert repeated["task"]["results_confirmed"] is False
assert repeated["copied_source_file_count"] == 1
assert repeated["copied_preview_count"] == 1
assert conn.created_files == [
{
"id": "source-new",
"task_id": target_task_id,
"storage_object_id": (
f"local://data-process/{target_task_id}/source-new/v1/source.jsonl"
),
"content": '{"id":1}\n',
}
]
assert conn.created_previews[0]["task_id"] == target_task_id
assert conn.created_previews[0]["source_file_id"] == "source-new"
assert conn.created_previews[0]["edited_content"] == '{"id":1,"checked":true}'
def test_decode_row_decodes_aggregated_output_datasets_json() -> None: def test_decode_row_decodes_aggregated_output_datasets_json() -> None:
decoded = _decode_row( decoded = _decode_row(
{ {

View File

@@ -0,0 +1,774 @@
"""
平台治理功能集成测试 —— 覆盖第 1-4 周交付内容。
测试策略:
- 在导入 app 模块前 mock psycopg / psycopg_pool避免依赖真实数据库驱动
- 使用 FastAPI TestClient 对真实路由栈发起请求
- 通过 mock.get_platform_store 替换为内存 FakeStore
- 每周交付内容对应一组 test class方便分阶段验收
覆盖范围:
第 1 周 — 登录、当前用户、用户列表、权限码、日志查询
第 2 周 — 租户、项目、项目成员、资源 ACL
第 3 周 — 审批实例、审批模板、审计日志查询和导出
第 4 周 — 写操作审计、审批拦截、权限校验
"""
from __future__ import annotations
import json
import sys
import types
from contextlib import contextmanager
from typing import Any, Iterator
from unittest.mock import MagicMock, patch
import pytest
from fastapi import FastAPI
from fastapi.testclient import TestClient
# ============================================================
# 在导入 app 之前 mock psycopg / psycopg_pool
# ============================================================
_psycopg_mock = types.ModuleType("psycopg")
_psycopg_mock.PgConn = type("PgConn", (), {})
_psycopg_mock.PostgresConnectionPool = MagicMock()
_psycopg_mock.connection = MagicMock()
sys.modules.setdefault("psycopg", _psycopg_mock)
_psycopg_pool_mock = types.ModuleType("psycopg_pool")
_psycopg_pool_mock.ConnectionPool = MagicMock()
sys.modules.setdefault("psycopg_pool", _psycopg_pool_mock)
# 现在安全导入 app 模块
from app.api.v1.endpoints.platform import ok, fail # noqa: E402
from app.modules.tenant.router import router as tenant_router # noqa: E402
from app.modules.project.router import router as project_router # noqa: E402
from app.modules.approval.router import router as approval_router # noqa: E402
from app.modules.system.router import router as system_router # noqa: E402
from app.modules.retention.router import router as retention_router # noqa: E402
from app.modules.resource.router import router as resource_router # noqa: E402
from app.api.v1.endpoints.platform import router as platform_router # noqa: E402
PREFIX = "/modelTF"
ADMIN_TOKEN = "platform-token-u_admin"
OP_TOKEN = "platform-token-u_op"
# ============================================================
# FakePlatformStore —— 内存实现,模拟 PlatformStore 全部治理接口
# ============================================================
class FakePlatformStore:
"""平台治理测试专用内存 store确保测试不连接真实数据库。"""
def __init__(self) -> None:
self._users: list[dict[str, Any]] = [
{
"id": "u_admin",
"username": "admin",
"display_name": "Admin",
"role": "admin",
"status": "active",
"permissions": [
"dashboard", "fine-tune", "model-eval", "model-inference",
"model-manage", "dataset", "data-process", "data-convert",
"compute", "hardware", "logs", "user-settings",
],
"last_login": "2026-08-01T10:00:00Z",
"protected": True,
},
{
"id": "u_op",
"username": "operator",
"display_name": "Operator",
"role": "operator",
"status": "active",
"permissions": ["dashboard", "fine-tune"],
"last_login": "2026-08-01T11:00:00Z",
"protected": False,
},
]
self._tenants: dict[str, dict[str, Any]] = {}
self._projects: dict[str, dict[str, Any]] = {}
self._members: dict[str, list[dict[str, Any]]] = {}
self._acl: dict[str, list[dict[str, Any]]] = {}
self._audit_logs: list[dict[str, Any]] = []
self._approval_templates: dict[str, dict[str, Any]] = {}
self._approval_instances: dict[str, dict[str, Any]] = {}
self._retention_policies: dict[str, dict[str, Any]] = {}
self._models: list[dict[str, Any]] = []
self._datasets: list[dict[str, Any]] = []
self._tasks: list[dict[str, Any]] = []
self._compute_nodes: list[dict[str, Any]] = []
self._gpus: list[dict[str, Any]] = []
self._sessions: list[dict[str, Any]] = []
self._seq = 0
@contextmanager
def connect(self) -> Iterator[Any]:
class FakeConn:
def execute(self, *a, **kw):
return []
def commit(self):
pass
def rollback(self):
pass
def close(self):
pass
yield FakeConn()
# ---- helpers ----
def _next_id(self, prefix: str) -> str:
self._seq += 1
return f"{prefix}_{self._seq}"
# ==================== 第1周登录 / 用户 / 权限码 / 日志 ====================
def login(self, username: str, password: str) -> dict[str, Any] | None:
for u in self._users:
if u["username"] == username and u["status"] == "active":
if password in ("admin123", "operator123", "test123"):
return dict(u)
return None
def users(self) -> list[dict[str, Any]]:
return [dict(u) for u in self._users]
def create_user(self, payload: dict[str, Any]) -> dict[str, Any]:
u = {"id": self._next_id("u"), "protected": False, **payload}
self._users.append(u)
return u
def update_user(self, user_id: str, payload: dict[str, Any]) -> dict[str, Any]:
for u in self._users:
if u["id"] == user_id:
u.update(payload)
return u
raise KeyError(user_id)
def delete_user(self, user_id: str) -> None:
self._users = [u for u in self._users if u["id"] != user_id]
def roles(self) -> list[dict[str, Any]]:
return [
{"name": "admin", "display_name": "管理员"},
{"name": "operator", "display_name": "操作员"},
{"name": "viewer", "display_name": "访客"},
]
def log_files(self, date: str | None = None) -> list[dict[str, Any]]:
return [{"name": "backend-2026-08-01.log", "size": "1 KB", "date": "2026-08-01"}]
def log_content(self, file: str) -> dict[str, Any]:
return {"file": file, "content": "[INFO] test line", "size": "1 KB"}
def training_log_files(self) -> list[dict[str, Any]]:
return [{"task_id": "ft_001", "name": "ft_001.log", "size": "2 KB"}]
def training_log_content(self, file: str) -> dict[str, Any]:
return {"file": file, "content": "epoch 0 loss 1.0", "size": "2 KB"}
# ==================== 第2周租户 / 项目 / 成员 / ACL ====================
def tenants(self) -> list[dict[str, Any]]:
return list(self._tenants.values())
def tenant(self, tenant_id: str) -> dict[str, Any]:
if tenant_id not in self._tenants:
raise KeyError(tenant_id)
return dict(self._tenants[tenant_id])
def create_tenant(self, payload: dict[str, Any]) -> dict[str, Any]:
tid = self._next_id("tnt")
t = {"id": tid, "status": "active", "quota": "{}", "retention_policy_id": None,
"create_time": "2026-08-01T00:00:00Z", **payload}
self._tenants[tid] = t
return dict(t)
def update_tenant(self, tenant_id: str, payload: dict[str, Any]) -> dict[str, Any]:
self._tenants[tenant_id].update(payload)
return dict(self._tenants[tenant_id])
def set_tenant_quota(self, tenant_id: str, quota: dict[str, Any]) -> dict[str, Any]:
self._tenants[tenant_id]["quota"] = json.dumps(quota)
return dict(self._tenants[tenant_id])
def set_tenant_retention(self, tenant_id: str, retention_policy_id: str | None) -> dict[str, Any]:
self._tenants[tenant_id]["retention_policy_id"] = retention_policy_id
return dict(self._tenants[tenant_id])
def projects(self, *, tenant_id: str = "default", status: str | None = None, keyword: str | None = None) -> list[dict[str, Any]]:
result = []
for p in self._projects.values():
if p.get("tenant_id") != tenant_id:
continue
if status and p.get("status") != status:
continue
if keyword and keyword.lower() not in p.get("name", "").lower():
continue
result.append(dict(p))
return result
def project(self, project_id: str) -> dict[str, Any]:
if project_id not in self._projects:
raise KeyError(project_id)
return dict(self._projects[project_id])
def create_project(self, payload: dict[str, Any]) -> dict[str, Any]:
pid = self._next_id("prj")
p = {"id": pid, "status": "active", "quota": "{}", "create_time": "2026-08-01T00:00:00Z", **payload}
self._projects[pid] = p
self._members[pid] = []
return dict(p)
def update_project(self, project_id: str, payload: dict[str, Any]) -> dict[str, Any]:
self._projects[project_id].update(payload)
return dict(self._projects[project_id])
def archive_project(self, project_id: str) -> dict[str, Any]:
self._projects[project_id]["status"] = "archived"
return dict(self._projects[project_id])
def delete_project(self, project_id: str) -> None:
self._projects.pop(project_id, None)
self._members.pop(project_id, None)
def project_members(self, project_id: str) -> list[dict[str, Any]]:
return [dict(m) for m in self._members.get(project_id, [])]
def add_project_member(self, project_id: str, payload: dict[str, Any]) -> dict[str, Any]:
m = {"joined_at": "2026-08-01T00:00:00Z", **payload}
self._members.setdefault(project_id, []).append(m)
return m
def update_project_member_role(self, project_id: str, user_id: str, role: str) -> dict[str, Any]:
for m in self._members.get(project_id, []):
if m["user_id"] == user_id:
m["role"] = role
return m
raise KeyError(user_id)
def remove_project_member(self, project_id: str, user_id: str) -> None:
self._members[project_id] = [m for m in self._members.get(project_id, []) if m["user_id"] != user_id]
# ---- ACL ----
def get_acl(self, resource_type: str, resource_id: str) -> list[dict[str, Any]]:
key = f"{resource_type}:{resource_id}"
return [dict(a) for a in self._acl.get(key, [])]
def set_acl(self, resource_type: str, resource_id: str, entries: list[dict[str, Any]]) -> list[dict[str, Any]]:
key = f"{resource_type}:{resource_id}"
self._acl[key] = [dict(e) for e in entries]
return self.get_acl(resource_type, resource_id)
def resource_acl(self, resource_type: str, resource_id: str) -> list[dict[str, Any]]:
rows = self.get_acl(resource_type, resource_id)
grouped: dict[str, dict[str, Any]] = {}
for r in rows:
k = f"{r.get('principal_type')}:{r.get('principal_id')}"
bucket = grouped.setdefault(k, {
"subject_type": r.get("principal_type"),
"subject_id": r.get("principal_id"),
"permissions": [],
})
perm = r.get("permission")
if perm and perm not in bucket["permissions"]:
bucket["permissions"].append(perm)
return list(grouped.values())
def set_resource_acl(self, resource_type: str, resource_id: str, entries: list[dict[str, Any]]) -> list[dict[str, Any]]:
flat: list[dict[str, Any]] = []
for e in entries:
for perm in e.get("permissions") or []:
flat.append({
"principal_type": e.get("subject_type"),
"principal_id": e.get("subject_id"),
"permission": perm,
})
self.set_acl(resource_type, resource_id, flat)
return self.resource_acl(resource_type, resource_id)
# ==================== 第3周审批 / 审计 / 留存 ====================
def approval_templates(self) -> list[dict[str, Any]]:
return list(self._approval_templates.values())
def create_approval_template(self, payload: dict[str, Any]) -> dict[str, Any]:
tid = payload.get("id") or self._next_id("tpl")
t = {"id": tid, "steps": [], "create_time": "2026-08-01T00:00:00Z", **payload}
self._approval_templates[tid] = t
return dict(t)
def approval_instances(self, *, status: str | None = None) -> list[dict[str, Any]]:
result = []
for i in self._approval_instances.values():
if status and i.get("status") != status:
continue
result.append(dict(i))
return result
def approval_instance(self, instance_id: str) -> dict[str, Any]:
if instance_id not in self._approval_instances:
raise KeyError(instance_id)
return dict(self._approval_instances[instance_id])
def create_approval_instance(self, payload: dict[str, Any]) -> dict[str, Any]:
iid = self._next_id("appr")
inst = {
"id": iid,
"status": "pending",
"current_step": 0,
"steps": [],
"create_time": "2026-08-01T00:00:00Z",
**payload,
}
self._approval_instances[iid] = inst
return dict(inst)
def decide_approval_step(self, instance_id: str, step_index: int, *, approver_id: str, approved: bool, comment: str | None = None) -> dict[str, Any]:
inst = self._approval_instances[instance_id]
inst["status"] = "approved" if approved else "rejected"
inst["current_step"] = step_index + 1
return dict(inst)
def audit_logs(self, **kw) -> dict[str, Any]:
items = [dict(l) for l in self._audit_logs]
for filter_key in ("tenant_id", "project_id", "actor_id", "action", "target_type"):
val = kw.get(filter_key)
if val:
items = [l for l in items if l.get(filter_key) == val]
limit = kw.get("limit", 50)
offset = kw.get("offset", 0)
total = len(items)
items = items[offset:offset + limit]
return {"items": items, "total": total}
def record_audit(self, **kw) -> None:
log = {"id": self._next_id("log"), "time": "2026-08-01T12:00:00Z", **kw}
self._audit_logs.append(log)
# ---- 留存策略 ----
def retention_policies(self) -> list[dict[str, Any]]:
return list(self._retention_policies.values())
def retention_policy(self, policy_id: str) -> dict[str, Any]:
if policy_id not in self._retention_policies:
raise KeyError(policy_id)
return dict(self._retention_policies[policy_id])
def create_retention_policy(self, payload: dict[str, Any]) -> dict[str, Any]:
pid = payload.get("id") or self._next_id("rpol")
p = {"id": pid, "status": "active", "create_time": "2026-08-01T00:00:00Z", **payload}
self._retention_policies[pid] = p
return dict(p)
def update_retention_policy(self, policy_id: str, payload: dict[str, Any]) -> dict[str, Any]:
self._retention_policies[policy_id].update(payload)
return dict(self._retention_policies[policy_id])
def delete_retention_policy(self, policy_id: str) -> None:
self._retention_policies.pop(policy_id, None)
# ---- dashboard & other stubs ----
def login_duration_rank(self, limit: int = 8, days: int = 30) -> list[dict[str, Any]]:
return [{"user": "admin", "role": "admin", "duration": 10.0}]
def models(self) -> list[dict[str, Any]]:
return self._models
def datasets(self) -> list[dict[str, Any]]:
return self._datasets
def tasks(self) -> list[dict[str, Any]]:
return self._tasks
def compute_nodes(self) -> list[dict[str, Any]]:
return self._compute_nodes
def gpus(self) -> list[dict[str, Any]]:
return self._gpus
def system_info(self) -> dict[str, Any]:
return {"cpu": {}, "memory": {}}
# ============================================================
# 测试 fixtures
# ============================================================
@pytest.fixture(scope="module")
def fake_store() -> FakePlatformStore:
return FakePlatformStore()
def _build_client(store: FakePlatformStore) -> TestClient:
"""构建 TestClientpatch 所有治理模块的 get_platform_store。"""
app = FastAPI()
app.include_router(platform_router, prefix=PREFIX)
app.include_router(system_router, prefix=PREFIX)
app.include_router(tenant_router, prefix=PREFIX)
app.include_router(project_router, prefix=PREFIX)
app.include_router(approval_router, prefix=PREFIX)
app.include_router(retention_router, prefix=PREFIX)
app.include_router(resource_router, prefix=PREFIX)
patches = [
patch("app.db.platform_store.get_platform_store", return_value=store),
patch("app.core.auth.get_platform_store", return_value=store),
patch("app.api.v1.endpoints.platform.get_platform_store", return_value=store),
patch("app.modules.system.router.get_platform_store", return_value=store),
patch("app.modules.tenant.router.get_platform_store", return_value=store),
patch("app.modules.project.router.get_platform_store", return_value=store),
patch("app.modules.approval.router.get_platform_store", return_value=store),
patch("app.modules.retention.router.get_platform_store", return_value=store),
patch("app.modules.resource.router.get_platform_store", return_value=store),
]
for p in patches:
p.start()
client = TestClient(app, raise_server_exceptions=False)
client._fake_store = store # type: ignore[attr-defined]
return client
@pytest.fixture(scope="module")
def client(fake_store: FakePlatformStore) -> TestClient:
c = _build_client(fake_store)
yield c
def _admin_headers() -> dict[str, str]:
return {"Authorization": f"Bearer {ADMIN_TOKEN}"}
def _op_headers() -> dict[str, str]:
return {"Authorization": f"Bearer {OP_TOKEN}"}
# ============================================================
# 第 1 周测试:登录、当前用户、用户列表、权限码、日志查询
# ============================================================
class TestWeek1AuthUserPermissionsLogs:
"""第 1 周:登录、当前用户、用户列表、权限码、日志查询接口。"""
def test_login_success(self, client: TestClient):
resp = client.post(f"{PREFIX}/login", json={"username": "admin", "password": "admin123"})
assert resp.status_code == 200
data = resp.json()["data"]
assert data["token"] == ADMIN_TOKEN
assert data["user"]["username"] == "admin"
def test_login_invalid(self, client: TestClient):
resp = client.post(f"{PREFIX}/login", json={"username": "admin", "password": "wrong"})
assert resp.status_code == 401
def test_me_with_valid_token(self, client: TestClient):
resp = client.get(f"{PREFIX}/me", headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["username"] == "admin"
def test_me_without_token(self, client: TestClient):
resp = client.get(f"{PREFIX}/me")
assert resp.status_code == 401
def test_users_list(self, client: TestClient):
resp = client.get(f"{PREFIX}/users", headers=_admin_headers())
assert resp.status_code == 200
users = resp.json()["data"]
assert len(users) >= 2
assert any(u["username"] == "admin" for u in users)
def test_create_user(self, client: TestClient):
resp = client.post(
f"{PREFIX}/users",
json={"username": "tester", "display_name": "Tester", "role": "viewer", "password": "test123"},
headers=_admin_headers(),
)
assert resp.status_code == 200
assert resp.json()["data"]["username"] == "tester"
def test_permission_codes(self, client: TestClient):
resp = client.get(f"{PREFIX}/system/permissions/codes")
assert resp.status_code == 200
codes = resp.json()["data"]["codes"]
assert "dashboard" in codes
assert "user-settings" in codes
def test_permissions_overview(self, client: TestClient):
resp = client.get(f"{PREFIX}/system/permissions")
assert resp.status_code == 200
data = resp.json()["data"]
assert "codes" in data
assert "roles" in data
def test_log_files(self, client: TestClient):
resp = client.get(f"{PREFIX}/log-files", headers=_admin_headers())
assert resp.status_code == 200
files = resp.json()["data"]
assert len(files) >= 1
def test_log_content(self, client: TestClient):
resp = client.get(f"{PREFIX}/log-content", params={"file": "backend.log"}, headers=_admin_headers())
assert resp.status_code == 200
assert "content" in resp.json()["data"]
def test_training_log_files(self, client: TestClient):
resp = client.get(f"{PREFIX}/training-log-files", headers=_admin_headers())
assert resp.status_code == 200
assert len(resp.json()["data"]) >= 1
def test_training_log_content(self, client: TestClient):
resp = client.get(f"{PREFIX}/training-log-content", params={"file": "ft_001.log"}, headers=_admin_headers())
assert resp.status_code == 200
assert "content" in resp.json()["data"]
# ============================================================
# 第 2 周测试:租户、项目、项目成员、资源 ACL
# ============================================================
class TestWeek2TenantProjectACL:
"""第 2 周:租户、项目、项目成员、资源 ACL。"""
def test_tenant_crud(self, client: TestClient):
# 创建
resp = client.post(f"{PREFIX}/tenants", json={"name": "Tenant-A", "code": "ta"}, headers=_admin_headers())
assert resp.status_code == 200
tid = resp.json()["data"]["id"]
# 查列表
resp = client.get(f"{PREFIX}/tenants", headers=_admin_headers())
assert resp.status_code == 200
assert any(t["id"] == tid for t in resp.json()["data"])
# 查详情
resp = client.get(f"{PREFIX}/tenants/{tid}", headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["name"] == "Tenant-A"
# 更新
resp = client.put(f"{PREFIX}/tenants/{tid}", json={"name": "Tenant-A2"}, headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["name"] == "Tenant-A2"
def test_tenant_quota(self, client: TestClient):
resp = client.post(f"{PREFIX}/tenants", json={"name": "Q-Tenant", "code": "qt"}, headers=_admin_headers())
tid = resp.json()["data"]["id"]
resp = client.put(f"{PREFIX}/tenants/{tid}/quota", json={"quota": {"gpu": 4}}, headers=_admin_headers())
assert resp.status_code == 200
def test_tenant_retention(self, client: TestClient):
resp = client.post(f"{PREFIX}/tenants", json={"name": "R-Tenant", "code": "rt"}, headers=_admin_headers())
tid = resp.json()["data"]["id"]
resp = client.put(f"{PREFIX}/tenants/{tid}/retention-policy", json={"retention_policy_id": "rpol_1"}, headers=_admin_headers())
assert resp.status_code == 200
def test_project_crud(self, client: TestClient):
# 创建项目
resp = client.post(f"{PREFIX}/projects", json={"name": "Proj-1", "code": "p1", "tenant_id": "default"}, headers=_admin_headers())
assert resp.status_code == 200
pid = resp.json()["data"]["id"]
# 查列表
resp = client.get(f"{PREFIX}/projects", params={"tenant_id": "default"}, headers=_admin_headers())
assert resp.status_code == 200
assert any(p["id"] == pid for p in resp.json()["data"])
# 查详情
resp = client.get(f"{PREFIX}/projects/{pid}", headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["name"] == "Proj-1"
# 更新
resp = client.put(f"{PREFIX}/projects/{pid}", json={"description": "updated"}, headers=_admin_headers())
assert resp.status_code == 200
# 归档
resp = client.post(f"{PREFIX}/projects/{pid}/archive", headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["status"] == "archived"
def test_project_members(self, client: TestClient):
resp = client.post(f"{PREFIX}/projects", json={"name": "Proj-M", "code": "pm", "tenant_id": "default"}, headers=_admin_headers())
pid = resp.json()["data"]["id"]
# 加成员
resp = client.post(f"{PREFIX}/projects/{pid}/members", json={"user_id": "u_op", "role": "developer"}, headers=_admin_headers())
assert resp.status_code == 200
# 列成员
resp = client.get(f"{PREFIX}/projects/{pid}/members", headers=_admin_headers())
assert resp.status_code == 200
assert len(resp.json()["data"]) >= 1
# 改角色
resp = client.put(f"{PREFIX}/projects/{pid}/members/u_op", json={"role": "maintainer"}, headers=_admin_headers())
assert resp.status_code == 200
# 删成员
resp = client.delete(f"{PREFIX}/projects/{pid}/members/u_op", headers=_admin_headers())
assert resp.status_code == 200
def test_resource_acl(self, client: TestClient):
# 设置 ACL
resp = client.put(
f"{PREFIX}/resources/model/m001/acl",
json={"entries": [{"subject_type": "user", "subject_id": "u_op", "permissions": ["read", "write"]}]},
headers=_admin_headers(),
)
assert resp.status_code == 200
result = resp.json()["data"]
assert len(result) == 1
assert set(result[0]["permissions"]) == {"read", "write"}
# 查询 ACL
resp = client.get(f"{PREFIX}/resources/model/m001/acl", headers=_admin_headers())
assert resp.status_code == 200
assert len(resp.json()["data"]) == 1
# ============================================================
# 第 3 周测试:审批实例、审批模板、审计日志查询和导出
# ============================================================
class TestWeek3ApprovalAudit:
"""第 3 周:审批实例、审批模板、审计日志查询和导出。"""
def test_approval_template_crud(self, client: TestClient):
# 创建模板
resp = client.post(f"{PREFIX}/approvals/templates", json={"name": "delete-approval", "steps": [{"approver_id": "u_admin", "status": "pending"}]}, headers=_admin_headers())
assert resp.status_code == 200
tpl_id = resp.json()["data"]["id"]
# 查列表
resp = client.get(f"{PREFIX}/approvals/templates", headers=_admin_headers())
assert resp.status_code == 200
assert any(t["id"] == tpl_id for t in resp.json()["data"])
def test_approval_instance_flow(self, client: TestClient):
# 创建审批实例
resp = client.post(f"{PREFIX}/approvals", json={
"resource_type": "dataset", "resource_id": "ds_001",
"applicant_id": "u_op",
}, headers=_admin_headers())
assert resp.status_code == 200
iid = resp.json()["data"]["id"]
# 查详情
resp = client.get(f"{PREFIX}/approvals/{iid}", headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["status"] == "pending"
# 审批决策
resp = client.post(f"{PREFIX}/approvals/{iid}/steps/0/decision", json={
"approver_id": "u_admin", "approved": True, "comment": "ok",
}, headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["status"] == "approved"
def test_approval_instance_reject(self, client: TestClient):
resp = client.post(f"{PREFIX}/approvals", json={
"resource_type": "model", "resource_id": "m_002",
"applicant_id": "u_op",
}, headers=_admin_headers())
iid = resp.json()["data"]["id"]
resp = client.post(f"{PREFIX}/approvals/{iid}/steps/0/decision", json={
"approver_id": "u_admin", "approved": False, "comment": "no",
}, headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["status"] == "rejected"
def test_approval_missing_field(self, client: TestClient):
resp = client.post(f"{PREFIX}/approvals", json={"resource_type": "dataset"}, headers=_admin_headers())
assert resp.status_code == 400
def test_audit_logs_query(self, client: TestClient):
# 通过 API 写操作触发审计
client.post(f"{PREFIX}/tenants", json={"name": "Audit-Tenant", "code": "at"}, headers=_admin_headers())
# 查询
resp = client.get(f"{PREFIX}/system/audit-logs", params={"limit": 50}, headers=_admin_headers())
assert resp.status_code == 200
data = resp.json()["data"]
assert "items" in data
assert "total" in data
assert data["total"] >= 1
def test_audit_logs_filter_by_action(self, client: TestClient):
resp = client.get(f"{PREFIX}/system/audit-logs", params={"action": "tenant.create"}, headers=_admin_headers())
assert resp.status_code == 200
items = resp.json()["data"]["items"]
assert all(i.get("action") == "tenant.create" for i in items)
def test_audit_logs_export_csv(self, client: TestClient):
resp = client.get(f"{PREFIX}/system/audit-logs/export", headers=_admin_headers())
assert resp.status_code == 200
assert "text/csv" in resp.headers.get("content-type", "")
# CSV 首行是表头
lines = resp.text.strip().split("\n")
assert "time" in lines[0]
# ============================================================
# 第 4 周测试:写操作审计、审批拦截、权限校验
# ============================================================
class TestWeek4AuditInterceptPermission:
"""第 4 周:写操作审计、审批拦截、权限校验。"""
def test_write_operation_produces_audit(self, client: TestClient, fake_store: FakePlatformStore):
# 清空审计日志便于断言
fake_store._audit_logs.clear()
# 创建租户 → 应产生 tenant.create 审计
client.post(f"{PREFIX}/tenants", json={"name": "W-Tenant", "code": "wt"}, headers=_admin_headers())
assert any(l["action"] == "tenant.create" for l in fake_store._audit_logs)
# 创建项目 → 应产生 project.create 审计
client.post(f"{PREFIX}/projects", json={"name": "W-Proj", "code": "wp", "tenant_id": "default"}, headers=_admin_headers())
assert any(l["action"] == "project.create" for l in fake_store._audit_logs)
# 设置 ACL → 应产生 resource.acl.set 审计
client.put(f"{PREFIX}/resources/model/w001/acl", json={"entries": []}, headers=_admin_headers())
assert any(l["action"] == "resource.acl.set" for l in fake_store._audit_logs)
def test_approval_intercept_on_project_archive(self, client: TestClient, fake_store: FakePlatformStore):
# 创建项目
resp = client.post(f"{PREFIX}/projects", json={"name": "I-Proj", "code": "ip", "tenant_id": "default"}, headers=_admin_headers())
pid = resp.json()["data"]["id"]
# 无待审批 → 可归档
resp = client.post(f"{PREFIX}/projects/{pid}/archive", headers=_admin_headers())
assert resp.status_code == 200
def test_approval_intercept_blocks_when_pending(self, client: TestClient, fake_store: FakePlatformStore):
# 创建项目
resp = client.post(f"{PREFIX}/projects", json={"name": "B-Proj", "code": "bp", "tenant_id": "default"}, headers=_admin_headers())
pid = resp.json()["data"]["id"]
# 注入一条待审批实例
fake_store.create_approval_instance({
"resource_type": "project",
"resource_id": pid,
"applicant_id": "u_op",
})
# 有待审批 → 归档应被拒绝
resp = client.post(f"{PREFIX}/projects/{pid}/archive", headers=_admin_headers())
assert resp.status_code == 409
def test_retention_policy_crud_with_audit(self, client: TestClient, fake_store: FakePlatformStore):
fake_store._audit_logs.clear()
# 创建
resp = client.post(f"{PREFIX}/retention-policies", json={"name": "30d-keep", "scope": "tenant"}, headers=_admin_headers())
assert resp.status_code == 200
rpid = resp.json()["data"]["id"]
assert any(l["action"] == "retention.create" for l in fake_store._audit_logs)
# 查列表
resp = client.get(f"{PREFIX}/retention-policies", headers=_admin_headers())
assert resp.status_code == 200
assert any(p["id"] == rpid for p in resp.json()["data"])
# 更新
resp = client.put(f"{PREFIX}/retention-policies/{rpid}", json={"status": "inactive"}, headers=_admin_headers())
assert resp.status_code == 200
assert resp.json()["data"]["status"] == "inactive"
# 删除
resp = client.delete(f"{PREFIX}/retention-policies/{rpid}", headers=_admin_headers())
assert resp.status_code == 200
def test_login_duration_rank_in_dashboard(self, client: TestClient):
resp = client.get(f"{PREFIX}/dashboard/stats", headers=_admin_headers())
assert resp.status_code == 200
data = resp.json()["data"]
assert "login_duration_rank" in data
assert "recent_login_users" in data
assert "service_status" in data
assert "training_7d" in data

View File

@@ -8,3 +8,5 @@ httpx>=0.27.0
sacrebleu>=2.4.0 sacrebleu>=2.4.0
rouge-score>=0.1.2 rouge-score>=0.1.2
scikit-learn>=1.3.0 scikit-learn>=1.3.0
# LLaMA-Factory 训练引擎
llamafactory

View File

@@ -83,6 +83,11 @@ assert.match(detailSource, /\.el-button\s*>\s*span[\s\S]*?width:\s*100%[\s\S]*?d
assert.match(detailSource, /\.el-button i[\s\S]*?margin-left:\s*auto/, '输出数据集跳转图标没有统一右对齐') assert.match(detailSource, /\.el-button i[\s\S]*?margin-left:\s*auto/, '输出数据集跳转图标没有统一右对齐')
assert.match(detailSource, /将发布三个独立数据集/, '发布说明仍未明确生成三个独立数据集') assert.match(detailSource, /将发布三个独立数据集/, '发布说明仍未明确生成三个独立数据集')
assert.match(detailSource, /function startRegeneration\(\)[\s\S]*?name: 'data-process-regenerate'[\s\S]*?params: \{ id: taskId\.value \}/, '重新生成按钮没有携带原任务 ID 进入命名路由') assert.match(detailSource, /function startRegeneration\(\)[\s\S]*?name: 'data-process-regenerate'[\s\S]*?params: \{ id: taskId\.value \}/, '重新生成按钮没有携带原任务 ID 进入命名路由')
assert.match(detailSource, /const canRepeatGeneration = computed[\s\S]*?status === 'completed'[\s\S]*?results_confirmed !== false[\s\S]*?previewCount\.value > 0/, '已完成任务缺少再次生成资格判断')
assert.match(detailSource, /repeatDataProcessTask\(taskId\.value,[\s\S]*?expected_updated_at: detail\.value\.updated_at[\s\S]*?request_id: repeatRequestId\.value/, '再次生成没有携带源任务版本和幂等请求 ID')
assert.match(detailSource, /name: 'data-process-workflow'[\s\S]*?params: \{ id: repeated\.task\.id \}/, '再次生成成功后没有进入新任务工作流')
assert.match(detailSource, /原任务和原结果不会被修改/, '再次生成确认提示没有说明原任务保持不变')
assert.match(detailSource, /v-if="canRepeatGeneration"[\s\S]*?@click="repeatGeneration"[\s\S]*?按原配置再生成一批/, '已完成任务详情缺少再次生成新批次入口')
assert.match(detailSource, /const canRegenerate = computed\(\(\) => \{[\s\S]*?status === 'pending'[\s\S]*?status === 'failed'[\s\S]*?status === 'stopped'[\s\S]*?status === 'completed'[\s\S]*?outputDatasetId\.value[\s\S]*?hasPublishedOutputs\.value/, '详情页没有覆盖指针已清空但旧发布数据集仍存在的重新生成任务') assert.match(detailSource, /const canRegenerate = computed\(\(\) => \{[\s\S]*?status === 'pending'[\s\S]*?status === 'failed'[\s\S]*?status === 'stopped'[\s\S]*?status === 'completed'[\s\S]*?outputDatasetId\.value[\s\S]*?hasPublishedOutputs\.value/, '详情页没有覆盖指针已清空但旧发布数据集仍存在的重新生成任务')
assert.match(detailSource, /v-if="canRegenerate"[\s\S]*?@click="startRegeneration"[\s\S]*?重新生成/, '可恢复任务没有收敛为单一重新生成入口') assert.match(detailSource, /v-if="canRegenerate"[\s\S]*?@click="startRegeneration"[\s\S]*?重新生成/, '可恢复任务没有收敛为单一重新生成入口')
assert.match(detailSource, /v-if="detail\.status === 'completed' && !hasCurrentPublishedDataset"[\s\S]*?@click="openPublishDialog"[\s\S]*?发布为三个数据集/, '未发布或发布指针失效的完成任务没有保留发布入口') assert.match(detailSource, /v-if="detail\.status === 'completed' && !hasCurrentPublishedDataset"[\s\S]*?@click="openPublishDialog"[\s\S]*?发布为三个数据集/, '未发布或发布指针失效的完成任务没有保留发布入口')
@@ -115,6 +120,11 @@ assert.match(detailSource, /inputMetricCount\.toLocaleString\(\) \}\} \{\{ input
assert.match(detailSource, /sourceFileCount\.toLocaleString\(\) \}\} 个/, '源文件数量缺少个数单位') assert.match(detailSource, /sourceFileCount\.toLocaleString\(\) \}\} 个/, '源文件数量缺少个数单位')
assert.match(detailSource, /<span>生成结果<\/span><strong>\{\{ numeric\(detail\.output_count\)\.toLocaleString\(\) \}\} 条<\/strong>/, '生成结果数量缺少条数单位或仍误称成功输出') assert.match(detailSource, /<span>生成结果<\/span><strong>\{\{ numeric\(detail\.output_count\)\.toLocaleString\(\) \}\} 条<\/strong>/, '生成结果数量缺少条数单位或仍误称成功输出')
assert.match(detailSource, /const configExpanded = ref\(false\)/, '处理配置没有默认收起') assert.match(detailSource, /const configExpanded = ref\(false\)/, '处理配置没有默认收起')
assert.match(detailSource, /appendGroup\(\['clean_invalid', 'deduplicate'\], '数据清洗'\)/, '详情页没有将完整清洗配置合并为数据清洗')
assert.match(detailSource, /appendGroup\(\['detect_structure', 'normalize_format'\], '结构标准化'\)/, '详情页没有将完整结构配置合并为结构标准化')
assert.match(detailSource, /历史部分配置/, '详情页没有标识旧任务的半组选项')
assert.match(detailSource, /异常数据过滤(历史规则)/, '详情页没有标识已停用的历史异常过滤规则')
assert.match(detailSource, /new Set\(value\.map/, '详情页没有去除历史预处理配置中的重复值')
assert.match(detailSource, /:aria-expanded="configExpanded"/, '处理配置折叠按钮缺少无障碍状态') assert.match(detailSource, /:aria-expanded="configExpanded"/, '处理配置折叠按钮缺少无障碍状态')
assert.match(detailSource, /<el-collapse-transition>[\s\S]*?v-show="configExpanded"/, '处理配置没有折叠过渡或内容状态') assert.match(detailSource, /<el-collapse-transition>[\s\S]*?v-show="configExpanded"/, '处理配置没有折叠过渡或内容状态')
assert.doesNotMatch(detailSource, /const (?:detailMap|completedResults)\b|TODO: 接入真实接口/, '详情页仍包含本地 Mock 数据') assert.doesNotMatch(detailSource, /const (?:detailMap|completedResults)\b|TODO: 接入真实接口/, '详情页仍包含本地 Mock 数据')
@@ -126,6 +136,7 @@ for (const apiName of [
'updateDataProcessResult', 'updateDataProcessResult',
'restoreDataProcessResult', 'restoreDataProcessResult',
'publishDataProcess', 'publishDataProcess',
'repeatDataProcessTask',
]) { ]) {
assert.match( assert.match(
apiSource, apiSource,
@@ -136,5 +147,8 @@ for (const apiName of [
assert.match(apiSource, /keyword\?: string; status\?: string; split\?: string/, '结果列表 API 缺少服务端筛选参数') assert.match(apiSource, /keyword\?: string; status\?: string; split\?: string/, '结果列表 API 缺少服务端筛选参数')
assert.match(apiSource, /\/results\/\$\{encodeURIComponent\(resultId\)\}/, '结果资源路径没有安全编码结果 ID') assert.match(apiSource, /\/results\/\$\{encodeURIComponent\(resultId\)\}/, '结果资源路径没有安全编码结果 ID')
assert.match(apiSource, /`\/data-process\/\$\{encodeURIComponent\(taskId\)\}\/publish`/, '发布 API 路径不正确') assert.match(apiSource, /`\/data-process\/\$\{encodeURIComponent\(taskId\)\}\/publish`/, '发布 API 路径不正确')
assert.match(apiSource, /`\/data-process\/\$\{encodeURIComponent\(taskId\)\}\/repeat`/, '再次生成 API 路径不正确')
assert.match(typesSource, /interface DataProcessRepeatPayload[\s\S]*?expected_updated_at: string[\s\S]*?request_id: string/, '再次生成请求契约不完整')
assert.match(typesSource, /interface DataProcessRepeatResult[\s\S]*?task: DataProcessTask[\s\S]*?source_task_id: string[\s\S]*?created: boolean/, '再次生成响应契约不完整')
console.log('数据处理任务详情真实 API 回归检查通过') console.log('数据处理任务详情真实 API 回归检查通过')

View File

@@ -183,8 +183,36 @@ for (const field of ['sourceStart', 'sourceEnd', 'originalContent', 'editedConte
assert.ok(typesSource.includes(field), `PreviewItem 缺少字段:${field}`) assert.ok(typesSource.includes(field), `PreviewItem 缺少字段:${field}`)
} }
assert.match(typesSource, /sourceFileId/, 'PreviewItem 缺少来源文件标识') assert.match(typesSource, /sourceFileId/, 'PreviewItem 缺少来源文件标识')
assert.match(typesSource, /sourceLocator\?: PreviewSourceLocator/, 'PreviewItem 缺少结构化来源定位契约')
assert.match(typesSource, /headingPath\?: string\[\]/, 'PreviewItem 缺少非结构化标题路径')
assert.match(typesSource, /PreviewSourceLocatorKind = 'json' \| 'jsonl' \| 'csv' \| 'xlsx'/, '前端来源定位 kind 未使用明确联合类型')
assert.match(contractTypesSource, /DataProcessSourceLocatorKind = 'json' \| 'jsonl' \| 'csv' \| 'xlsx'/, 'API 来源定位 kind 未使用明确联合类型')
for (const field of ['kind', 'record_index', 'start_line', 'end_line', 'source_start', 'source_end', 'json_pointer', 'sheet_index', 'sheet_name', 'row_number', 'sheet_record_index']) {
assert.ok(typesSource.includes(field), `PreviewSourceLocator 缺少字段:${field}`)
assert.ok(contractTypesSource.includes(field), `后端来源定位契约缺少字段:${field}`)
}
assert.match(contractTypesSource, /source_locator\?: DataProcessSourceLocator/, '质量信息缺少来源定位契约')
assert.match(contractTypesSource, /heading_path\?: string\[\]/, '质量信息缺少标题路径契约')
assert.match(viewSource, /const sourceLocator = item\.quality_score\?\.source_locator/, '预览映射丢失来源定位')
assert.match(viewSource, /sourceStart:\s*item\.source_start\s*\?\?\s*sourceLocator\?\.source_start/, 'JSON locator 的字符起点没有映射到预览项')
assert.match(viewSource, /sourceEnd:\s*item\.source_end\s*\?\?\s*sourceLocator\?\.source_end/, 'JSON locator 的字符终点没有映射到预览项')
assert.match(viewSource, /sourceStartLine:\s*item\.source_start_line\s*\?\?\s*sourceLocator\?\.start_line/, 'JSON locator 的起始行没有映射到预览项')
assert.match(viewSource, /sourceEndLine:\s*item\.source_end_line\s*\?\?\s*sourceLocator\?\.end_line/, 'JSON locator 的结束行没有映射到预览项')
assert.match(viewSource, /headingPath:[\s\S]*?item\.quality_score\?\.heading_path/, '预览映射丢失标题路径')
assert.match(typesSource, /export type StepId = 'create' \| 'model' \| 'upload' \| 'preview' \| 'generate' \| 'results'/, '步骤类型缺少独立大模型选择步骤') assert.match(typesSource, /export type StepId = 'create' \| 'model' \| 'upload' \| 'preview' \| 'generate' \| 'results'/, '步骤类型缺少独立大模型选择步骤')
assert.match(modelSource, /export function sourceLines/, '缺少源文件行偏移生成函数') assert.match(modelSource, /export function sourceLineWindow/, '缺少有界源文件行窗口函数')
assert.match(modelSource, /maxLines:\s*number/, '源文件行窗口缺少最大渲染行数参数')
assert.doesNotMatch(modelSource, /\.split\(\s*['"]\\n['"]\s*\)/, '源文件行窗口仍会先对全文 split')
assert.match(modelSource, /lines\.length < limit/, '源文件行扫描没有受最大行数约束')
assert.match(modelSource, /unicodeCodePointLength/, '源文件字符偏移未与后端 Unicode code point 计数保持一致')
assert.match(modelSource, /export function sourceLineNumberAtOffset/, '字符偏移缺少无数组的行号解析函数')
const manualPreviewHelperStart = modelSource.indexOf('export function isManualPreviewItem(')
const manualPreviewHelperEnd = modelSource.indexOf('\n}', manualPreviewHelperStart)
assert.ok(manualPreviewHelperStart >= 0, '缺少统一的手动预览项判定函数')
const manualPreviewHelperSource = modelSource.slice(manualPreviewHelperStart, manualPreviewHelperEnd + 2)
for (const field of ['status', 'originalContent', 'sourceStart', 'sourceEnd', 'sourceStartLine', 'sourceEndLine', 'sourcePages', 'sourceLocator']) {
assert.ok(manualPreviewHelperSource.includes(field), `手动预览项判定缺少来源字段:${field}`)
}
assert.doesNotMatch(modelSource, /buildPreviewItems/, '前端不应保留与后端重复的本地切片算法') assert.doesNotMatch(modelSource, /buildPreviewItems/, '前端不应保留与后端重复的本地切片算法')
assert.match(viewSource, /selectedPreviewFileId/, '父页面缺少当前预览文件状态') assert.match(viewSource, /selectedPreviewFileId/, '父页面缺少当前预览文件状态')
const previewBuildBindingStart = viewSource.indexOf('useDataProcessPreviewBuild()') const previewBuildBindingStart = viewSource.indexOf('useDataProcessPreviewBuild()')
@@ -210,6 +238,30 @@ for (const marker of [
} }
assert.match(previewSource, /sourceStart/, '第四步未使用来源起始偏移') assert.match(previewSource, /sourceStart/, '第四步未使用来源起始偏移')
assert.match(previewSource, /sourceEnd/, '第四步未使用来源结束偏移') assert.match(previewSource, /sourceEnd/, '第四步未使用来源结束偏移')
const lineRangeStart = previewSource.indexOf('function lineRange(item: PreviewItem)')
const lineRangeEnd = previewSource.indexOf('\n}', lineRangeStart)
const lineRangeSource = previewSource.slice(lineRangeStart, lineRangeEnd + 2)
assert.match(lineRangeSource, /isManualPreviewItem\(item\)[\s\S]*?手动新增,无源文件定位/, '来源标签仍会把缺少行偏移的正常记录误判为手动新增')
assert.match(lineRangeSource, /props\.processType === 'unstructured'[\s\S]*?来源:源文件记录/, '结构化来源记录缺少无行偏移时的准确标签')
assert.doesNotMatch(lineRangeSource, /sourceStartLine == null[^\n]*手动新增/, '来源标签仍直接以缺少行号判定手动新增')
assert.match(lineRangeSource, /sheet_name[\s\S]*?row_number[\s\S]*?来源:\$\{sheet\} · 第 \$\{locator\.row_number\} 行/, 'XLSX 来源标签没有展示工作表和物理行号')
assert.match(lineRangeSource, /json_pointer[\s\S]*?JSON 路径/, 'JSON 来源标签没有展示 JSON 路径')
assert.match(lineRangeSource, /locator\?\.kind === 'json'[\s\S]*?JSON 根对象/, 'JSON 根对象来源标签被空 JSON Pointer 错误降级')
assert.match(lineRangeSource, /locatedLines[\s\S]*?第 \$\{locatedLines\.start\}[\s\S]*?locatedLines\.end/, 'JSONL/CSV 来源标签没有展示行范围')
assert.match(lineRangeSource, /headingPath[\s\S]*?章节:/, '非结构化来源标签没有合并标题路径')
assert.match(previewSource, /sourceLocator\?\.start_line[\s\S]*?sourceLocator\?\.end_line/, '文本预览没有优先使用后端行号定位')
assert.match(previewSource, /sourceLocator\?\.source_start\s*\?\?\s*item\.sourceStart/, '文本预览没有优先使用 locator 字符起点')
assert.match(previewSource, /sourceLocator\?\.source_end\s*\?\?\s*item\.sourceEnd/, '文本预览没有优先使用 locator 字符终点')
assert.match(previewSource, /data-line-number="line\.number"/, '文本预览行缺少稳定行号定位标识')
assert.match(previewSource, /isLineHighlighted\(line\.number, line\.start, line\.end\)/, '文本预览没有按物理行号高亮')
assert.match(previewSource, /querySelector<HTMLElement>\(`\[data-line-number=/, '选中记录后没有按物理行号滚动定位')
assert.match(previewSource, /const SOURCE_LINE_RENDER_LIMIT = 240/, '源文件查看器缺少安全渲染上限')
assert.match(previewSource, /const SOURCE_LINE_CHARACTER_LIMIT = 4_000/, '源文件查看器缺少单行字符渲染上限')
assert.match(previewSource, /sourceLineWindow\([\s\S]*?SOURCE_LINE_RENDER_LIMIT/, '源文件查看器没有使用有界行窗口')
assert.match(previewSource, /SOURCE_LINE_RENDER_LIMIT,[\s\S]*?SOURCE_LINE_CHARACTER_LIMIT,[\s\S]*?selectedSourceLine\.value,[\s\S]*?selectedSourceOffset\.value/, '单行超大 JSON 没有围绕选中来源构建字符窗口')
assert.match(previewSource, /sourceWindowStartLine/, '源文件查看器缺少窗口起始行状态')
assert.match(previewSource, /showPreviousSourceWindow[\s\S]*?showNextSourceWindow/, '源文件查看器缺少前后窗口导航')
assert.match(previewSource, /sourceLineNumberAtOffset\(props\.sourceText/, '仅有字符偏移时没有解析目标物理行')
assert.match(previewSource, /filterable/, '文件选择器必须可搜索') assert.match(previewSource, /filterable/, '文件选择器必须可搜索')
assert.match(previewSource, /当前文件/, '预览缺少当前文件切换器') assert.match(previewSource, /当前文件/, '预览缺少当前文件切换器')
assert.doesNotMatch(previewSource, /located-badge|sync-label|已定位到/, '源文件栏不应显示冗余定位提示') assert.doesNotMatch(previewSource, /located-badge|sync-label|已定位到/, '源文件栏不应显示冗余定位提示')
@@ -269,6 +321,18 @@ for (const marker of [
]) { ]) {
assert.ok(officeViewerSource.includes(marker), `Word/XLSX 预览缺少结构或行为:${marker}`) assert.ok(officeViewerSource.includes(marker), `Word/XLSX 预览缺少结构或行为:${marker}`)
} }
assert.match(officeViewerSource, /const selectedXlsxLocator = computed/, 'XLSX 查看器没有读取精确来源定位')
assert.match(officeViewerSource, /row\.row_number === locator\.row_number/, 'XLSX 查看器没有按物理行号精确高亮')
assert.match(officeViewerSource, /row\.record_index === locator\.sheet_record_index/, 'XLSX 查看器没有按工作表记录序号精确高亮')
assert.match(officeViewerSource, /Math\.floor\(locator\.sheet_record_index \/ XLSX_PAGE_SIZE\) \* XLSX_PAGE_SIZE/, 'XLSX 查看器没有按记录序号自动计算分页')
assert.match(officeViewerSource, /activeSheetIndex\.value = targetSheet[\s\S]*?pageOffset\.value = targetOffset[\s\S]*?loadPreview\(\)/, '切换记录时 XLSX 查看器没有自动切工作表和分页')
const xlsxHighlightStart = officeViewerSource.indexOf('function xlsxRowHighlighted(')
const xlsxHighlightEnd = officeViewerSource.indexOf('\n}', xlsxHighlightStart)
const xlsxHighlightSource = officeViewerSource.slice(xlsxHighlightStart, xlsxHighlightEnd + 2)
assert.ok(
xlsxHighlightSource.indexOf('locator.row_number') < xlsxHighlightSource.indexOf('selectedRecordKey.value'),
'XLSX 查看器没有把精确定位放在原内容比对 fallback 之前',
)
const taskSetupPath = path.join(createDir, 'TaskSetupStep.vue') const taskSetupPath = path.join(createDir, 'TaskSetupStep.vue')
const structuredOptionsPath = path.join(createDir, 'StructuredOptionsPanel.vue') const structuredOptionsPath = path.join(createDir, 'StructuredOptionsPanel.vue')
@@ -454,7 +518,7 @@ assert.match(
) )
assert.match( assert.match(
workflowInitializationSource, workflowInitializationSource,
/sourceTask\.status === 'running'[\s\S]*?resumeStep = 'generate'[\s\S]*?goToStep\(resumeStep\)[\s\S]*?resumeGeneration/, /sourceTask\.status === 'running'[\s\S]*?resumeStep = 'generate'[\s\S]*?resumeGeneration\(\)[\s\S]*?goToStep\(resumeStep\)/,
'生成运行中时没有强制回到第五步并接管后台进度', '生成运行中时没有强制回到第五步并接管后台进度',
) )
const startGenerationHandler = viewSource.slice( const startGenerationHandler = viewSource.slice(
@@ -463,7 +527,35 @@ const startGenerationHandler = viewSource.slice(
) )
assert.match(startGenerationHandler, /await persistWorkflowStep\('generate'\)[\s\S]*?await startGeneration\(\)[\s\S]*?dirty\.value = false/, '开始生成没有持久化第五步或启动真实后台任务') assert.match(startGenerationHandler, /await persistWorkflowStep\('generate'\)[\s\S]*?await startGeneration\(\)[\s\S]*?dirty\.value = false/, '开始生成没有持久化第五步或启动真实后台任务')
assert.doesNotMatch(startGenerationHandler, /router\.(?:push|replace)|allowLeave\s*=\s*true/, '开始生成后应停留在第五步,不得自动跳回列表') assert.doesNotMatch(startGenerationHandler, /router\.(?:push|replace)|allowLeave\s*=\s*true/, '开始生成后应停留在第五步,不得自动跳回列表')
assert.match(viewSource, /:disabled="currentStepId === 'generate' \|\| previewBuilding \|\| sourceUploading"/, '第五步底部返回按钮没有固定禁用') assert.match(
generationSource,
/const canReturnFromGeneration = computed\(\(\) => \([\s\S]*?generation\.status === 'idle'[\s\S]*?!generationStarting\.value[\s\S]*?!generationRestoring\.value/,
'第五步返回权限没有区分未启动、启动中和恢复中状态',
)
assert.match(
viewSource,
/:disabled="\(currentStepId === 'generate' && !canReturnFromGeneration\) \|\| previewBuilding \|\| sourceUploading"/,
'第五步尚未启动生成时返回按钮仍被禁用',
)
const handleBackStart = viewSource.indexOf('async function handleBack()')
const handleBackEnd = viewSource.indexOf('\n}', handleBackStart)
const handleBackSource = viewSource.slice(handleBackStart, handleBackEnd + 2)
assert.match(
handleBackSource,
/currentStepId\.value === 'generate' && !canReturnFromGeneration\.value/,
'第五步处理函数仍无条件拦截返回',
)
assert.match(
generationSource,
/async function resumeGeneration\(\)[\s\S]*?generationRestoring\.value = true[\s\S]*?await getDataProcessProgress\(taskId\)[\s\S]*?generationRestoring\.value = false/,
'恢复已启动任务时存在短暂可返回的 idle 窗口',
)
assert.match(viewSource, /const resume = resumeGeneration\(\)[\s\S]*?goToStep\(resumeStep\)[\s\S]*?await resume/, '第五步展示时未先启动恢复锁')
assert.match(
generationSource,
/const generationStarting = ref\(false\)[\s\S]*?generationStarting\.value = true[\s\S]*?generationStarting\.value = false/,
'点击开始生成后到请求启动前没有锁定返回状态',
)
assert.match( assert.match(
viewSource, viewSource,
/generation\.status === 'success'[\s\S]*?persistWorkflowStep\('results'\)/, /generation\.status === 'success'[\s\S]*?persistWorkflowStep\('results'\)/,
@@ -506,32 +598,34 @@ assert.match(viewSource, /watch\(processType,[\s\S]*?resetSourceDataForProcessTy
assert.match(viewSource, /function resetSourceDataForProcessTypeChange\(\)[\s\S]*?uploadedFiles\.value = \[\][\s\S]*?selectedPreviewFileId\.value = null/, '旧源数据失效没有同步清理文件与预览选择') assert.match(viewSource, /function resetSourceDataForProcessTypeChange\(\)[\s\S]*?uploadedFiles\.value = \[\][\s\S]*?selectedPreviewFileId\.value = null/, '旧源数据失效没有同步清理文件与预览选择')
assert.match(taskSetupSource, /v-if="processType === 'structured'"/, '结构化配置必须仅在结构化数据类型下显示') assert.match(taskSetupSource, /v-if="processType === 'structured'"/, '结构化配置必须仅在结构化数据类型下显示')
const expectedStructuredOptions = [ const expectedStructuredGroups = [
['clean_invalid', '清理无效数据', '清理全空列,并剔除关键字段残缺的数据行'],
[ [
'detect_structure', "values: ['clean_invalid', 'deduplicate']",
'嵌套结构展平', '数据清洗',
'展平嵌套对象和可解析的 JSON 字段Excel 表头与合并单元格在上传时自动解析', '清理全空列和空记录,并删除内容完全相同的记录;不会猜测可空字段是否必填',
], ],
[ [
'deduplicate', "values: ['detect_structure', 'normalize_format']",
'重复记录去重', '结构标准化',
'按整行内容或 id、uuid、key、code、*_id 等身份字段去重,暂不支持自定义组合字段', '展平嵌套对象和可解析的 JSON 字段,并统一编码、空白、字段名和 JSON 序列化格式',
], ],
['normalize_format', '数据格式标准化', '按所选规则统一编码、空白、字段名及 JSON 序列化格式'], ["values: ['desensitize']", '敏感信息脱敏', '识别并脱敏姓名、手机号、邮箱和身份证号'],
['filter_anomaly', '异常数据过滤', '使用 IQR 识别数值离群值,并过滤乱码等异常记录'],
['desensitize', '敏感信息脱敏', '识别并脱敏姓名、手机号、邮箱和身份证号'],
] ]
for (const [value, label, description] of expectedStructuredOptions) { for (const [values, label, description] of expectedStructuredGroups) {
assert.ok(structuredOptionsSource.includes(`value: '${value}'`), `结构化预处理缺少值${value}`) assert.ok(structuredOptionsSource.includes(values), `结构化预处理组合值不准确${label}`)
assert.ok(structuredOptionsSource.includes(`label: '${label}'`), `结构化预处理缺少标签:${label}`) assert.ok(structuredOptionsSource.includes(`label: '${label}'`), `结构化预处理缺少标签:${label}`)
assert.ok(structuredOptionsSource.includes(`description: '${description}'`), `结构化预处理语义不准确:${value}`) assert.ok(structuredOptionsSource.includes(`description: '${description}'`), `结构化预处理语义不准确:${label}`)
} }
const structuredOptionValues = [...structuredOptionsSource.matchAll(/\{\s*value: '([^']+)',\s*label:/g)] assert.equal(expectedStructuredGroups.length, 3, '结构化预处理应收敛为 3 项')
.map((match) => match[1]) const preprocessGroupsSource = structuredOptionsSource.slice(
assert.deepEqual(structuredOptionValues, expectedStructuredOptions.map(([value]) => value), '结构化预处理值集合不准确') structuredOptionsSource.indexOf('const PREPROCESS_GROUPS'),
assert.equal(new Set(structuredOptionValues).size, structuredOptionValues.length, '结构化预处理 value 必须唯一') structuredOptionsSource.indexOf('const legacyAnomalyFilterEnabled'),
assert.match(structuredOptionsSource, /Array\.from\(new Set\(value\.filter\(/, '结构化预处理选中值没有去重') )
assert.doesNotMatch(preprocessGroupsSource, /异常数据过滤|filter_anomaly|IQR/, '结构化新任务仍暴露异常数据过滤')
assert.match(structuredOptionsSource, /:indeterminate="groupIndeterminate\(group\.values\)"/, '历史部分选中的组合项没有半选回显')
assert.match(structuredOptionsSource, /function updatePreprocessGroup\([\s\S]*?new Set\(props\.options\.preprocessOptions\)[\s\S]*?next\.add\(value\)[\s\S]*?next\.delete\(value\)[\s\S]*?\[\.\.\.next\]/, '结构化预处理组合开关没有原子化更新或去重内部选项')
assert.match(typesSource, /仅用于恢复历史任务[\s\S]*?\| 'filter_anomaly'/, '异常数据过滤缺少历史兼容类型')
assert.match(structuredOptionsSource, /legacyAnomalyFilterEnabled[\s\S]*?历史任务[\s\S]*?结果可复现/, '历史异常过滤配置没有透明提示')
assert.ok(structuredOptionsSource.includes('生成选项'), '结构化配置缺少生成选项分类') assert.ok(structuredOptionsSource.includes('生成选项'), '结构化配置缺少生成选项分类')
for (const splitName of ['训练集', '验证集', '测试集']) { for (const splitName of ['训练集', '验证集', '测试集']) {
assert.ok(datasetSplitEditorSource.includes(splitName), `生成选项缺少数据集划分:${splitName}`) assert.ok(datasetSplitEditorSource.includes(splitName), `生成选项缺少数据集划分:${splitName}`)
@@ -585,8 +679,20 @@ for (const extension of ['txt', 'md', 'markdown', 'pdf', 'docx', 'pptx', 'json',
} }
assert.match(sourceUploadWorkerSource, /LEGACY_OFFICE_EXTENSIONS = new Set\(\['doc', 'xls', 'ppt'\]\)/, '缺少旧版 Office 格式识别') assert.match(sourceUploadWorkerSource, /LEGACY_OFFICE_EXTENSIONS = new Set\(\['doc', 'xls', 'ppt'\]\)/, '缺少旧版 Office 格式识别')
assert.ok(sourceUploadWorkerSource.includes('请分别转换为 DOCX、XLSX、PPTX 后上传'), '旧版 Office 文件缺少转换提示') assert.ok(sourceUploadWorkerSource.includes('请分别转换为 DOCX、XLSX、PPTX 后上传'), '旧版 Office 文件缺少转换提示')
assert.match(sourceUploadWorkerSource, /if \(!BINARY_FILE_EXTENSIONS\.has\(job\.extension\)\) \{[\s\S]*?TextDecoder/, '文本格式没有执行 UTF-8 客户端校验') const sourceValidationStart = sourceUploadWorkerSource.indexOf('export function validateSourceFileSelection(')
assert.match(sourceUploadWorkerSource, /if \(BINARY_FILE_EXTENSIONS\.has\(job\.extension\)\) \{[\s\S]*?getDataProcessSourceContent\(currentTaskId, source\.id,[\s\S]*?start_line:\s*1,[\s\S]*?line_count:\s*10_000/, '二进制文档上传后没有读取后端解析文本') const sourceValidationEnd = sourceUploadWorkerSource.indexOf('\n}\n\nfunction unicodeCodePointLength', sourceValidationStart)
assert.ok(sourceValidationStart >= 0 && sourceValidationEnd > sourceValidationStart, '无法定位源文件选择校验函数')
const sourceValidationSource = sourceUploadWorkerSource.slice(sourceValidationStart, sourceValidationEnd + 2)
assert.doesNotMatch(sourceValidationSource, /file\.name === raw\.name[\s\S]{0,160}file\.size === raw\.size|同名且同大小/, '不同内容但同名同大小的文件仍会被前端误拒绝')
assert.match(sourceValidationSource, /selectedFiles\.length >= MAX_SOURCE_FILE_COUNT/, '移除伪重复校验时误删了文件数量限制')
assert.match(sourceValidationSource, /selectedBytes \+ raw\.size > MAX_SOURCE_BATCH_BYTES/, '移除伪重复校验时误删了批次大小限制')
assert.doesNotMatch(sourceUploadWorkerSource, /job\.file\.arrayBuffer\(|new TextDecoder/, '上传前仍把整个文本文件读入浏览器内存')
assert.match(sourceUploadWorkerSource, /export async function loadCanonicalSourceContent[\s\S]*?offset,[\s\S]*?limit: SOURCE_CONTENT_PAGE_CHARS/, '服务端 canonical content 没有按有界字符窗口读取')
assert.match(sourceUploadWorkerSource, /pending\.content = await loadCanonicalSourceContent\(currentTaskId, source\.id\)/, '上传成功后没有统一使用服务端 canonical content')
assert.doesNotMatch(sourceUploadWorkerSource, /\brawFile:\s*job\.file\b/, '上传成功状态仍长期保留原始 File')
assert.doesNotMatch(typesSource, /\brawFile\??:\s*File\b/, '上传状态类型仍长期持有原始 File')
assert.doesNotMatch(viewSource, /\brawFile:\s*raw\b/, '待上传列表仍复制保存原始 File')
assert.match(apiSource, /params:\s*\{[\s\S]*?offset\?: number[\s\S]*?limit\?: number[\s\S]*?\}/, '正文 API 前端契约缺少字符窗口参数')
assert.match(apiSource, /formData\.append\('files', file\)/, '上传 API 没有使用 files 多文件表单字段') assert.match(apiSource, /formData\.append\('files', file\)/, '上传 API 没有使用 files 多文件表单字段')
assert.match(apiSource, /onUploadProgress:[\s\S]*?event\.loaded \/ event\.total[\s\S]*?Math\.min\(99,/, '上传 API 没有接入真实字节进度或响应前未限制在 99%') assert.match(apiSource, /onUploadProgress:[\s\S]*?event\.loaded \/ event\.total[\s\S]*?Math\.min\(99,/, '上传 API 没有接入真实字节进度或响应前未限制在 99%')
assert.match(apiSource, /source-files`[\s\S]*?timeout: 5 \* 60 \* 1000/, '源文件上传缺少 5 分钟超时') assert.match(apiSource, /source-files`[\s\S]*?timeout: 5 \* 60 \* 1000/, '源文件上传缺少 5 分钟超时')
@@ -609,7 +715,7 @@ assert.match(
/export interface DataProcessPreviewProgress[\s\S]*?workflow_step: DataProcessWorkflowStep[\s\S]*?preview_status: DataProcessPreviewStatus[\s\S]*?preview_progress: number[\s\S]*?preview_run_id/, /export interface DataProcessPreviewProgress[\s\S]*?workflow_step: DataProcessWorkflowStep[\s\S]*?preview_status: DataProcessPreviewStatus[\s\S]*?preview_progress: number[\s\S]*?preview_run_id/,
'后台切分进度契约缺少步骤、状态、进度或任务代次', '后台切分进度契约缺少步骤、状态、进度或任务代次',
) )
for (const field of ['rawFile', 'status', 'uploadProgress', 'uploadError', 'previewStatus', 'previewProgress', 'previewError', 'previewConfigSignature']) { for (const field of ['status', 'uploadProgress', 'uploadError', 'previewStatus', 'previewProgress', 'previewError', 'previewConfigSignature']) {
assert.ok(typesSource.includes(field), `上传文件缺少逐文件预览字段:${field}`) assert.ok(typesSource.includes(field), `上传文件缺少逐文件预览字段:${field}`)
} }
assert.match(typesSource, /status: 'queued' \| 'uploading' \| 'ready' \| 'failed'/, '上传文件状态机不完整') assert.match(typesSource, /status: 'queued' \| 'uploading' \| 'ready' \| 'failed'/, '上传文件状态机不完整')
@@ -853,13 +959,23 @@ const defaultStructuredPreprocess = defaultPreprocessValues(
) )
assert.deepEqual( assert.deepEqual(
defaultStructuredPreprocess, defaultStructuredPreprocess,
['clean_invalid', 'detect_structure', 'deduplicate', 'normalize_format'], [],
'结构化默认预处理配置不准确', '结构化新任务不应默认勾选预处理',
) )
assert.equal(new Set(defaultStructuredPreprocess).size, defaultStructuredPreprocess.length, '结构化默认预处理值重复') assert.equal(new Set(defaultStructuredPreprocess).size, defaultStructuredPreprocess.length, '结构化默认预处理值重复')
const defaultUnstructuredPreprocess = defaultPreprocessValues('createDefaultUnstructuredOptions') const defaultUnstructuredPreprocess = defaultPreprocessValues('createDefaultUnstructuredOptions')
assert.deepEqual(defaultUnstructuredPreprocess, expectedSmartPreprocessOptions, '智能预处理默认值不完整') assert.deepEqual(defaultUnstructuredPreprocess, [], '非结构化新任务不应默认勾选预处理')
assert.equal(new Set(defaultUnstructuredPreprocess).size, defaultUnstructuredPreprocess.length, '非结构化默认预处理值重复') assert.equal(new Set(defaultUnstructuredPreprocess).size, defaultUnstructuredPreprocess.length, '非结构化默认预处理值重复')
for (const field of ['preserveTables', 'preserveCodeBlocks', 'preserveLists']) {
assert.match(
stateSource,
new RegExp(`${field}:\\s*false`),
`非结构化预处理选项 ${field} 不应默认开启`,
)
}
assert.match(structuredOptionsSource, /默认不执行预处理,请按数据情况自行选择/, '结构化预处理缺少默认不勾选说明')
assert.match(unstructuredOptionsSource, /默认不执行预处理,请按文档情况自行选择/, '非结构化预处理缺少默认不勾选说明')
assert.doesNotMatch(unstructuredOptionsSource, /默认启用结构感知/, '非结构化预处理仍保留默认启用的误导文案')
const backendConfigStart = viewSource.indexOf('function toBackendConfig()') const backendConfigStart = viewSource.indexOf('function toBackendConfig()')
const backendConfigEnd = viewSource.indexOf('function taskPayload()', backendConfigStart) const backendConfigEnd = viewSource.indexOf('function taskPayload()', backendConfigStart)
@@ -934,7 +1050,7 @@ for (const [field, fallback] of [
) )
} }
assert.match(regenerationSource, /getDataProcessTask\(sourceTaskId\.value\)/, '重新生成没有加载原任务') assert.match(regenerationSource, /getDataProcessTask\(sourceTaskId\.value\)/, '重新生成没有加载原任务')
assert.match(regenerationSource, /while \(true\)[\s\S]*?getDataProcessSourceContent[\s\S]*?has_more/, '重新生成没有分页加载完整源正文') assert.match(regenerationSource, /loadCanonicalSourceContent\(taskId, file\.id\)/, '重新生成没有复用分页 canonical 正文加载器')
assert.match(regenerationSource, /getDataProcessPreview\(taskId, \{ page: 1, page_size: 500 \}\)[\s\S]*?for \(let page = 2; page <= pages;/, '重新生成没有分页加载全部现有切片') assert.match(regenerationSource, /getDataProcessPreview\(taskId, \{ page: 1, page_size: 500 \}\)[\s\S]*?for \(let page = 2; page <= pages;/, '重新生成没有分页加载全部现有切片')
assert.match(viewSource, /if \(hydrating\.value\) return/, '任务水合期间仍可能触发重置副作用') assert.match(viewSource, /if \(hydrating\.value\) return/, '任务水合期间仍可能触发重置副作用')
assert.match(regenerationSource, /currentSignature !== originalPreviewConfigSignature\.value[\s\S]*?currentSignature === confirmedPreviewConfigSignature\.value/, '切分变更确认没有按原签名和已确认签名去重') assert.match(regenerationSource, /currentSignature !== originalPreviewConfigSignature\.value[\s\S]*?currentSignature === confirmedPreviewConfigSignature\.value/, '切分变更确认没有按原签名和已确认签名去重')
@@ -948,7 +1064,7 @@ assert.match(regenerationSource, /if \(regenerationPrepared\.value\) \{[\s\S]*?g
assert.match(regenerationSource, /regenerationPrepared\.value = true/, '重新生成提交成功后没有记录服务端已变更状态') assert.match(regenerationSource, /regenerationPrepared\.value = true/, '重新生成提交成功后没有记录服务端已变更状态')
assert.match(regenerationSource, /hydrateWorkspace\(regeneratedTask, !regenerated\.preview_invalidated\)/, '重新生成没有按 preview_invalidated 决定保留或清空切片') assert.match(regenerationSource, /hydrateWorkspace\(regeneratedTask, !regenerated\.preview_invalidated\)/, '重新生成没有按 preview_invalidated 决定保留或清空切片')
assert.match(regenerationSource, /重新生成配置已保存,但工作区恢复失败/, '重新生成配置已保存但水合失败时缺少可恢复错误状态') assert.match(regenerationSource, /重新生成配置已保存,但工作区恢复失败/, '重新生成配置已保存但水合失败时缺少可恢复错误状态')
assert.match(regenerationSource, /return chunks\.join\(''\)/, '分页恢复源正文时不应额外插入换行') assert.match(sourceUploadWorkerSource, /return chunks\.join\(''\)/, '分页恢复源正文时不应额外插入换行')
assert.doesNotMatch(regenerationSource, /binaryDocument[\s\S]*?mapDataProcessSourceFile\(file, ''\)/, '二进制源正文加载失败时不能静默降级为空内容') assert.doesNotMatch(regenerationSource, /binaryDocument[\s\S]*?mapDataProcessSourceFile\(file, ''\)/, '二进制源正文加载失败时不能静默降级为空内容')
assert.match(nextFromModelSource, /if \(isRegeneration\.value\) \{[\s\S]*?prepareRegeneration\(taskPayload\(\)\)/, '重新生成每次从模型步骤继续时没有调用专用接口') assert.match(nextFromModelSource, /if \(isRegeneration\.value\) \{[\s\S]*?prepareRegeneration\(taskPayload\(\)\)/, '重新生成每次从模型步骤继续时没有调用专用接口')
assert.doesNotMatch(nextFromModelSource, /isRegeneration\.value && !taskId\.value/, '重新生成提交一次后可能错误转为普通任务更新') assert.doesNotMatch(nextFromModelSource, /isRegeneration\.value && !taskId\.value/, '重新生成提交一次后可能错误转为普通任务更新')
@@ -1017,6 +1133,17 @@ for (const mutationFunction of [
const mutationSource = viewSource.slice(mutationStart, mutationEnd === -1 ? undefined : mutationEnd) const mutationSource = viewSource.slice(mutationStart, mutationEnd === -1 ? undefined : mutationEnd)
assert.ok(mutationSource.includes('resetDownstream()'), `预览变更 ${mutationFunction} 后没有失效旧生成结果`) assert.ok(mutationSource.includes('resetDownstream()'), `预览变更 ${mutationFunction} 后没有失效旧生成结果`)
} }
const updatePreviewContentStart = viewSource.indexOf('function updatePreviewContent(')
const updatePreviewContentEnd = viewSource.indexOf('\n}', updatePreviewContentStart)
const updatePreviewContentSource = viewSource.slice(updatePreviewContentStart, updatePreviewContentEnd + 2)
assert.match(updatePreviewContentSource, /isManualPreviewItem\(item\)/, '编辑预览内容仍未按稳定来源信息区分手动项')
assert.doesNotMatch(updatePreviewContentSource, /sourceStart == null/, '结构化来源记录编辑后仍会被误标为手动项')
const restorePreviewItemStart = viewSource.indexOf('function restorePreviewItem(')
const restorePreviewItemEnd = viewSource.indexOf('\n}', restorePreviewItemStart)
const restorePreviewItemSource = viewSource.slice(restorePreviewItemStart, restorePreviewItemEnd + 2)
assert.match(restorePreviewItemSource, /isManualPreviewItem\(item\)/, '恢复预览内容没有使用统一的手动项判定')
assert.doesNotMatch(restorePreviewItemSource, /sourceStart == null/, '结构化来源记录仍因缺少字符偏移而无法恢复')
assert.match(previewSource, /v-if="!isManualPreviewItem\(editingItem\)"/, '结构化来源记录的恢复原文按钮仍被错误隐藏')
assert.doesNotMatch(modelSource, /createResults\(/, '纯预览映射模块不应承担结果生成职责') assert.doesNotMatch(modelSource, /createResults\(/, '纯预览映射模块不应承担结果生成职责')
function findNextStyleBlockStart(source, startIndex) { function findNextStyleBlockStart(source, startIndex) {

View File

@@ -1,9 +1,62 @@
<script setup lang="ts"> <script setup lang="ts">
import { onMounted, onUnmounted } from 'vue'
import { useRouter } from 'vue-router'
import zhCn from 'element-plus/es/locale/lang/zh-cn' import zhCn from 'element-plus/es/locale/lang/zh-cn'
import { ElMessage } from 'element-plus'
import { routeLoading } from '@/router'
import { useAuthStore } from '@/stores/auth'
import { SESSION_TIMEOUT } from '@/constants'
const router = useRouter()
const auth = useAuthStore()
/**
* 离开页面超时:
* - 标签页切走/最小化document.hidden时记录时间
* - 切回来时若超过 SESSION_TIMEOUT5分钟强制跳登录
* - 不管是否在操作,只要离开页面超过 5 分钟就跳
*/
let hiddenAt = 0
function handleVisibility() {
if (document.hidden) {
hiddenAt = Date.now()
} else {
if (hiddenAt > 0 && Date.now() - hiddenAt >= SESSION_TIMEOUT) {
auth.logout()
ElMessage.warning('登录已过期,请重新登录')
router.push('/login')
}
hiddenAt = 0
}
}
onMounted(() => {
document.addEventListener('visibilitychange', handleVisibility)
})
onUnmounted(() => {
document.removeEventListener('visibilitychange', handleVisibility)
})
</script> </script>
<template> <template>
<el-config-provider :locale="zhCn"> <el-config-provider :locale="zhCn">
<div v-loading="routeLoading" element-loading-text="加载中..." class="app-root">
<router-view /> <router-view />
</div>
</el-config-provider> </el-config-provider>
</template> </template>
<style>
html,
body,
#app {
height: 100%;
margin: 0;
}
.app-root {
height: 100%;
position: relative;
}
</style>

View File

@@ -0,0 +1,15 @@
import { get, put } from '../request'
export interface AclEntry {
subject_type: string
subject_id: string
permissions: string[]
}
/** 资源 ACL 查询 */
export const getAcl = (resourceType: string, resourceId: string) =>
get<AclEntry[]>(`/resources/${resourceType}/${resourceId}/acl`)
/** 资源 ACL 设置 */
export const setAcl = (resourceType: string, resourceId: string, entries: AclEntry[]) =>
put<AclEntry[]>(`/resources/${resourceType}/${resourceId}/acl`, { entries })

View File

@@ -0,0 +1,50 @@
import { get, post } from '../request'
export interface ApprovalStep {
approver_id?: string | null
status: string
}
export interface ApprovalTemplate {
id: string
name: string
steps: ApprovalStep[]
create_time?: string
}
export interface ApprovalInstance {
id: string
template_id?: string | null
resource_type: string
resource_id: string
applicant_id: string
status: string
current_step: number
create_time?: string
steps: Array<ApprovalStep & { step_index: number; comment?: string | null; time?: string | null }>
}
export const getApprovalTemplates = () =>
get<ApprovalTemplate[]>('/approvals/templates')
export const createApprovalTemplate = (payload: { name: string; steps: ApprovalStep[] }) =>
post<ApprovalTemplate>('/approvals/templates', payload)
export const getApprovalInstances = (status?: string) =>
get<ApprovalInstance[]>('/approvals', { status })
export const createApprovalInstance = (payload: {
template_id?: string
resource_type: string
resource_id: string
applicant_id: string
}) => post<ApprovalInstance>('/approvals', payload)
export const getApprovalInstance = (id: string) =>
get<ApprovalInstance>(`/approvals/${id}`)
export const decideApproval = (
id: string,
step_index: number,
payload: { approver_id: string; approved: boolean; comment?: string },
) => post<ApprovalInstance>(`/approvals/${id}/steps/${step_index}/decision`, payload)

View File

@@ -0,0 +1,40 @@
import { get } from '../request'
import request from '../request'
export interface AuditLog {
id: string
tenant_id?: string
project_id?: string
actor_id?: string
action?: string
target_type?: string
target_id?: string
detail?: string
client_ip?: string
time?: string
}
export interface AuditQuery {
tenant_id?: string
project_id?: string
actor_id?: string
action?: string
target_type?: string
start_time?: string
end_time?: string
limit?: number
offset?: number
}
/** 审计日志查询:使用 get 辅助函数,拦截器已解包,直接返回 { items, total } */
export const getAuditLogs = (query: AuditQuery = {}) =>
get<{ items: AuditLog[]; total: number }>('/system/audit-logs', query)
/** 审计日志导出 CSVblob 响应走完整 axios response需手动取 data */
export const exportAuditLogs = (query: AuditQuery = {}) =>
request<Blob>({
url: '/system/audit-logs/export',
method: 'get',
params: query,
responseType: 'blob',
}).then((res) => res.data)

View File

@@ -0,0 +1,35 @@
import { get } from '../request'
export interface ServiceStatusStat {
type: string
status: 'normal' | 'busy' | 'error'
count: number
}
export interface TrainingTaskStat {
id: string
name: string
status: string
train_type: string
train_method: string
base_model: string
progress: number
accuracy: number | null
started_at: string
}
export interface DashboardStats {
online_services: number
running_tasks: number
pending_alerts: number
training_7d: { date: string; train: number; gpu: number; accuracy: number | null }[]
service_status: ServiceStatusStat[]
training_tasks: TrainingTaskStat[]
operation_distribution: { name: string; value: number }[]
login_duration_rank: { user: string; role: string; duration: number }[]
recent_login_users: { user: string; role: string; last_login: string }[]
}
export function getDashboardStats() {
return get<DashboardStats>('/dashboard/stats')
}

View File

@@ -14,6 +14,8 @@ import type {
DataProcessProgress, DataProcessProgress,
DataProcessRegeneratePayload, DataProcessRegeneratePayload,
DataProcessRegenerateResult, DataProcessRegenerateResult,
DataProcessRepeatPayload,
DataProcessRepeatResult,
DataProcessPublishPayload, DataProcessPublishPayload,
DataProcessPublishResult, DataProcessPublishResult,
DataProcessQualityScore, DataProcessQualityScore,
@@ -56,6 +58,8 @@ export type {
DataProcessProgress, DataProcessProgress,
DataProcessRegeneratePayload, DataProcessRegeneratePayload,
DataProcessRegenerateResult, DataProcessRegenerateResult,
DataProcessRepeatPayload,
DataProcessRepeatResult,
DataProcessPublishPayload, DataProcessPublishPayload,
DataProcessPublishResult, DataProcessPublishResult,
DataProcessQualityScore, DataProcessQualityScore,
@@ -117,6 +121,15 @@ export const regenerateDataProcessTask = (
payload, payload,
) )
export const repeatDataProcessTask = (
taskId: string | number,
payload: DataProcessRepeatPayload,
) => post<DataProcessRepeatResult>(
`/data-process/${encodeURIComponent(taskId)}/repeat`,
payload,
{ timeout: 5 * 60 * 1000 },
)
export const deleteDataProcessTask = (taskId: string | number) => export const deleteDataProcessTask = (taskId: string | number) =>
del<{ deleted: string | number }>(`/data-process/${encodeURIComponent(taskId)}`) del<{ deleted: string | number }>(`/data-process/${encodeURIComponent(taskId)}`)
@@ -150,7 +163,12 @@ export const deleteDataProcessSourceFile = (taskId: string | number, fileId: str
export const getDataProcessSourceContent = ( export const getDataProcessSourceContent = (
taskId: string | number, taskId: string | number,
fileId: string | number, fileId: string | number,
params: { start_line?: number; line_count?: number } = {}, params: {
start_line?: number
line_count?: number
offset?: number
limit?: number
} = {},
) => get<DataProcessSourceContent>( ) => get<DataProcessSourceContent>(
`/data-process/${encodeURIComponent(taskId)}/source-files/${encodeURIComponent(fileId)}/content`, `/data-process/${encodeURIComponent(taskId)}/source-files/${encodeURIComponent(fileId)}/content`,
params, params,

View File

@@ -0,0 +1,55 @@
import { del, get, post, put } from '../request'
export interface Project {
id: string
tenant_id: string
name: string
code: string
description?: string
status: string
quota?: Record<string, unknown>
member_count?: number
task_count?: number
create_time?: string
}
export interface ProjectMember {
user_id: string
role: string
joined_at?: string
}
/** 项目列表(按租户过滤,默认 default */
export const getProjects = (tenantId = 'default') =>
get<Project[]>('/projects', { tenant_id: tenantId })
/** 项目详情 */
export const getProject = (id: string) => get<Project>(`/projects/${id}`)
/** 创建项目 */
export const createProject = (payload: Partial<Project>) =>
post<Project>('/projects', payload)
/** 更新项目 */
export const updateProject = (id: string, payload: Partial<Project>) =>
put<Project>(`/projects/${id}`, payload)
/** 归档项目 */
export const archiveProject = (id: string) =>
post<Project>(`/projects/${id}/archive`)
/** 项目成员列表 */
export const getProjectMembers = (id: string) =>
get<ProjectMember[]>(`/projects/${id}/members`)
/** 添加成员 */
export const addProjectMember = (id: string, payload: { user_id: string; role: string }) =>
post<ProjectMember>(`/projects/${id}/members`, payload)
/** 更新成员角色 */
export const updateProjectMember = (id: string, userId: string, role: string) =>
put<ProjectMember>(`/projects/${id}/members/${userId}`, { role })
/** 移除成员 */
export const removeProjectMember = (id: string, userId: string) =>
del(`/projects/${id}/members/${userId}`)

View File

@@ -0,0 +1,29 @@
import { del, get, post, put } from '../request'
export interface RetentionPolicy {
id: string
name: string
scope?: string | null
rule?: string | null
status: string
create_time?: string
create_by?: string | null
updated_at?: string
}
/** 留存策略列表 */
export const getRetentionPolicies = () => get<RetentionPolicy[]>('/retention-policies')
/** 留存策略详情 */
export const getRetentionPolicy = (id: string) => get<RetentionPolicy>(`/retention-policies/${id}`)
/** 创建留存策略 */
export const createRetentionPolicy = (payload: Partial<RetentionPolicy>) =>
post<RetentionPolicy>('/retention-policies', payload)
/** 更新留存策略 */
export const updateRetentionPolicy = (id: string, payload: Partial<RetentionPolicy>) =>
put<RetentionPolicy>(`/retention-policies/${id}`, payload)
/** 删除留存策略 */
export const deleteRetentionPolicy = (id: string) => del(`/retention-policies/${id}`)

View File

@@ -25,12 +25,14 @@ export const getUsers = () => get<SystemUser[]>('/users')
export const createUser = (payload: CreateUserPayload) => export const createUser = (payload: CreateUserPayload) =>
post<SystemUser>('/users', payload) post<SystemUser>('/users', payload)
/** 删除用户currentUsername 用于防止删除当前登录账号 */
export const deleteUser = (id: string, currentUsername: string) =>
del<{ deleted: string }>(`/users/${encodeURIComponent(id)}`, {
current_username: currentUsername,
})
/** 更新用户角色、状态及页面权限 */ /** 更新用户角色、状态及页面权限 */
export const updateUserAccess = (id: string, payload: UpdateUserAccessPayload) => export const updateUserAccess = (id: string, payload: UpdateUserAccessPayload) =>
put<SystemUser>(`/users/${encodeURIComponent(id)}`, payload) put<SystemUser>(`/users/${encodeURIComponent(id)}`, payload)
/** 重置用户密码 */
export const resetUserPassword = (id: string, password?: string) =>
post<{ reset: string }>(`/users/${encodeURIComponent(id)}/reset-password`, { password })
/** 删除用户protected 管理员账号不允许删除) */
export const deleteUser = (id: string) =>
del<{ deleted: string }>(`/users/${encodeURIComponent(id)}`)

View File

@@ -0,0 +1,34 @@
import { get, post, put } from '../request'
export interface Tenant {
id: string
name: string
code: string
status: string
owner_user_id?: string | null
quota: Record<string, unknown>
retention_policy_id?: string | null
create_time?: string
}
/** 租户列表 */
export const getTenants = () => get<Tenant[]>('/tenants')
/** 租户详情 */
export const getTenant = (id: string) => get<Tenant>(`/tenants/${id}`)
/** 创建租户 */
export const createTenant = (payload: Partial<Tenant>) =>
post<Tenant>('/tenants', payload)
/** 更新租户 */
export const updateTenant = (id: string, payload: Partial<Tenant>) =>
put<Tenant>(`/tenants/${id}`, payload)
/** 设置租户配额 */
export const setTenantQuota = (id: string, quota: Record<string, unknown>) =>
put<Tenant>(`/tenants/${id}/quota`, { quota })
/** 设置租户留存策略 */
export const setTenantRetention = (id: string, retention_policy_id: string) =>
put<Tenant>(`/tenants/${id}/retention-policy`, { retention_policy_id })

View File

@@ -1,6 +1,5 @@
import axios, { type AxiosInstance, type AxiosRequestConfig } from 'axios' import axios, { type AxiosInstance, type AxiosRequestConfig } from 'axios'
import { ElMessage } from 'element-plus' import { ElMessage } from 'element-plus'
import { touchSessionActivity } from '@/utils/sessionActivity'
/** /**
* 后端统一响应格式 * 后端统一响应格式
@@ -18,9 +17,37 @@ const service: AxiosInstance = axios.create({
timeout: 30000, timeout: 30000,
}) })
// 请求拦截器 /**
* 从 localStorage 取当前用户 token登录时后端返回 platform-token-{user_id})。
* 后端鉴权中间件依赖此 header 解析当前用户身份。
*/
function getAuthToken(): string | null {
const USER_STORAGE_KEY = 'currentUser'
const raw = localStorage.getItem(USER_STORAGE_KEY)
if (raw) {
try {
const user = JSON.parse(raw)
// 后端 login 返回的 token 格式为 platform-token-{user.id}
if (user?.id) return `platform-token-${user.id}`
} catch {
/* ignore */
}
}
// 兼容改造前 admin 会话
if (localStorage.getItem('username') === 'admin') return 'platform-token-admin'
return null
}
// 请求拦截器:注入 Authorization header
service.interceptors.request.use( service.interceptors.request.use(
(config) => config, (config) => {
const token = getAuthToken()
if (token) {
config.headers = config.headers || {}
config.headers['Authorization'] = `Bearer ${token}`
}
return config
},
(error) => Promise.reject(error), (error) => Promise.reject(error),
) )
@@ -30,12 +57,9 @@ service.interceptors.response.use(
const res = response.data as ApiResult const res = response.data as ApiResult
// 二进制流等非 JSON 响应直接返回 // 二进制流等非 JSON 响应直接返回
if (response.config.responseType === 'blob' || response.config.responseType === 'arraybuffer') { if (response.config.responseType === 'blob' || response.config.responseType === 'arraybuffer') {
touchSessionActivity()
return response return response
} }
if (res.code === 0) { if (res.code === 0) {
// 生成进度轮询也属于用户正在使用系统,避免长任务结束后被误判为会话过期。
touchSessionActivity()
return res.data return res.data
} }
// 业务错误 // 业务错误

View File

@@ -0,0 +1,79 @@
<script setup lang="ts">
import { computed, ref, watch } from 'vue'
import { ElMessage } from 'element-plus'
import { getAcl, setAcl, type AclEntry } from '@/api/modules/acl'
const props = defineProps<{
modelValue: boolean
resourceType: string
resourceId: string
}>()
const emit = defineEmits<{ 'update:modelValue': [boolean] }>()
const visible = computed({
get: () => props.modelValue,
set: (v) => emit('update:modelValue', v),
})
const entries = ref<AclEntry[]>([])
const loading = ref(false)
const ALL_PERMS = ['read', 'write', 'execute', 'download', 'delete', 'share']
async function load() {
loading.value = true
try {
entries.value = await getAcl(props.resourceType, props.resourceId)
} finally {
loading.value = false
}
}
watch(visible, (v) => { if (v) load() })
function addEntry() {
entries.value.push({ subject_type: 'user', subject_id: '', permissions: [] })
}
function removeEntry(idx: number) {
entries.value.splice(idx, 1)
}
async function save() {
await setAcl(props.resourceType, props.resourceId, entries.value)
ElMessage.success('ACL 已保存')
visible.value = false
}
</script>
<template>
<el-dialog v-model="visible" title="资源授权 (ACL)" width="640px">
<div v-loading="loading">
<el-button type="primary" size="small" @click="addEntry">添加授权项</el-button>
<div v-for="(entry, idx) in entries" :key="idx" class="acl-row">
<el-select v-model="entry.subject_type" style="width: 140px">
<el-option label="用户" value="user" />
<el-option label="项目角色" value="project_role" />
</el-select>
<el-input v-model="entry.subject_id" placeholder="subject ID" style="width: 200px" />
<el-checkbox-group v-model="entry.permissions">
<el-checkbox v-for="p in ALL_PERMS" :key="p" :value="p">{{ p }}</el-checkbox>
</el-checkbox-group>
<el-button link type="danger" @click="removeEntry(idx)">删除</el-button>
</div>
<el-empty v-if="entries.length === 0" description="暂无授权" />
</div>
<template #footer>
<el-button @click="visible = false">取消</el-button>
<el-button type="primary" @click="save">保存</el-button>
</template>
</el-dialog>
</template>
<style scoped lang="scss">
.acl-row {
display: flex;
align-items: center;
gap: 12px;
margin-top: 12px;
flex-wrap: wrap;
}
</style>

View File

@@ -74,6 +74,16 @@ const menuGroups: MenuGroup[] = [
{ key: 'compute', label: '算力节点', icon: 'fa-microchip', to: '/compute', permission: 'compute' }, { key: 'compute', label: '算力节点', icon: 'fa-microchip', to: '/compute', permission: 'compute' },
], ],
}, },
{
title: '平台治理',
items: [
{ key: 'tenants', label: '租户管理', icon: 'fa-building', to: '/tenants', permission: 'user-settings' },
{ key: 'projects', label: '项目空间', icon: 'fa-folder', to: '/projects', permission: 'user-settings' },
{ key: 'audit-logs', label: '审计日志', icon: 'fa-history', to: '/audit-logs', permission: 'user-settings' },
{ key: 'approval-templates', label: '审批模板', icon: 'fa-list-alt', to: '/approval-templates', permission: 'user-settings' },
{ key: 'approval-instances', label: '审批中心', icon: 'fa-check-square', to: '/approval-instances', permission: 'user-settings' },
],
},
{ {
title: '系统设置', title: '系统设置',
items: [ items: [

View File

@@ -1,7 +1,11 @@
import { createRouter, createWebHistory, type RouteRecordRaw } from 'vue-router' import { createRouter, createWebHistory, type RouteRecordRaw } from 'vue-router'
import { ref } from 'vue'
import { useAuthStore } from '@/stores/auth' import { useAuthStore } from '@/stores/auth'
import type { PermissionCode } from '@/types' import type { PermissionCode } from '@/types'
/** 路由切换时的全局加载态,供 App.vue 显示全屏转圈遮罩,消除懒加载时的空白卡顿感 */
export const routeLoading = ref(false)
const routes: RouteRecordRaw[] = [ const routes: RouteRecordRaw[] = [
{ {
path: '/login', path: '/login',
@@ -27,6 +31,49 @@ const routes: RouteRecordRaw[] = [
component: () => import('@/views/dashboard/DashboardView.vue'), component: () => import('@/views/dashboard/DashboardView.vue'),
meta: { title: '服务看板' }, meta: { title: '服务看板' },
}, },
// 平台治理
{
path: 'tenants',
name: 'tenants',
component: () => import('@/views/tenants/TenantListView.vue'),
meta: { title: '租户管理', permission: 'user-settings' },
},
{
path: 'tenants/:id',
name: 'tenant-detail',
component: () => import('@/views/tenants/TenantDetailView.vue'),
meta: { title: '租户详情', permission: 'user-settings' },
},
{
path: 'projects',
name: 'projects',
component: () => import('@/views/projects/ProjectListView.vue'),
meta: { title: '项目空间', permission: 'user-settings' },
},
{
path: 'projects/:id',
name: 'project-detail',
component: () => import('@/views/projects/ProjectDetailView.vue'),
meta: { title: '项目详情', permission: 'user-settings' },
},
{
path: 'audit-logs',
name: 'audit-logs',
component: () => import('@/views/audit/AuditLogView.vue'),
meta: { title: '审计日志', permission: 'user-settings' },
},
{
path: 'approval-templates',
name: 'approval-templates',
component: () => import('@/views/approvals/ApprovalTemplateView.vue'),
meta: { title: '审批模板', permission: 'user-settings' },
},
{
path: 'approval-instances',
name: 'approval-instances',
component: () => import('@/views/approvals/ApprovalInstanceView.vue'),
meta: { title: '审批中心', permission: 'user-settings' },
},
// 模型调优 // 模型调优
{ {
path: 'fine-tune', path: 'fine-tune',
@@ -299,6 +346,11 @@ const permissionBySegment: Record<string, PermissionCode> = {
hardware: 'hardware', hardware: 'hardware',
logs: 'logs', logs: 'logs',
'user-settings': 'user-settings', 'user-settings': 'user-settings',
tenants: 'user-settings',
projects: 'user-settings',
'audit-logs': 'user-settings',
'approval-templates': 'user-settings',
'approval-instances': 'user-settings',
} }
function requiredPermission(path: string, explicit?: unknown) { function requiredPermission(path: string, explicit?: unknown) {
@@ -307,10 +359,11 @@ function requiredPermission(path: string, explicit?: unknown) {
return permissionBySegment[segment] return permissionBySegment[segment]
} }
// 全局守卫:登录校验 + 会话超时 // 全局守卫:登录校验
// 离开页面超时由 App.vue 的 visibilitychange 监听接管
router.beforeEach((to, _from, next) => { router.beforeEach((to, _from, next) => {
if (!to.meta.public) routeLoading.value = true
const auth = useAuthStore() const auth = useAuthStore()
auth.syncSession()
document.title = to.meta.title ? `${to.meta.title} - 远光软件微调平台` : '远光软件微调平台' document.title = to.meta.title ? `${to.meta.title} - 远光软件微调平台` : '远光软件微调平台'
if (to.meta.public) { if (to.meta.public) {
@@ -324,6 +377,7 @@ router.beforeEach((to, _from, next) => {
} }
if (!auth.isLoggedIn) { if (!auth.isLoggedIn) {
auth.logout()
next({ name: 'login' }) next({ name: 'login' })
return return
} }
@@ -336,9 +390,11 @@ router.beforeEach((to, _from, next) => {
} }
} }
// 续期会话
auth.refresh()
next() next()
}) })
router.afterEach(() => {
routeLoading.value = false
})
export default router export default router

View File

@@ -1,15 +1,7 @@
import { defineStore } from 'pinia' import { defineStore } from 'pinia'
import { ref, computed } from 'vue' import { ref, computed } from 'vue'
import { login as loginApi } from '@/api/modules/system' import { login as loginApi } from '@/api/modules/system'
import { SESSION_TIMEOUT } from '@/constants'
import type { PermissionCode, SystemUser } from '@/types' import type { PermissionCode, SystemUser } from '@/types'
import {
clearSessionActivity,
sessionActivityTime,
startSessionActivity,
syncSessionActivity,
touchSessionActivity,
} from '@/utils/sessionActivity'
const USER_STORAGE_KEY = 'currentUser' const USER_STORAGE_KEY = 'currentUser'
@@ -56,7 +48,8 @@ function restoreUser(): SystemUser | null {
/** /**
* 认证 store * 认证 store
* 沿用原项目 localStorage 的登录时间戳 + 5 分钟会话超时机制 * 登录态管理:有 currentUser 即视为已登录。
* 离开页面超时由 App.vue 的 visibilitychange 监听接管。
*/ */
export const useAuthStore = defineStore('auth', () => { export const useAuthStore = defineStore('auth', () => {
const currentUser = ref<SystemUser | null>(restoreUser()) const currentUser = ref<SystemUser | null>(restoreUser())
@@ -67,18 +60,13 @@ export const useAuthStore = defineStore('auth', () => {
if (currentUser.value?.role === 'operator') return '操作员' if (currentUser.value?.role === 'operator') return '操作员'
return '观察员' return '观察员'
}) })
const loginTime = sessionActivityTime
const isLoggedIn = computed(() => { const isLoggedIn = computed(() => currentUser.value !== null)
if (!loginTime.value) return false
return Date.now() - loginTime.value < SESSION_TIMEOUT
})
/** 登录 */ /** 登录 */
async function login(user: string, password: string) { async function login(user: string, password: string) {
const response = await loginApi(user, password) const response = await loginApi(user, password)
currentUser.value = response.user currentUser.value = response.user
startSessionActivity()
localStorage.setItem('username', response.user.username) localStorage.setItem('username', response.user.username)
localStorage.setItem(USER_STORAGE_KEY, JSON.stringify(response.user)) localStorage.setItem(USER_STORAGE_KEY, JSON.stringify(response.user))
} }
@@ -89,20 +77,9 @@ export const useAuthStore = defineStore('auth', () => {
return currentUser.value?.permissions.includes(permission) ?? false return currentUser.value?.permissions.includes(permission) ?? false
} }
/** 续期会话(活跃时刷新) */
function refresh() {
if (currentUser.value) touchSessionActivity()
}
/** 在路由判断前吸收其他标签页写入的最后活跃时间。 */
function syncSession() {
syncSessionActivity()
}
/** 退出 */ /** 退出 */
function logout() { function logout() {
currentUser.value = null currentUser.value = null
clearSessionActivity()
localStorage.removeItem('username') localStorage.removeItem('username')
localStorage.removeItem(USER_STORAGE_KEY) localStorage.removeItem(USER_STORAGE_KEY)
} }
@@ -112,12 +89,9 @@ export const useAuthStore = defineStore('auth', () => {
username, username,
displayName, displayName,
roleLabel, roleLabel,
loginTime,
isLoggedIn, isLoggedIn,
hasPermission, hasPermission,
login, login,
refresh,
syncSession,
logout, logout,
} }
}) })

View File

@@ -99,6 +99,20 @@ export interface DataProcessRegenerateResult {
published_outputs_preserved: boolean published_outputs_preserved: boolean
} }
export interface DataProcessRepeatPayload {
expected_updated_at: string
request_id: string
}
export interface DataProcessRepeatResult {
task: DataProcessTask
source_task_id: string
created: boolean
copied_source_file_count: number
copied_preview_count: number
progress: DataProcessProgress
}
export type DataProcessTaskUpdatePayload = Partial<DataProcessTaskCreatePayload> export type DataProcessTaskUpdatePayload = Partial<DataProcessTaskCreatePayload>
export interface DataProcessSourceFile { export interface DataProcessSourceFile {
@@ -232,6 +246,22 @@ export interface DataProcessPreviewItem {
updated_at?: string updated_at?: string
} }
export type DataProcessSourceLocatorKind = 'json' | 'jsonl' | 'csv' | 'xlsx'
export interface DataProcessSourceLocator {
kind: DataProcessSourceLocatorKind
record_index?: number | null
start_line?: number | null
end_line?: number | null
source_start?: number | null
source_end?: number | null
json_pointer?: string | null
sheet_index?: number | null
sheet_name?: string | null
row_number?: number | null
sheet_record_index?: number | null
}
export interface DataProcessPreviewBuildPayload { export interface DataProcessPreviewBuildPayload {
replace_existing?: true replace_existing?: true
source_file_ids?: Array<string | number> source_file_ids?: Array<string | number>
@@ -369,6 +399,9 @@ export interface DataProcessQualityScore {
is_valid?: boolean is_valid?: boolean
flags?: string[] flags?: string[]
fingerprint?: string fingerprint?: string
source_pages?: number[]
heading_path?: string[]
source_locator?: DataProcessSourceLocator
[key: string]: unknown [key: string]: unknown
} }

View File

@@ -8,7 +8,7 @@ function storedActivityTime() {
/** /**
* 会话按“最后活跃时间”计算,而不是从首次登录起固定倒计时。 * 会话按“最后活跃时间”计算,而不是从首次登录起固定倒计时。
* 该 ref 被认证 store 与请求层共享,确保 API 活动可以立即影响路由守卫 * 该 ref 被认证 store 与路由守卫共享,确保真实用户活动可以立即影响超时判断
*/ */
export const sessionActivityTime = ref(storedActivityTime()) export const sessionActivityTime = ref(storedActivityTime())
@@ -30,3 +30,45 @@ export function clearSessionActivity() {
sessionActivityTime.value = 0 sessionActivityTime.value = 0
localStorage.removeItem(LOGIN_TIME_STORAGE_KEY) localStorage.removeItem(LOGIN_TIME_STORAGE_KEY)
} }
/**
* 仅在用户真实活跃时续期会话:
* - 鼠标移动 / 键盘 / 点击 / 触摸(说明用户正在操作)
* - 标签页切回可见(说明用户回到界面)
* 页面后台轮询接口、切走标签页不会续期,从而“无操作”或“不在当前界面”
* 超过空闲时长才会被判定为会话过期并跳回登录。
*/
let userActivityBound = false
let lastTouch = 0
const ACTIVITY_THROTTLE = 5000 // 5s 内最多续期一次,避免 mousemove 过于频繁
const activityEvents = ['mousemove', 'mousedown', 'keydown', 'click', 'touchstart'] as const
function handleUserActivity() {
const now = Date.now()
if (now - lastTouch < ACTIVITY_THROTTLE) return
lastTouch = now
touchSessionActivity()
}
function handleVisibility() {
if (!document.hidden) {
touchSessionActivity()
}
}
export function bindUserActivityListeners() {
if (userActivityBound) return
userActivityBound = true
activityEvents.forEach((evt) =>
window.addEventListener(evt, handleUserActivity, { passive: true })
)
document.addEventListener('visibilitychange', handleVisibility)
}
export function unbindUserActivityListeners() {
if (!userActivityBound) return
userActivityBound = false
activityEvents.forEach((evt) => window.removeEventListener(evt, handleUserActivity))
document.removeEventListener('visibilitychange', handleVisibility)
}

View File

@@ -0,0 +1,128 @@
<script setup lang="ts">
import { onMounted, reactive, ref } from 'vue'
import { ElMessage } from 'element-plus'
import DataTablePage from '@/components/DataTablePage.vue'
import { getApprovalInstances, decideApproval, type ApprovalInstance } from '@/api/modules/approval'
import { getUsers, type SystemUser } from '@/api/modules/system'
const loading = ref(false)
const instances = ref<ApprovalInstance[]>([])
const users = ref<SystemUser[]>([])
const statusFilter = ref<string | undefined>(undefined)
const showDecide = ref(false)
const current = ref<ApprovalInstance | null>(null)
const decision = ref({ step_index: 0, approver_id: '', approved: true, comment: '' })
const statusOptions = [
{ label: '待审批', value: 'pending' },
{ label: '已通过', value: 'approved' },
{ label: '已拒绝', value: 'rejected' },
]
async function load() {
loading.value = true
try {
instances.value = await getApprovalInstances(statusFilter.value)
} finally {
loading.value = false
}
}
async function loadUsers() {
try {
users.value = await getUsers()
} catch {
users.value = []
}
}
function userName(id?: string) {
if (!id) return '—'
return users.value.find((u) => u.id === id)?.username || id
}
function openDecide(inst: ApprovalInstance) {
current.value = inst
const step = inst.steps.find((s) => s.status === 'pending')
decision.value = { step_index: step ? step.step_index : 0, approver_id: '', approved: true, comment: '' }
showDecide.value = true
}
async function submitDecision() {
if (!current.value) return
if (!decision.value.approver_id) {
ElMessage.warning('请选择审批人')
return
}
await decideApproval(current.value.id, decision.value.step_index, {
approver_id: decision.value.approver_id,
approved: decision.value.approved,
comment: decision.value.comment,
})
ElMessage.success('审批已提交')
showDecide.value = false
load()
}
onMounted(() => {
loadUsers()
load()
})
</script>
<template>
<div class="page">
<DataTablePage title="审批实例" :data="instances" :loading="loading" searchable search-fields="resource_type,resource_id">
<template #toolbar-extra>
<el-select v-model="statusFilter" placeholder="状态" clearable style="width: 140px" @change="load">
<el-option v-for="s in statusOptions" :key="s.value" :label="s.label" :value="s.value" />
</el-select>
</template>
<template #columns>
<el-table-column prop="resource_type" label="资源类型" min-width="120" />
<el-table-column prop="resource_id" label="资源 ID" min-width="160" show-overflow-tooltip />
<el-table-column prop="applicant_id" label="申请人" min-width="120">
<template #default="{ row }">{{ userName(row.applicant_id) }}</template>
</el-table-column>
<el-table-column prop="status" label="状态" min-width="100" />
<el-table-column prop="current_step" label="当前步骤" min-width="100" />
<el-table-column prop="create_time" label="创建时间" min-width="180" />
</template>
<template #actions="{ row }">
<el-button v-if="row.status === 'pending'" link type="primary" @click="openDecide(row)">审批</el-button>
</template>
</DataTablePage>
<el-dialog v-model="showDecide" title="审批决策" width="480px">
<el-form label-width="80px" v-if="current">
<el-form-item label="实例">
{{ current.resource_type }} / {{ current.resource_id }}
</el-form-item>
<el-form-item label="步骤">
{{ decision.step_index + 1 }}
</el-form-item>
<el-form-item label="审批人" required>
<el-select v-model="decision.approver_id" filterable style="width: 100%">
<el-option v-for="u in users" :key="u.id" :label="u.username" :value="u.id" />
</el-select>
</el-form-item>
<el-form-item label="结果">
<el-radio-group v-model="decision.approved">
<el-radio :value="true">通过</el-radio>
<el-radio :value="false">拒绝</el-radio>
</el-radio-group>
</el-form-item>
<el-form-item label="意见">
<el-input v-model="decision.comment" type="textarea" :rows="3" />
</el-form-item>
</el-form>
<template #footer>
<el-button @click="showDecide = false">取消</el-button>
<el-button type="primary" @click="submitDecision">提交</el-button>
</template>
</el-dialog>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
</style>

View File

@@ -0,0 +1,77 @@
<script setup lang="ts">
import { onMounted, ref } from 'vue'
import { ElMessage } from 'element-plus'
import { Plus } from '@element-plus/icons-vue'
import DataTablePage from '@/components/DataTablePage.vue'
import { createApprovalTemplate, getApprovalTemplates, type ApprovalTemplate } from '@/api/modules/approval'
const loading = ref(false)
const templates = ref<ApprovalTemplate[]>([])
const showCreate = ref(false)
const form = ref({ name: '', stepsText: '[]' })
async function load() {
loading.value = true
try {
templates.value = await getApprovalTemplates()
} finally {
loading.value = false
}
}
async function submitCreate() {
if (!form.value.name) {
ElMessage.warning('请填写模板名称')
return
}
let steps: unknown[] = []
try {
steps = JSON.parse(form.value.stepsText || '[]')
} catch {
ElMessage.error('步骤需为合法 JSON 数组')
return
}
await createApprovalTemplate({ name: form.value.name, steps: steps as any })
ElMessage.success('模板创建成功')
showCreate.value = false
form.value = { name: '', stepsText: '[]' }
load()
}
onMounted(load)
</script>
<template>
<div class="page">
<DataTablePage title="审批模板" :data="templates" :loading="loading">
<template #toolbar-extra>
<el-button type="primary" :icon="Plus" @click="showCreate = true">新建模板</el-button>
</template>
<template #columns>
<el-table-column prop="name" label="模板名" min-width="160" />
<el-table-column label="步骤数" min-width="100">
<template #default="{ row }">{{ (row.steps || []).length }}</template>
</el-table-column>
<el-table-column prop="create_time" label="创建时间" min-width="180" />
</template>
</DataTablePage>
<el-dialog v-model="showCreate" title="新建审批模板" width="560px">
<el-form label-width="90px">
<el-form-item label="名称" required>
<el-input v-model="form.name" placeholder="模板名" />
</el-form-item>
<el-form-item label="步骤 JSON">
<el-input v-model="form.stepsText" type="textarea" :rows="5" placeholder='[{"approver_id":"u1"},{"approver_id":"u2"}]' />
</el-form-item>
</el-form>
<template #footer>
<el-button @click="showCreate = false">取消</el-button>
<el-button type="primary" @click="submitCreate">创建</el-button>
</template>
</el-dialog>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
</style>

View File

@@ -0,0 +1,125 @@
<script setup lang="ts">
import { onMounted, reactive, ref } from 'vue'
import { ElMessage } from 'element-plus'
import { getAuditLogs, exportAuditLogs, type AuditLog, type AuditQuery } from '@/api/modules/audit'
const loading = ref(false)
const logs = ref<AuditLog[]>([])
const total = ref(0)
const query = reactive<AuditQuery>({
tenant_id: '',
project_id: '',
actor_id: '',
action: '',
target_type: '',
start_time: '',
end_time: '',
limit: 50,
offset: 0,
})
// 时间范围el-date-picker 双向绑定数组 [start, end]
const timeRange = ref<[string, string] | null>(null)
function applyTimeRange() {
if (timeRange.value && timeRange.value.length === 2) {
query.start_time = timeRange.value[0]
query.end_time = timeRange.value[1]
} else {
query.start_time = ''
query.end_time = ''
}
}
async function load() {
loading.value = true
try {
const res = await getAuditLogs({ ...query })
logs.value = res.items
total.value = res.total
} finally {
loading.value = false
}
}
async function handleExport() {
try {
const blob = await exportAuditLogs({ ...query, limit: 10000, offset: 0 })
const url = URL.createObjectURL(blob)
const a = document.createElement('a')
a.href = url
a.download = `audit_logs_${Date.now()}.csv`
a.click()
URL.revokeObjectURL(url)
} catch {
ElMessage.error('导出失败')
}
}
onMounted(load)
</script>
<template>
<div class="page">
<div class="page-header">
<h2 class="page-title">审计日志</h2>
<el-button @click="handleExport">导出 CSV</el-button>
</div>
<el-card class="filter-card">
<el-form :inline="true">
<el-form-item label="租户">
<el-input v-model="query.tenant_id" placeholder="tenant_id" clearable />
</el-form-item>
<el-form-item label="项目">
<el-input v-model="query.project_id" placeholder="project_id" clearable />
</el-form-item>
<el-form-item label="操作人">
<el-input v-model="query.actor_id" placeholder="actor_id" clearable />
</el-form-item>
<el-form-item label="动作">
<el-input v-model="query.action" placeholder="action" clearable />
</el-form-item>
<el-form-item label="目标类型">
<el-input v-model="query.target_type" placeholder="target_type" clearable />
</el-form-item>
<el-form-item label="时间范围">
<el-date-picker
v-model="timeRange"
type="datetimerange"
value-format="YYYY-MM-DDTHH:mm:ss"
range-separator=""
start-placeholder="开始时间"
end-placeholder="结束时间"
clearable
style="width: 360px"
@change="applyTimeRange"
/>
</el-form-item>
<el-form-item>
<el-button type="primary" @click="load">查询</el-button>
</el-form-item>
</el-form>
</el-card>
<el-table :data="logs" v-loading="loading" border stripe class="log-table">
<el-table-column prop="time" label="时间" min-width="180" />
<el-table-column prop="tenant_id" label="租户" min-width="120" />
<el-table-column prop="project_id" label="项目" min-width="120" />
<el-table-column prop="actor_id" label="操作人" min-width="120" />
<el-table-column prop="action" label="动作" min-width="140" />
<el-table-column prop="target_type" label="目标类型" min-width="120" />
<el-table-column prop="target_id" label="目标 ID" min-width="140" show-overflow-tooltip />
<el-table-column prop="detail" label="详情" min-width="200" show-overflow-tooltip />
<el-table-column prop="client_ip" label="IP" min-width="120" />
</el-table>
<div class="pager"> {{ total }} </div>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
.page-header { display: flex; align-items: center; justify-content: space-between; margin-bottom: 16px; }
.page-title { margin: 0; font-size: 18px; }
.filter-card { margin-bottom: 16px; }
.log-table { margin-top: 8px; }
.pager { margin-top: 12px; text-align: right; color: #909399; }
</style>

View File

@@ -1,9 +1,10 @@
<script setup lang="ts"> <script setup lang="ts">
import { computed, ref } from 'vue' import { computed, onMounted, ref } from 'vue'
import { useRouter } from 'vue-router' import { useRouter } from 'vue-router'
import VChart from 'vue-echarts' import VChart from 'vue-echarts'
import '@/plugins/echarts' import '@/plugins/echarts'
import type { EChartsOption } from 'echarts' import type { EChartsOption } from 'echarts'
import { getDashboardStats } from '@/api/modules/dashboard'
type ServiceState = 'normal' | 'busy' | 'error' type ServiceState = 'normal' | 'busy' | 'error'
type TaskState = 'running' | 'pending' | 'completed' | 'failed' type TaskState = 'running' | 'pending' | 'completed' | 'failed'
@@ -16,7 +17,7 @@ interface ServiceStatus {
} }
interface DashboardTask { interface DashboardTask {
id: number id: string
name: string name: string
state: TaskState state: TaskState
trainType: string trainType: string
@@ -43,59 +44,37 @@ interface RecentLoginUser {
const router = useRouter() const router = useRouter()
const period = ref('7d') const period = ref('7d')
const serviceStatuses: ServiceStatus[] = [ const onlineServices = ref(0)
{ name: '模型推理', icon: 'fa-cube', state: 'normal', instances: '6 / 6' }, const runningTasks = ref(0)
{ name: '模型微调', icon: 'fa-sliders', state: 'busy', instances: '4 / 6' }, const pendingAlerts = ref(0)
{ name: '模型评测', icon: 'fa-bar-chart', state: 'normal', instances: '3 / 3' },
{ name: '数据处理', icon: 'fa-filter', state: 'error', instances: '1 / 3' },
]
const trainingTasks: DashboardTask[] = [ const serviceStatuses = ref<ServiceStatus[]>([])
{ const trainingTasks = ref<DashboardTask[]>([])
id: 103942, const loginDurationStats = ref<LoginDurationStat[]>([])
name: 'finance-sft-003', const recentLoginUsers = ref<RecentLoginUser[]>([])
state: 'running', const training7d = ref<{ date: string; train: number; gpu: number; accuracy: number | null }[]>([])
trainType: 'SFT',
trainMethod: 'LoRA', const onlineServicesHint = computed(() => {
baseModel: 'Qwen2.5-7B-Instruct', if (onlineServices.value === 0) return '暂无在线服务'
progress: 68, const abnormal = serviceStatuses.value.filter(
accuracy: 89.2, (s) => s.state === 'busy' || s.state === 'error'
startedAt: '今天 09:18', ).length
}, return abnormal > 0 ? `${abnormal} 个异常` : '全部在线'
{ })
id: 593021, const operationDistribution = ref<{ name: string; value: number }[]>([])
name: 'legal-eval-008',
state: 'pending', const serviceIcon: Record<string, string> = {
trainType: 'DPO', '模型推理': 'fa-cube',
trainMethod: 'LoRA', '模型微调': 'fa-sliders',
baseModel: 'Qwen2.5-7B-Instruct', '模型评测': 'fa-bar-chart',
progress: 0, '数据处理': 'fa-filter',
accuracy: null, }
startedAt: '今天 08:55', const roleLabel: Record<string, string> = {
}, admin: '超级管理员',
{ operator: '操作员',
id: 849301, observer: '观察员',
name: 'medical-cpt-002', guest: '访客',
state: 'completed', }
trainType: 'CPT',
trainMethod: 'Full',
baseModel: 'Qwen2.5-14B-Instruct',
progress: 100,
accuracy: 91.6,
startedAt: '07/10 16:20',
},
{
id: 201948,
name: 'finance-sft-002',
state: 'failed',
trainType: 'SFT',
trainMethod: 'LoRA',
baseModel: 'Qwen2.5-7B-Instruct',
progress: 42,
accuracy: null,
startedAt: '07/10 11:08',
},
]
const serviceStateMeta: Record<ServiceState, { label: string; className: string }> = { const serviceStateMeta: Record<ServiceState, { label: string; className: string }> = {
normal: { label: '正常', className: 'is-normal' }, normal: { label: '正常', className: 'is-normal' },
@@ -140,7 +119,7 @@ const chartOption = computed<EChartsOption>(() => ({
}, },
xAxis: { xAxis: {
type: 'category', type: 'category',
data: ['07/05', '07/06', '07/07', '07/08', '07/09', '07/10', '07/11\n今天'], data: training7d.value.map((d) => d.date),
axisLine: { lineStyle: { color: '#e2e8f0' } }, axisLine: { lineStyle: { color: '#e2e8f0' } },
axisTick: { show: false }, axisTick: { show: false },
axisLabel: { color: '#64748b', fontSize: 11, lineHeight: 16, margin: 12 }, axisLabel: { color: '#64748b', fontSize: 11, lineHeight: 16, margin: 12 },
@@ -175,7 +154,7 @@ const chartOption = computed<EChartsOption>(() => ({
{ {
name: '训练次数(次)', name: '训练次数(次)',
type: 'bar', type: 'bar',
data: [8, 12, 10, 15, 13, 18, 11], data: training7d.value.map((d) => d.train),
barMaxWidth: 16, barMaxWidth: 16,
itemStyle: { borderRadius: [3, 3, 0, 0] }, itemStyle: { borderRadius: [3, 3, 0, 0] },
label: { show: true, position: 'top', color: '#64748b', fontSize: 10 }, label: { show: true, position: 'top', color: '#64748b', fontSize: 10 },
@@ -183,7 +162,7 @@ const chartOption = computed<EChartsOption>(() => ({
{ {
name: 'GPU 使用数(个)', name: 'GPU 使用数(个)',
type: 'bar', type: 'bar',
data: [3, 4, 4, 6, 5, 7, 5], data: training7d.value.map((d) => d.gpu),
barMaxWidth: 16, barMaxWidth: 16,
itemStyle: { borderRadius: [3, 3, 0, 0] }, itemStyle: { borderRadius: [3, 3, 0, 0] },
label: { show: true, position: 'top', color: '#64748b', fontSize: 10 }, label: { show: true, position: 'top', color: '#64748b', fontSize: 10 },
@@ -192,7 +171,7 @@ const chartOption = computed<EChartsOption>(() => ({
name: '平均准确率(%', name: '平均准确率(%',
type: 'bar', type: 'bar',
yAxisIndex: 1, yAxisIndex: 1,
data: [82, 85, 84, 88, 87, 91, 89], data: training7d.value.map((d) => d.accuracy ?? null),
barMaxWidth: 16, barMaxWidth: 16,
itemStyle: { borderRadius: [3, 3, 0, 0] }, itemStyle: { borderRadius: [3, 3, 0, 0] },
label: { show: true, position: 'top', color: '#d97706', fontSize: 10 }, label: { show: true, position: 'top', color: '#d97706', fontSize: 10 },
@@ -200,58 +179,61 @@ const chartOption = computed<EChartsOption>(() => ({
], ],
})) }))
const operationChartOption = computed<EChartsOption>(() => ({ // 模块固定配色,保证每个模块颜色不同
const OPERATION_COLORS = ['#4f46e5', '#10b981', '#f59e0b', '#3b82f6', '#ec4899', '#8b5cf6', '#ef4444', '#14b8a6']
const operationChartOption = computed<EChartsOption>(() => {
const items = operationDistribution.value
const total = items.reduce((s, d) => s + (d.value || 0), 0)
// 完全没有操作数据时,用等分灰色占位扇区,保证 6 个模块都可见
const data =
total > 0
? items.map((d) => ({ value: d.value || 0, name: d.name }))
: items.map((d) => ({ value: 1, name: d.name, itemStyle: { color: '#e2e8f0' } }))
return {
animationDuration: 500, animationDuration: 500,
tooltip: { trigger: 'item' }, tooltip: { trigger: 'item', formatter: '{b}: {c} ({d}%)' },
color: ['#4f46e5', '#10b981', '#f59e0b', '#3b82f6', '#ec4899'], color: OPERATION_COLORS,
legend: {
type: 'scroll',
bottom: 0,
textStyle: { color: '#64748b', fontSize: 11 },
itemWidth: 10,
itemHeight: 10,
},
series: [ series: [
{ {
name: '操作分类', name: '操作分类',
type: 'pie', type: 'pie',
radius: ['40%', '64%'], radius: ['38%', '60%'],
center: ['50%', '50%'], center: ['50%', '42%'],
avoidLabelOverlap: true, avoidLabelOverlap: true,
itemStyle: { itemStyle: {
borderRadius: 6, borderRadius: 6,
borderColor: '#fff', borderColor: '#fff',
borderWidth: 2 borderWidth: 2,
}, },
label: { label: {
show: true, show: true,
position: 'outside', position: 'outside',
formatter: '{b}', formatter: '{b}\n{d}%',
color: '#475569', color: '#475569',
fontSize: 11, fontSize: 11,
lineHeight: 16, lineHeight: 15,
width: 70,
overflow: 'truncate',
}, },
emphasis: { emphasis: {
label: { show: true, fontSize: 12, fontWeight: 'bold', color: '#1e293b' } label: { show: true, fontSize: 12, fontWeight: 'bold', color: '#1e293b' },
}, },
labelLine: { labelLine: {
show: true, show: true,
length: 10, length: 8,
length2: 8, length2: 8,
lineStyle: { color: '#94a3b8', width: 1 }, lineStyle: { color: '#94a3b8', width: 1 },
}, },
data: [ data,
{ value: 1048, name: '模型训练' }, },
{ value: 735, name: '数据处理' }, ],
{ value: 580, name: '模型评测' },
{ value: 484, name: '模型推理' },
{ value: 300, name: '系统设置' }
]
} }
] })
}))
const loginDurationStats: LoginDurationStat[] = [
{ id: 1, username: 'admin', duration: 124 },
{ id: 2, username: 'zhangsan', duration: 86 },
{ id: 3, username: 'lisi', duration: 42 },
{ id: 4, username: 'wangwu', duration: 18 },
]
const loginDurationChartOption = computed<EChartsOption>(() => ({ const loginDurationChartOption = computed<EChartsOption>(() => ({
animationDuration: 500, animationDuration: 500,
@@ -263,7 +245,7 @@ const loginDurationChartOption = computed<EChartsOption>(() => ({
}, },
xAxis: { xAxis: {
type: 'value', type: 'value',
max: Math.ceil(Math.max(...loginDurationStats.map((user) => user.duration)) * 1.15 / 10) * 10, max: Math.max(10, Math.ceil(Math.max(...loginDurationStats.value.map((user) => user.duration), 0) * 1.15 / 10) * 10),
splitNumber: 4, splitNumber: 4,
axisLabel: { color: '#94a3b8', fontSize: 11, formatter: '{value}h' }, axisLabel: { color: '#94a3b8', fontSize: 11, formatter: '{value}h' },
axisLine: { show: false }, axisLine: { show: false },
@@ -273,7 +255,7 @@ const loginDurationChartOption = computed<EChartsOption>(() => ({
yAxis: { yAxis: {
type: 'category', type: 'category',
inverse: true, inverse: true,
data: loginDurationStats.map((user) => user.username), data: loginDurationStats.value.map((user) => user.username),
axisLabel: { color: '#475569', fontSize: 12 }, axisLabel: { color: '#475569', fontSize: 12 },
axisLine: { show: false }, axisLine: { show: false },
axisTick: { show: false }, axisTick: { show: false },
@@ -282,7 +264,7 @@ const loginDurationChartOption = computed<EChartsOption>(() => ({
{ {
name: '登录时长', name: '登录时长',
type: 'bar', type: 'bar',
data: loginDurationStats.map((user) => user.duration), data: loginDurationStats.value.map((user) => user.duration),
barMaxWidth: 18, barMaxWidth: 18,
barCategoryGap: '34%', barCategoryGap: '34%',
itemStyle: { color: '#4f46e5', borderRadius: [0, 4, 4, 0] }, itemStyle: { color: '#4f46e5', borderRadius: [0, 4, 4, 0] },
@@ -291,19 +273,51 @@ const loginDurationChartOption = computed<EChartsOption>(() => ({
], ],
})) }))
const recentLoginUsers: RecentLoginUser[] = [
{ id: 1, username: 'admin', role: '超级管理员', lastLogin: '10 分钟前' },
{ id: 2, username: 'zhangsan', role: '操作员', lastLogin: '2 小时前' },
{ id: 5, username: 'zhaoliu', role: '观察员', lastLogin: '5 小时前' },
{ id: 3, username: 'lisi', role: '操作员', lastLogin: '昨天 15:30' },
]
const roleTagType: Record<string, 'danger' | 'primary' | 'info'> = { const roleTagType: Record<string, 'danger' | 'primary' | 'info'> = {
'超级管理员': 'danger', '超级管理员': 'danger',
'操作员': 'primary', '操作员': 'primary',
'观察员': 'info', '观察员': 'info',
} }
async function loadStats() {
const stats = await getDashboardStats()
onlineServices.value = stats.online_services
runningTasks.value = stats.running_tasks
pendingAlerts.value = stats.pending_alerts
serviceStatuses.value = stats.service_status.map((s) => ({
name: s.type,
icon: serviceIcon[s.type] || 'fa-cube',
state: s.status as ServiceState,
instances: String(s.count),
}))
trainingTasks.value = stats.training_tasks.map((t) => ({
id: String(t.id),
name: t.name,
state: t.status as TaskState,
trainType: t.train_type,
trainMethod: t.train_method,
baseModel: t.base_model,
progress: t.progress,
accuracy: t.accuracy,
startedAt: t.started_at,
}))
loginDurationStats.value = stats.login_duration_rank.map((u, i) => ({
id: i + 1,
username: u.user,
duration: u.duration,
}))
recentLoginUsers.value = stats.recent_login_users.map((u, i) => ({
id: i + 1,
username: u.user,
role: roleLabel[u.role] || u.role,
lastLogin: u.last_login,
}))
training7d.value = stats.training_7d
operationDistribution.value = stats.operation_distribution
}
onMounted(loadStats)
function viewAllTasks() { function viewAllTasks() {
router.push('/fine-tune') router.push('/fine-tune')
} }
@@ -329,17 +343,17 @@ function viewTask(task: DashboardTask) {
<div class="overview-metrics"> <div class="overview-metrics">
<div class="overview-metric"> <div class="overview-metric">
<span>在线服务</span> <span>在线服务</span>
<strong>12</strong> <strong>{{ onlineServices }}</strong>
<small>全部在线</small> <small>{{ onlineServicesHint }}</small>
</div> </div>
<div class="overview-metric"> <div class="overview-metric">
<span>运行中任务</span> <span>运行中任务</span>
<strong>5</strong> <strong>{{ runningTasks }}</strong>
<small>较昨日 +1</small> <small>较昨日 +1</small>
</div> </div>
<div class="overview-metric is-alert"> <div class="overview-metric is-alert">
<span>待处理告警</span> <span>待处理告警</span>
<strong>2</strong> <strong>{{ pendingAlerts }}</strong>
<small>较昨日 -1</small> <small>较昨日 -1</small>
</div> </div>
</div> </div>
@@ -391,7 +405,10 @@ function viewTask(task: DashboardTask) {
<section class="stat-card" aria-labelledby="login-dur-title"> <section class="stat-card" aria-labelledby="login-dur-title">
<h2 id="login-dur-title" class="section-title">登录时长排行 (本月)</h2> <h2 id="login-dur-title" class="section-title">登录时长排行 (本月)</h2>
<div v-if="loginDurationStats.length" class="chart-container">
<VChart class="duration-chart" :option="loginDurationChartOption" autoresize /> <VChart class="duration-chart" :option="loginDurationChartOption" autoresize />
</div>
<div v-else class="empty-hint">暂无数据</div>
</section> </section>
<section class="stat-card" aria-labelledby="recent-login-title"> <section class="stat-card" aria-labelledby="recent-login-title">
@@ -515,6 +532,15 @@ function viewTask(task: DashboardTask) {
height: 224px; height: 224px;
} }
.empty-hint {
flex: 1 1 auto;
display: grid;
place-items: center;
min-height: 224px;
color: #94a3b8;
font-size: 13px;
}
.duration-chart { .duration-chart {
width: 100%; width: 100%;
height: 224px; height: 224px;

View File

@@ -10,7 +10,7 @@ import SourceUploadStep from './create/SourceUploadStep.vue'
import PreviewCompareStep from './create/PreviewCompareStep.vue' import PreviewCompareStep from './create/PreviewCompareStep.vue'
import GenerationStep from './create/GenerationStep.vue' import GenerationStep from './create/GenerationStep.vue'
import ResultEditorStep from './create/ResultEditorStep.vue' import ResultEditorStep from './create/ResultEditorStep.vue'
import { DEFAULT_SOURCE_TEXT, estimateTokenCount } from './create/previewModel' import { DEFAULT_SOURCE_TEXT, estimateTokenCount, isManualPreviewItem } from './create/previewModel'
import { import {
createDefaultStructuredOptions, createDefaultStructuredOptions,
createDefaultUnstructuredOptions, createDefaultUnstructuredOptions,
@@ -21,6 +21,7 @@ import { useDataProcessGeneration } from './create/useDataProcessGeneration'
import { useDataProcessPreviewBuild } from './create/useDataProcessPreviewBuild' import { useDataProcessPreviewBuild } from './create/useDataProcessPreviewBuild'
import { useDataProcessRegeneration } from './create/useDataProcessRegeneration' import { useDataProcessRegeneration } from './create/useDataProcessRegeneration'
import { import {
loadCanonicalSourceContent,
mapDataProcessSourceFile, mapDataProcessSourceFile,
useDataProcessSourceUpload, useDataProcessSourceUpload,
validateSourceFileSelection, validateSourceFileSelection,
@@ -33,7 +34,6 @@ import {
deleteDataProcessPreview, deleteDataProcessPreview,
deleteDataProcessSourceFile, deleteDataProcessSourceFile,
getDataProcessPreview, getDataProcessPreview,
getDataProcessSourceContent,
pullDataProcessExternalSource, pullDataProcessExternalSource,
testDataProcessExternalSource, testDataProcessExternalSource,
updateDataProcessPreview, updateDataProcessPreview,
@@ -116,7 +116,9 @@ const modelSubmitLoading = ref(false)
let allowLeave = false let allowLeave = false
const { const {
bulkRegeneration, bulkRegeneration,
canReturnFromGeneration,
generation, generation,
generationStarting,
regeneratingResultId, regeneratingResultId,
resultRegenerationBusy, resultRegenerationBusy,
results, results,
@@ -189,7 +191,6 @@ const primaryActionIcon = computed(() => {
if (currentStepId.value === 'generate' && generation.status !== 'success') return 'fa-play' if (currentStepId.value === 'generate' && generation.status !== 'success') return 'fa-play'
return 'fa-arrow-right' return 'fa-arrow-right'
}) })
const previousStepLabel = computed(() => currentStep.value > 0 const previousStepLabel = computed(() => currentStep.value > 0
? WIZARD_STEPS[currentStep.value - 1].title ? WIZARD_STEPS[currentStep.value - 1].title
: '') : '')
@@ -290,21 +291,26 @@ function externalPayload(): DataProcessExternalSourcePayload {
} }
function mapPreviewItem(item: DataProcessPreviewItem): PreviewItem { function mapPreviewItem(item: DataProcessPreviewItem): PreviewItem {
const sourceLocator = item.quality_score?.source_locator
return { return {
id: String(item.id), id: String(item.id),
sourceFileId: String(item.source_file_id), sourceFileId: String(item.source_file_id),
originalContent: item.original_content, originalContent: item.original_content,
editedContent: item.edited_content, editedContent: item.edited_content,
savedEditedContent: item.edited_content, savedEditedContent: item.edited_content,
sourceStart: item.source_start, sourceStart: item.source_start ?? sourceLocator?.source_start ?? null,
sourceEnd: item.source_end, sourceEnd: item.source_end ?? sourceLocator?.source_end ?? null,
sourceStartLine: item.source_start_line, sourceStartLine: item.source_start_line ?? sourceLocator?.start_line ?? null,
sourceEndLine: item.source_end_line, sourceEndLine: item.source_end_line ?? sourceLocator?.end_line ?? null,
tokenCount: item.token_count, tokenCount: item.token_count,
status: item.status, status: item.status,
sourcePages: Array.isArray(item.quality_score?.source_pages) sourcePages: Array.isArray(item.quality_score?.source_pages)
? item.quality_score.source_pages.filter((value): value is number => typeof value === 'number') ? item.quality_score.source_pages.filter((value): value is number => typeof value === 'number')
: [], : [],
sourceLocator,
headingPath: Array.isArray(item.quality_score?.heading_path)
? item.quality_score.heading_path.filter((value): value is string => typeof value === 'string')
: [],
updatedAt: item.updated_at, updatedAt: item.updated_at,
} }
} }
@@ -419,7 +425,6 @@ function handleFileChange(uploadFile: UploadFile) {
const localUid = `local-${uploadFile.uid}-${Date.now()}-${uploadedFiles.value.length}` const localUid = `local-${uploadFile.uid}-${Date.now()}-${uploadedFiles.value.length}`
uploadedFiles.value.push({ uploadedFiles.value.push({
uid: localUid, uid: localUid,
rawFile: raw,
name: raw.name, name: raw.name,
size: raw.size, size: raw.size,
count: 0, count: 0,
@@ -431,7 +436,7 @@ function handleFileChange(uploadFile: UploadFile) {
previewProgress: 0, previewProgress: 0,
}) })
dirty.value = true dirty.value = true
enqueueSourceUpload({ uid: localUid, file: raw, extension: validation.extension }) enqueueSourceUpload({ uid: localUid, file: raw })
} }
async function useSampleFile() { async function useSampleFile() {
@@ -488,11 +493,8 @@ async function handlePullData() {
const response = await pullDataProcessExternalSource(taskId.value, externalPayload()) const response = await pullDataProcessExternalSource(taskId.value, externalPayload())
const newFiles: UploadedDataFile[] = [] const newFiles: UploadedDataFile[] = []
for (const file of response.files) { for (const file of response.files) {
const source = await getDataProcessSourceContent(taskId.value, file.id, { const content = await loadCanonicalSourceContent(taskId.value, file.id)
start_line: 1, newFiles.push(mapDataProcessSourceFile(file, content))
line_count: 5000,
})
newFiles.push(mapDataProcessSourceFile(file, source.content))
} }
uploadedFiles.value.push(...newFiles) uploadedFiles.value.push(...newFiles)
externalConnected.value = true externalConnected.value = true
@@ -734,9 +736,14 @@ function selectPreviewItem(id: string) {
function updatePreviewContent(id: string, value: string) { function updatePreviewContent(id: string, value: string) {
const item = previewItems.value.find((entry) => entry.id === id) const item = previewItems.value.find((entry) => entry.id === id)
if (!item) return if (!item) return
const isManual = isManualPreviewItem(item)
item.editedContent = value item.editedContent = value
item.tokenCount = estimateTokenCount(value) item.tokenCount = estimateTokenCount(value)
item.status = value === item.originalContent ? 'original' : item.sourceStart == null ? 'manual' : 'modified' item.status = !value.trim()
? 'invalid'
: value === item.originalContent
? 'original'
: isManual ? 'manual' : 'modified'
resetDownstream() resetDownstream()
dirty.value = true dirty.value = true
} }
@@ -758,7 +765,7 @@ async function syncPreviewChanges() {
function restorePreviewItem(id: string) { function restorePreviewItem(id: string) {
const item = previewItems.value.find((entry) => entry.id === id) const item = previewItems.value.find((entry) => entry.id === id)
if (!item || item.sourceStart == null) return if (!item || isManualPreviewItem(item)) return
item.editedContent = item.originalContent item.editedContent = item.originalContent
item.tokenCount = estimateTokenCount(item.originalContent) item.tokenCount = estimateTokenCount(item.originalContent)
item.status = 'original' item.status = 'original'
@@ -897,7 +904,7 @@ async function handleBack() {
ElMessage.warning('请等待当前文件切分完成') ElMessage.warning('请等待当前文件切分完成')
return return
} }
if (currentStepId.value === 'generate') return if (currentStepId.value === 'generate' && !canReturnFromGeneration.value) return
if (currentStep.value > 0) { if (currentStep.value > 0) {
const targetStep = WIZARD_STEPS[currentStep.value - 1]?.id const targetStep = WIZARD_STEPS[currentStep.value - 1]?.id
if (!targetStep) return if (!targetStep) return
@@ -1014,8 +1021,13 @@ async function initializeExistingWorkflow() {
if (sourceTask.status === 'running') resumeStep = 'generate' if (sourceTask.status === 'running') resumeStep = 'generate'
if (resumeStep === 'preview' && !previewItems.value.length) resumeStep = 'upload' if (resumeStep === 'preview' && !previewItems.value.length) resumeStep = 'upload'
if (resumeStep === 'results' && sourceTask.status !== 'completed') resumeStep = 'generate' if (resumeStep === 'results' && sourceTask.status !== 'completed') resumeStep = 'generate'
if (resumeStep === 'generate' || resumeStep === 'results') {
const resume = resumeGeneration()
goToStep(resumeStep)
await resume
return
}
goToStep(resumeStep) goToStep(resumeStep)
if (resumeStep === 'generate' || resumeStep === 'results') await resumeGeneration()
} }
onBeforeUnmount(() => { onBeforeUnmount(() => {
@@ -1154,7 +1166,7 @@ onMounted(() => {
<div class="footer-left"> <div class="footer-left">
<el-button <el-button
v-if="currentStep > 0" v-if="currentStep > 0"
:disabled="currentStepId === 'generate' || previewBuilding || sourceUploading" :disabled="(currentStepId === 'generate' && !canReturnFromGeneration) || previewBuilding || sourceUploading"
@click="handleBack" @click="handleBack"
> >
<i class="fa fa-arrow-left" style="margin-right: 6px;" /> 返回{{ previousStepLabel }} <i class="fa fa-arrow-left" style="margin-right: 6px;" /> 返回{{ previousStepLabel }}
@@ -1167,8 +1179,8 @@ onMounted(() => {
<el-button <el-button
class="wizard-primary-action" class="wizard-primary-action"
type="primary" type="primary"
:loading="modelSubmitLoading || generation.status === 'running' || resultRegenerationBusy || (currentStepId === 'upload' && (sourceUploading || previewBuilding))" :loading="modelSubmitLoading || generationStarting || generation.status === 'running' || resultRegenerationBusy || (currentStepId === 'upload' && (sourceUploading || previewBuilding))"
:disabled="hydrating || modelSubmitLoading || Boolean(initializationError) || resultRegenerationBusy || (currentStepId === 'generate' && generation.status === 'running') || previewBuilding || sourceUploading || (currentStepId === 'upload' && hasUnfinishedUploads)" :disabled="hydrating || modelSubmitLoading || generationStarting || Boolean(initializationError) || resultRegenerationBusy || (currentStepId === 'generate' && generation.status === 'running') || previewBuilding || sourceUploading || (currentStepId === 'upload' && hasUnfinishedUploads)"
@click="handlePrimaryAction" @click="handlePrimaryAction"
> >
{{ primaryActionLabel }} <i class="fa" :class="primaryActionIcon" style="margin-left: 6px;" /> {{ primaryActionLabel }} <i class="fa" :class="primaryActionIcon" style="margin-left: 6px;" />

View File

@@ -9,6 +9,7 @@ import {
getDataProcessResults, getDataProcessResults,
getDataProcessTask, getDataProcessTask,
publishDataProcess, publishDataProcess,
repeatDataProcessTask,
restoreDataProcessResult, restoreDataProcessResult,
updateDataProcessResult, updateDataProcessResult,
} from '@/api/modules/dataProcess' } from '@/api/modules/dataProcess'
@@ -42,6 +43,8 @@ const savingResult = ref(false)
const restoringResultId = ref<string | number | null>(null) const restoringResultId = ref<string | number | null>(null)
const publishDialogVisible = ref(false) const publishDialogVisible = ref(false)
const publishing = ref(false) const publishing = ref(false)
const repeatGenerating = ref(false)
const repeatRequestId = ref('')
const configExpanded = ref(false) const configExpanded = ref(false)
const resultCellTooltipOptions = { const resultCellTooltipOptions = {
popperClass: 'data-process-result-tooltip', popperClass: 'data-process-result-tooltip',
@@ -113,6 +116,51 @@ const preprocessOptionLabelMap: Record<string, string> = {
preserve_context: '保留上下文', preserve_context: '保留上下文',
} }
const structuredPreprocessOptionKeys = new Set([
'clean_invalid',
'deduplicate',
'detect_structure',
'normalize_format',
'desensitize',
'filter_anomaly',
])
function formatStructuredPreprocessOptions(value: unknown[]) {
const options = [...new Set(value.map((item) => String(item)))]
const selected = new Set(options)
const consumed = new Set<string>()
const labels: string[] = []
function appendGroup(values: string[], groupLabel: string) {
const selectedValues = values.filter((item) => selected.has(item))
selectedValues.forEach((item) => consumed.add(item))
if (selectedValues.length === values.length) {
labels.push(groupLabel)
return
}
selectedValues.forEach((item) => {
labels.push(`${preprocessOptionLabelMap[item] || item}(历史部分配置)`)
})
}
appendGroup(['clean_invalid', 'deduplicate'], '数据清洗')
appendGroup(['detect_structure', 'normalize_format'], '结构标准化')
if (selected.has('desensitize')) {
consumed.add('desensitize')
labels.push('敏感信息脱敏')
}
if (selected.has('filter_anomaly')) {
consumed.add('filter_anomaly')
labels.push('异常数据过滤(历史规则)')
}
options.forEach((item) => {
if (!consumed.has(item)) labels.push(preprocessOptionLabelMap[item] || item)
})
return labels.length ? labels.join('、') : '-'
}
function numeric(value: unknown) { function numeric(value: unknown) {
const parsed = typeof value === 'number' ? value : Number(value) const parsed = typeof value === 'number' ? value : Number(value)
return Number.isFinite(parsed) ? parsed : 0 return Number.isFinite(parsed) ? parsed : 0
@@ -199,6 +247,11 @@ const canRegenerate = computed(() => {
|| status === 'stopped' || status === 'stopped'
|| (status === 'completed' && (Boolean(outputDatasetId.value) || hasPublishedOutputs.value)) || (status === 'completed' && (Boolean(outputDatasetId.value) || hasPublishedOutputs.value))
}) })
const canRepeatGeneration = computed(() => (
detail.value?.status === 'completed'
&& detail.value.results_confirmed !== false
&& previewCount.value > 0
))
const creatorName = computed(() => detail.value?.creator_name || detail.value?.creator || '-') const creatorName = computed(() => detail.value?.creator_name || detail.value?.creator || '-')
const createTime = computed(() => detail.value?.create_time || detail.value?.created_at) const createTime = computed(() => detail.value?.create_time || detail.value?.created_at)
const startTime = computed(() => detail.value?.start_time || detail.value?.started_at) const startTime = computed(() => detail.value?.start_time || detail.value?.started_at)
@@ -263,7 +316,12 @@ function formatConfigValue(key: string, value: unknown) {
} }
if (Array.isArray(value)) { if (Array.isArray(value)) {
if (key === 'preprocess_options') { if (key === 'preprocess_options') {
return value.length const containsStructuredOption = value.some((item) => (
structuredPreprocessOptionKeys.has(String(item))
))
return containsStructuredOption
? formatStructuredPreprocessOptions(value)
: value.length
? value.map((item) => preprocessOptionLabelMap[String(item)] || String(item)).join('、') ? value.map((item) => preprocessOptionLabelMap[String(item)] || String(item)).join('、')
: '-' : '-'
} }
@@ -503,6 +561,48 @@ function startRegeneration() {
void router.push({ name: 'data-process-regenerate', params: { id: taskId.value } }) void router.push({ name: 'data-process-regenerate', params: { id: taskId.value } })
} }
function createRepeatRequestId() {
if (typeof globalThis.crypto?.randomUUID === 'function') {
return globalThis.crypto.randomUUID()
}
return `${Date.now()}_${Math.random().toString(36).slice(2, 14)}`
}
async function repeatGeneration() {
if (!detail.value?.updated_at || repeatGenerating.value) return
try {
await ElMessageBox.confirm(
'系统会复制当前配置、源文件和切分结果,创建一个独立的新任务并在后台生成。原任务和原结果不会被修改。',
'按原配置再生成一批?',
{
confirmButtonText: '创建并开始生成',
cancelButtonText: '取消',
type: 'info',
},
)
} catch {
return
}
repeatGenerating.value = true
repeatRequestId.value ||= createRepeatRequestId()
try {
const repeated = await repeatDataProcessTask(taskId.value, {
expected_updated_at: detail.value.updated_at,
request_id: repeatRequestId.value,
})
ElMessage.success(repeated.created ? '已创建新任务,正在后台生成' : '已恢复此前创建的新任务')
await router.push({
name: 'data-process-workflow',
params: { id: repeated.task.id },
})
} catch {
// 保留幂等请求 ID网络超时后再次点击不会重复创建任务。
} finally {
repeatGenerating.value = false
}
}
watch([currentPage, pageSize], () => void loadResults()) watch([currentPage, pageSize], () => void loadResults())
onMounted(loadPage) onMounted(loadPage)
@@ -520,23 +620,33 @@ onBeforeUnmount(() => {
<el-tag :type="displayStatus.type" size="small" effect="light"> <el-tag :type="displayStatus.type" size="small" effect="light">
{{ displayStatus.label }} {{ displayStatus.label }}
</el-tag> </el-tag>
<div class="heading-actions">
<el-button <el-button
v-if="detail.status === 'completed' && !hasCurrentPublishedDataset" v-if="detail.status === 'completed' && !hasCurrentPublishedDataset"
class="publish-button"
type="primary" type="primary"
@click="openPublishDialog" @click="openPublishDialog"
> >
<i class="fa fa-database" style="margin-right: 4px;" />发布为三个数据集 <i class="fa fa-database" style="margin-right: 4px;" />发布为三个数据集
</el-button> </el-button>
<el-button <el-button
v-if="canRegenerate" v-if="canRepeatGeneration"
class="publish-button"
type="primary" type="primary"
:loading="repeatGenerating"
:disabled="repeatGenerating"
@click="repeatGeneration"
>
<i class="fa fa-clone" style="margin-right: 4px;" />按原配置再生成一批
</el-button>
<el-button
v-if="canRegenerate"
type="warning"
plain
@click="startRegeneration" @click="startRegeneration"
> >
<i class="fa fa-refresh" style="margin-right: 4px;" />重新生成 <i class="fa fa-refresh" style="margin-right: 4px;" />覆盖当前任务重新生成
</el-button> </el-button>
</div> </div>
</div>
<p>{{ detail.description || '暂无任务描述' }}</p> <p>{{ detail.description || '暂无任务描述' }}</p>
<dl class="heading-meta"> <dl class="heading-meta">
<div><dt>任务 ID</dt><dd>{{ detail.id }}</dd></div> <div><dt>任务 ID</dt><dd>{{ detail.id }}</dd></div>
@@ -804,7 +914,15 @@ onBeforeUnmount(() => {
> p { margin: 8px 0 0; color: #64748b; font-size: 13px; } > p { margin: 8px 0 0; color: #64748b; font-size: 13px; }
} }
.publish-button { margin-left: auto; } .heading-actions {
margin-left: auto;
display: flex;
flex-wrap: wrap;
justify-content: flex-end;
gap: 8px;
:deep(.el-button + .el-button) { margin-left: 0; }
}
.load-state-actions { display: flex; gap: 10px; } .load-state-actions { display: flex; gap: 10px; }
.compact-empty { padding: 28px 18px; color: #94a3b8; font-size: 13px; text-align: center; } .compact-empty { padding: 28px 18px; color: #94a3b8; font-size: 13px; text-align: center; }
.publish-form-grid { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); gap: 12px; } .publish-form-grid { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); gap: 12px; }
@@ -968,7 +1086,8 @@ onBeforeUnmount(() => {
@media (max-width: 720px) { @media (max-width: 720px) {
.metric-grid, .config-grid { grid-template-columns: 1fr; } .metric-grid, .config-grid { grid-template-columns: 1fr; }
.detail-heading .heading-row { align-items: flex-start; flex-wrap: wrap; } .detail-heading .heading-row { align-items: flex-start; flex-wrap: wrap; }
.publish-button { width: 100%; margin-left: 0; } .heading-actions { width: 100%; margin-left: 0; }
.heading-actions :deep(.el-button) { width: 100%; }
.publish-form-grid { grid-template-columns: 1fr; gap: 0; } .publish-form-grid { grid-template-columns: 1fr; gap: 0; }
.result-toolbar { align-items: stretch; flex-direction: column; } .result-toolbar { align-items: stretch; flex-direction: column; }
.result-filters { padding: 0 16px 16px; flex-direction: column; } .result-filters { padding: 0 16px 16px; flex-direction: column; }

View File

@@ -46,6 +46,13 @@ const sourceUrl = computed(() => (
? getDataProcessSourceRawUrl(props.taskId, props.sourceFileId) ? getDataProcessSourceRawUrl(props.taskId, props.sourceFileId)
: '' : ''
)) ))
const selectedXlsxLocator = computed(() => {
const locator = props.selectedItem?.sourceLocator
if (!locator) return null
const hasSheet = locator.sheet_index != null || Boolean(locator.sheet_name)
const hasRow = locator.row_number != null || locator.sheet_record_index != null
return hasSheet && hasRow ? locator : null
})
const visibleRowRange = computed(() => { const visibleRowRange = computed(() => {
const sheet = xlsxPreview.value?.active_sheet const sheet = xlsxPreview.value?.active_sheet
if (!sheet || !sheet.rows.length) return '当前工作表没有可预览记录' if (!sheet || !sheet.rows.length) return '当前工作表没有可预览记录'
@@ -95,6 +102,16 @@ const selectedRecordKey = computed(() => {
}) })
function xlsxRowHighlighted(row: DataProcessXlsxPreviewRow) { function xlsxRowHighlighted(row: DataProcessXlsxPreviewRow) {
const locator = selectedXlsxLocator.value
const sheet = xlsxPreview.value?.active_sheet
if (locator && sheet) {
const sheetMatches = locator.sheet_index != null
? sheet.index === locator.sheet_index
: sheet.name === locator.sheet_name
if (!sheetMatches) return false
if (locator.row_number != null) return row.row_number === locator.row_number
return row.record_index === locator.sheet_record_index
}
return Boolean(selectedRecordKey.value && recordKey(row.record) === selectedRecordKey.value) return Boolean(selectedRecordKey.value && recordKey(row.record) === selectedRecordKey.value)
} }
@@ -119,8 +136,11 @@ async function locateSelectedItem() {
async function loadPreview(options: { reset?: boolean } = {}) { async function loadPreview(options: { reset?: boolean } = {}) {
const sequence = ++loadSequence const sequence = ++loadSequence
if (options.reset) { if (options.reset) {
activeSheetIndex.value = 0 const locator = selectedXlsxLocator.value
pageOffset.value = 0 activeSheetIndex.value = locator?.sheet_index ?? 0
pageOffset.value = locator?.sheet_record_index == null
? 0
: Math.floor(locator.sheet_record_index / XLSX_PAGE_SIZE) * XLSX_PAGE_SIZE
preview.value = null preview.value = null
} }
errorMessage.value = '' errorMessage.value = ''
@@ -178,8 +198,34 @@ watch(
) )
watch( watch(
() => props.selectedItem?.id, () => [
() => void locateSelectedItem(), props.selectedItem?.id,
props.selectedItem?.sourceLocator?.sheet_index,
props.selectedItem?.sourceLocator?.sheet_record_index,
props.selectedItem?.sourceLocator?.row_number,
],
() => {
const locator = selectedXlsxLocator.value
if (!locator || isDocx.value) {
void locateSelectedItem()
return
}
const targetSheet = locator.sheet_index ?? activeSheetIndex.value
const targetOffset = locator.sheet_record_index == null
? pageOffset.value
: Math.floor(locator.sheet_record_index / XLSX_PAGE_SIZE) * XLSX_PAGE_SIZE
const activeSheet = xlsxPreview.value?.active_sheet
if (
activeSheet?.index === targetSheet
&& activeSheet.offset === targetOffset
) {
void locateSelectedItem()
return
}
activeSheetIndex.value = targetSheet
pageOffset.value = targetOffset
void loadPreview()
},
) )
</script> </script>
@@ -301,6 +347,8 @@ watch(
:key="row.row_number" :key="row.row_number"
class="xlsx-row" class="xlsx-row"
:class="{ 'is-highlighted': xlsxRowHighlighted(row) }" :class="{ 'is-highlighted': xlsxRowHighlighted(row) }"
:data-row-number="row.row_number"
:data-record-index="row.record_index"
> >
<th class="row-number-cell">{{ row.row_number }}</th> <th class="row-number-cell">{{ row.row_number }}</th>
<td <td

View File

@@ -2,7 +2,11 @@
import { computed, nextTick, ref, watch } from 'vue' import { computed, nextTick, ref, watch } from 'vue'
import OfficeSourceViewer from './OfficeSourceViewer.vue' import OfficeSourceViewer from './OfficeSourceViewer.vue'
import PdfSourceViewer from './PdfSourceViewer.vue' import PdfSourceViewer from './PdfSourceViewer.vue'
import { sourceLines } from './previewModel' import {
isManualPreviewItem,
sourceLineNumberAtOffset,
sourceLineWindow,
} from './previewModel'
import type { PreviewItem, ProcessType } from './types' import type { PreviewItem, ProcessType } from './types'
const props = defineProps<{ const props = defineProps<{
@@ -31,9 +35,11 @@ const sourceViewerRef = ref<HTMLElement | null>(null)
const search = ref('') const search = ref('')
const currentPage = ref(1) const currentPage = ref(1)
const PREVIEW_PAGE_SIZE = 10 const PREVIEW_PAGE_SIZE = 10
const SOURCE_LINE_RENDER_LIMIT = 240
const SOURCE_LINE_CHARACTER_LIMIT = 4_000
const sourceWindowStartLine = ref(1)
const editingItemId = ref<string | null>(null) const editingItemId = ref<string | null>(null)
const editorDraft = ref('') const editorDraft = ref('')
const lines = computed(() => sourceLines(props.sourceText))
const selectedItem = computed(() => props.items.find((item) => item.id === props.selectedId) ?? props.items[0]) const selectedItem = computed(() => props.items.find((item) => item.id === props.selectedId) ?? props.items[0])
const editingItem = computed(() => props.items.find((item) => item.id === editingItemId.value)) const editingItem = computed(() => props.items.find((item) => item.id === editingItemId.value))
const normalizedFileFormat = computed(() => ( const normalizedFileFormat = computed(() => (
@@ -43,6 +49,27 @@ const normalizedFileFormat = computed(() => (
)) ))
const isPdfSource = computed(() => normalizedFileFormat.value === 'pdf') const isPdfSource = computed(() => normalizedFileFormat.value === 'pdf')
const isOfficeSource = computed(() => ['docx', 'xlsx'].includes(normalizedFileFormat.value)) const isOfficeSource = computed(() => ['docx', 'xlsx'].includes(normalizedFileFormat.value))
const selectedSourceOffset = computed(() => {
const item = selectedItem.value
return item ? sourceOffsetRange(item)?.start ?? null : null
})
const selectedSourceLine = computed(() => {
const item = selectedItem.value
if (!item) return null
return sourceLineRange(item)?.start
?? (selectedSourceOffset.value == null
? null
: sourceLineNumberAtOffset(props.sourceText, selectedSourceOffset.value))
})
const visibleSourceWindow = computed(() => sourceLineWindow(
props.sourceText,
sourceWindowStartLine.value,
SOURCE_LINE_RENDER_LIMIT,
SOURCE_LINE_CHARACTER_LIMIT,
selectedSourceLine.value,
selectedSourceOffset.value,
))
const lines = computed(() => visibleSourceWindow.value.lines)
const filteredItems = computed(() => props.items.filter((item, index) => { const filteredItems = computed(() => props.items.filter((item, index) => {
const matchesSearch = !search.value.trim() const matchesSearch = !search.value.trim()
@@ -58,10 +85,27 @@ const pagedItems = computed(() => {
const selectedIndex = computed(() => props.items.findIndex((item) => item.id === selectedItem.value?.id)) const selectedIndex = computed(() => props.items.findIndex((item) => item.id === selectedItem.value?.id))
function isLineHighlighted(lineStart: number, lineEnd: number) { function sourceLineRange(item: PreviewItem) {
const start = item.sourceLocator?.start_line ?? item.sourceStartLine
const end = item.sourceLocator?.end_line ?? item.sourceEndLine ?? start
return start == null ? null : { start, end: end ?? start }
}
function sourceOffsetRange(item: PreviewItem) {
const start = item.sourceLocator?.source_start ?? item.sourceStart
const end = item.sourceLocator?.source_end ?? item.sourceEnd ?? start
return start == null ? null : { start, end: Math.max(start, end ?? start) }
}
function isLineHighlighted(lineNumber: number, lineStart: number, lineEnd: number) {
const item = selectedItem.value const item = selectedItem.value
if (!item || item.sourceStart == null || item.sourceEnd == null) return false if (!item) return false
return lineEnd >= item.sourceStart && lineStart <= item.sourceEnd const lineRange = sourceLineRange(item)
if (lineRange) return lineNumber >= lineRange.start && lineNumber <= lineRange.end
const offsetRange = sourceOffsetRange(item)
if (!offsetRange) return false
const effectiveEnd = Math.max(offsetRange.start + 1, offsetRange.end)
return lineEnd >= offsetRange.start && lineStart < effectiveEnd
} }
function selectItem(id: string) { function selectItem(id: string) {
@@ -107,34 +151,88 @@ watch(search, () => {
watch(() => props.selectedFileId, closeEditor) watch(() => props.selectedFileId, closeEditor)
watch(selectedItem, async (item) => { watch([selectedItem, () => props.sourceText], async ([item]) => {
if (!item) return if (!item) return
const visibleIndex = filteredItems.value.findIndex((entry) => entry.id === item.id) const visibleIndex = filteredItems.value.findIndex((entry) => entry.id === item.id)
if (visibleIndex >= 0) { if (visibleIndex >= 0) {
currentPage.value = Math.floor(visibleIndex / PREVIEW_PAGE_SIZE) + 1 currentPage.value = Math.floor(visibleIndex / PREVIEW_PAGE_SIZE) + 1
} }
if (isPdfSource.value || isOfficeSource.value || item.sourceStart == null) return if (isPdfSource.value || isOfficeSource.value) return
const itemLineRange = sourceLineRange(item)
const itemOffsetRange = sourceOffsetRange(item)
if (!itemLineRange && !itemOffsetRange) {
sourceWindowStartLine.value = 1
return
}
const targetLine = selectedSourceLine.value
?? sourceLineNumberAtOffset(props.sourceText, itemOffsetRange?.start ?? 0)
sourceWindowStartLine.value = Math.max(1, targetLine - Math.floor(SOURCE_LINE_RENDER_LIMIT / 3))
await nextTick() await nextTick()
const target = sourceViewerRef.value?.querySelector<HTMLElement>(`[data-source-start="${item.sourceStart}"]`) const exactTarget = sourceViewerRef.value
?.querySelector<HTMLElement>(`[data-line-number="${targetLine}"]`)
const target = exactTarget
?? sourceViewerRef.value?.querySelector<HTMLElement>('.source-line.is-highlighted') ?? sourceViewerRef.value?.querySelector<HTMLElement>('.source-line.is-highlighted')
target?.scrollIntoView({ block: 'center', behavior: 'smooth' }) target?.scrollIntoView({ block: 'center', behavior: 'smooth' })
}, { immediate: true }) }, { immediate: true })
async function showPreviousSourceWindow() {
sourceWindowStartLine.value = Math.max(1, sourceWindowStartLine.value - SOURCE_LINE_RENDER_LIMIT)
await nextTick()
if (sourceViewerRef.value) sourceViewerRef.value.scrollTop = 0
}
async function showNextSourceWindow() {
if (!visibleSourceWindow.value.hasMore) return
sourceWindowStartLine.value = visibleSourceWindow.value.endLine + 1
await nextTick()
if (sourceViewerRef.value) sourceViewerRef.value.scrollTop = 0
}
function itemNumber(item: PreviewItem) { function itemNumber(item: PreviewItem) {
return props.items.findIndex((entry) => entry.id === item.id) + 1 return props.items.findIndex((entry) => entry.id === item.id) + 1
} }
function lineRange(item: PreviewItem) { function lineRange(item: PreviewItem) {
if (isManualPreviewItem(item)) return '手动新增,无源文件定位'
const locator = item.sourceLocator
const locatedLines = sourceLineRange(item)
if (props.processType === 'unstructured') {
const parts: string[] = []
if (item.sourcePages?.length) { if (item.sourcePages?.length) {
const first = item.sourcePages[0] const first = item.sourcePages[0]
const last = item.sourcePages[item.sourcePages.length - 1] const last = item.sourcePages[item.sourcePages.length - 1]
return first === last ? `来源:${first}` : `来源:${first}${last}` parts.push(first === last ? `${first}` : `${first}${last}`)
} }
if (item.sourceStartLine == null || item.sourceEndLine == null) return '手动新增,无源文件定位' if (locatedLines) {
return item.sourceStartLine === item.sourceEndLine parts.push(
? `来源:第 ${item.sourceStartLine}` locatedLines.start === locatedLines.end
: `来源:${item.sourceStartLine}${item.sourceEndLine}` ? `${locatedLines.start}`
: `${locatedLines.start}${locatedLines.end}`,
)
}
if (item.headingPath?.length) parts.push(`章节:${item.headingPath.join(' / ')}`)
return parts.length ? `来源:${parts.join(' · ')}` : '来源:源文件内容(无精确定位)'
}
if (locator?.kind === 'xlsx') {
const sheet = locator.sheet_name || `工作表 ${Number(locator.sheet_index ?? 0) + 1}`
return locator.row_number != null
? `来源:${sheet} · 第 ${locator.row_number}`
: `来源:${sheet}`
}
if (locator?.kind === 'json') {
return locator.json_pointer
? `来源JSON 路径 ${locator.json_pointer}`
: '来源JSON 根对象'
}
if (locatedLines) {
return locatedLines.start === locatedLines.end
? `来源:第 ${locatedLines.start}`
: `来源:第 ${locatedLines.start}${locatedLines.end}`
}
return '来源:源文件记录'
} }
</script> </script>
@@ -175,6 +273,31 @@ function lineRange(item: PreviewItem) {
<div> <div>
<strong>源文件 · {{ fileName }}</strong> <strong>源文件 · {{ fileName }}</strong>
</div> </div>
<div
v-if="!isPdfSource && !isOfficeSource && lines.length"
class="source-window-controls"
aria-label="源文件行窗口"
>
<span> {{ visibleSourceWindow.startLine }}{{ visibleSourceWindow.endLine }} </span>
<el-button
link
size="small"
aria-label="查看上一段源文件"
:disabled="!visibleSourceWindow.hasPrevious"
@click="showPreviousSourceWindow"
>
上一段
</el-button>
<el-button
link
size="small"
aria-label="查看下一段源文件"
:disabled="!visibleSourceWindow.hasMore"
@click="showNextSourceWindow"
>
下一段
</el-button>
</div>
</div> </div>
<PdfSourceViewer <PdfSourceViewer
@@ -197,8 +320,9 @@ function lineRange(item: PreviewItem) {
v-for="line in lines" v-for="line in lines"
:key="line.number" :key="line.number"
class="source-line" class="source-line"
:class="{ 'is-highlighted': isLineHighlighted(line.start, line.end) }" :class="{ 'is-highlighted': isLineHighlighted(line.number, line.start, line.end) }"
:data-source-start="line.start" :data-source-start="line.start"
:data-line-number="line.number"
> >
<span class="line-number">{{ line.number }}</span> <span class="line-number">{{ line.number }}</span>
<span class="line-content">{{ line.content || ' ' }}</span> <span class="line-content">{{ line.content || ' ' }}</span>
@@ -282,7 +406,7 @@ function lineRange(item: PreviewItem) {
/> />
<div class="editor-actions"> <div class="editor-actions">
<el-button <el-button
v-if="editingItem.sourceStart != null" v-if="!isManualPreviewItem(editingItem)"
link link
@click="restoreItem" @click="restoreItem"
> >
@@ -431,6 +555,22 @@ function lineRange(item: PreviewItem) {
} }
} }
.source-window-controls {
flex: none;
gap: 2px !important;
> span {
margin-right: 4px;
color: #8a93a3;
font-size: 11px;
white-space: nowrap;
}
:deep(.el-button) {
margin-left: 0;
}
}
.source-viewer { .source-viewer {
flex: 1; flex: 1;
height: 538px; height: 538px;

View File

@@ -1,4 +1,5 @@
<script setup lang="ts"> <script setup lang="ts">
import { computed } from 'vue'
import type { import type {
GenerationControlOptions, GenerationControlOptions,
PreprocessOption, PreprocessOption,
@@ -21,27 +22,32 @@ const emit = defineEmits<{
'update:options': [value: StructuredProcessOptions] 'update:options': [value: StructuredProcessOptions]
}>() }>()
const PREPROCESS_OPTIONS: Array<{ const PREPROCESS_GROUPS: Array<{
value: PreprocessOption values: PreprocessOption[]
label: string label: string
description: string description: string
}> = [ }> = [
{ value: 'clean_invalid', label: '清理无效数据', description: '清理全空列,并剔除关键字段残缺的数据行' },
{ {
value: 'detect_structure', values: ['clean_invalid', 'deduplicate'],
label: '嵌套结构展平', label: '数据清洗',
description: '展平嵌套对象和可解析的 JSON 字段Excel 表头与合并单元格在上传时自动解析', description: '清理全空列和空记录,并删除内容完全相同的记录;不会猜测可空字段是否必填',
}, },
{ {
value: 'deduplicate', values: ['detect_structure', 'normalize_format'],
label: '重复记录去重', label: '结构标准化',
description: '按整行内容或 id、uuid、key、code、*_id 等身份字段去重,暂不支持自定义组合字段', description: '展平嵌套对象和可解析的 JSON 字段,并统一编码、空白、字段名和 JSON 序列化格式',
},
{
values: ['desensitize'],
label: '敏感信息脱敏',
description: '识别并脱敏姓名、手机号、邮箱和身份证号',
}, },
{ value: 'normalize_format', label: '数据格式标准化', description: '按所选规则统一编码、空白、字段名及 JSON 序列化格式' },
{ value: 'filter_anomaly', label: '异常数据过滤', description: '使用 IQR 识别数值离群值,并过滤乱码等异常记录' },
{ value: 'desensitize', label: '敏感信息脱敏', description: '识别并脱敏姓名、手机号、邮箱和身份证号' },
] ]
const legacyAnomalyFilterEnabled = computed(() => (
props.options.preprocessOptions.includes('filter_anomaly')
))
function updateField<K extends keyof StructuredProcessOptions>( function updateField<K extends keyof StructuredProcessOptions>(
field: K, field: K,
value: StructuredProcessOptions[K], value: StructuredProcessOptions[K],
@@ -57,14 +63,26 @@ function updateQaPairsPerRow(value: number | undefined) {
updateField('qaPairsPerRow', normalizeQaPairsGenerationCount(value)) updateField('qaPairsPerRow', normalizeQaPairsGenerationCount(value))
} }
function updatePreprocessOptions(value: Array<string | number | boolean>) { function selectedCount(values: PreprocessOption[]) {
const allowedValues = new Set(PREPROCESS_OPTIONS.map((option) => option.value)) return values.filter((value) => props.options.preprocessOptions.includes(value)).length
const preprocessOptions = Array.from(new Set(value.filter( }
(option): option is PreprocessOption => (
typeof option === 'string' && allowedValues.has(option as PreprocessOption) function groupSelected(values: PreprocessOption[]) {
), return selectedCount(values) === values.length
))) }
updateField('preprocessOptions', preprocessOptions)
function groupIndeterminate(values: PreprocessOption[]) {
const count = selectedCount(values)
return count > 0 && count < values.length
}
function updatePreprocessGroup(values: PreprocessOption[], checked: string | number | boolean) {
const next = new Set(props.options.preprocessOptions)
values.forEach((value) => {
if (Boolean(checked)) next.add(value)
else next.delete(value)
})
updateField('preprocessOptions', [...next])
} }
</script> </script>
@@ -73,26 +91,34 @@ function updatePreprocessOptions(value: Array<string | number | boolean>) {
<div class="section-title-row"> <div class="section-title-row">
<div> <div>
<h3>预处理选项</h3> <h3>预处理选项</h3>
<p>选择在生成问答对之前需要执行的数据处理方式</p> <p>默认不执行预处理请按数据情况自行选择</p>
</div> </div>
</div> </div>
<el-checkbox-group <div class="preprocess-option-grid">
:model-value="options.preprocessOptions" <label
class="preprocess-option-grid" v-for="group in PREPROCESS_GROUPS"
@update:model-value="updatePreprocessOptions" :key="group.label"
class="preprocess-option"
:class="{ 'is-checked': groupSelected(group.values) }"
> >
<el-checkbox <el-checkbox
v-for="option in PREPROCESS_OPTIONS" :model-value="groupSelected(group.values)"
:key="option.value" :indeterminate="groupIndeterminate(group.values)"
:value="option.value" @update:model-value="updatePreprocessGroup(group.values, $event)"
class="preprocess-option" />
>
<span class="preprocess-option-copy"> <span class="preprocess-option-copy">
<strong>{{ option.label }}</strong> <strong>{{ group.label }}</strong>
<small>{{ option.description }}</small> <small>{{ group.description }}</small>
</span> </span>
</el-checkbox> </label>
</el-checkbox-group> </div>
<el-alert
v-if="legacyAnomalyFilterEnabled"
class="legacy-preprocess-alert"
type="warning"
:closable="false"
title="该历史任务仍启用了已停用的“异常数据过滤”;为保证结果可复现,本次继续保留"
/>
</div> </div>
<div class="form-section generation-options-section"> <div class="form-section generation-options-section">

View File

@@ -119,7 +119,7 @@ defineExpose({ revealValidation })
<div class="section-title-row"> <div class="section-title-row">
<div> <div>
<h3>预处理选项</h3> <h3>预处理选项</h3>
<p>默认启用结构感知的推荐策略只需决定是否需要脱敏</p> <p>默认不执行预处理请按文档情况自行选择</p>
</div> </div>
</div> </div>
<div class="preprocess-option-grid"> <div class="preprocess-option-grid">

View File

@@ -97,7 +97,7 @@ export function isBuiltInGenerationPrompt(value: string) {
export function createDefaultStructuredOptions(): StructuredProcessOptions { export function createDefaultStructuredOptions(): StructuredProcessOptions {
return { return {
preprocessOptions: ['clean_invalid', 'detect_structure', 'deduplicate', 'normalize_format'], preprocessOptions: [],
semanticEnrichment: false, semanticEnrichment: false,
qaPairsPerRow: 1, qaPairsPerRow: 1,
datasetSplit: { train: 80, validation: 10, test: 10 }, datasetSplit: { train: 80, validation: 10, test: 10 },
@@ -117,22 +117,15 @@ export function createDefaultStructuredOptions(): StructuredProcessOptions {
export function createDefaultUnstructuredOptions(): UnstructuredProcessOptions { export function createDefaultUnstructuredOptions(): UnstructuredProcessOptions {
return { return {
preprocessOptions: [ preprocessOptions: [],
'clean_invalid_content',
'detect_document_structure',
'merge_short_content',
'filter_low_quality',
'deduplicate_content',
'preserve_context',
],
chunkMethod: 'layout_hybrid', chunkMethod: 'layout_hybrid',
chunkSize: 800, chunkSize: 800,
chunkOverlap: 100, chunkOverlap: 100,
minChunkSize: 100, minChunkSize: 100,
semanticBreakpointPercentile: 95, semanticBreakpointPercentile: 95,
preserveTables: true, preserveTables: false,
preserveCodeBlocks: true, preserveCodeBlocks: false,
preserveLists: true, preserveLists: false,
semanticEnrichment: false, semanticEnrichment: false,
qaPairsPerChunk: 1, qaPairsPerChunk: 1,
datasetSplit: { train: 80, validation: 10, test: 10 }, datasetSplit: { train: 80, validation: 10, test: 10 },
@@ -218,11 +211,21 @@ function generationOptionsFromConfig(
export function createStructuredOptionsFromConfig(config: DataProcessConfig): StructuredProcessOptions { export function createStructuredOptionsFromConfig(config: DataProcessConfig): StructuredProcessOptions {
const defaults = createDefaultStructuredOptions() const defaults = createDefaultStructuredOptions()
const preprocessOptions = configValue<unknown>(config, 'preprocess_options', []) const preprocessOptions = configValue<unknown>(config, 'preprocess_options', [])
const supportedPreprocessOptions = new Set<PreprocessOption>([
'clean_invalid',
'deduplicate',
'detect_structure',
'normalize_format',
'desensitize',
'filter_anomaly',
])
return { return {
...defaults, ...defaults,
...generationOptionsFromConfig(config, defaults), ...generationOptionsFromConfig(config, defaults),
preprocessOptions: Array.isArray(preprocessOptions) preprocessOptions: Array.isArray(preprocessOptions)
? preprocessOptions.map(String) as PreprocessOption[] ? Array.from(new Set(preprocessOptions.map(String).filter(
(option): option is PreprocessOption => supportedPreprocessOptions.has(option as PreprocessOption),
)))
: defaults.preprocessOptions, : defaults.preprocessOptions,
semanticEnrichment: Boolean(configValue( semanticEnrichment: Boolean(configValue(
config, config,

View File

@@ -1,4 +1,4 @@
import type { SourceLine } from './types' import type { PreviewItem, SourceLine } from './types'
/** 仅用于“使用示例”上传;正式预览和切片全部由后端生成。 */ /** 仅用于“使用示例”上传;正式预览和切片全部由后端生成。 */
export const DEFAULT_SOURCE_TEXT = [ export const DEFAULT_SOURCE_TEXT = [
@@ -12,19 +12,141 @@ export const DEFAULT_SOURCE_TEXT = [
'答:复利是将上一期利息加入本金,再计算下一期利息。', '答:复利是将上一期利息加入本金,再计算下一期利息。',
].join('\n') ].join('\n')
/** export interface SourceLineWindow {
* 把后端返回的字符偏移映射为源文件行,仅负责界面高亮,不参与切片。 lines: SourceLine[]
*/ startLine: number
export function sourceLines(sourceText: string): SourceLine[] { endLine: number
const rawLines = sourceText.split('\n') hasPrevious: boolean
let cursor = 0 hasMore: boolean
}
return rawLines.map((content, index) => { function unicodeCodePointLength(value: string, start = 0, end = value.length) {
const start = cursor let length = 0
const end = start + content.length let index = start
cursor = end + (index < rawLines.length - 1 ? 1 : 0) while (index < end) {
return { number: index + 1, content, start, end } const codePoint = value.codePointAt(index)
}) index += codePoint != null && codePoint > 0xffff ? 2 : 1
length += 1
}
return length
}
function advanceCodePoints(value: string, start: number, end: number, count: number) {
let index = start
let remaining = Math.max(0, count)
while (index < end && remaining > 0) {
const codePoint = value.codePointAt(index)
index += codePoint != null && codePoint > 0xffff ? 2 : 1
remaining -= 1
}
return index
}
/**
* 只扫描并返回当前可见行窗口,不对全文 split避免大文件生成巨量字符串数组。
* 字符定位场景可开启 code point 偏移,以与后端 Python 的字符计数保持一致。
*/
export function sourceLineWindow(
sourceText: string,
requestedStartLine: number,
maxLines: number,
maxCharactersPerLine: number,
focusLine: number | null = null,
focusOffset: number | null = null,
): SourceLineWindow {
const startLine = Math.max(1, Math.trunc(requestedStartLine) || 1)
const limit = Math.max(1, Math.trunc(maxLines) || 1)
const characterLimit = Math.max(1, Math.trunc(maxCharactersPerLine) || 1)
const trackUnicodeOffsets = focusOffset != null
const lines: SourceLine[] = []
let lineNumber = 1
let jsCursor = 0
let sourceCursor = 0
while (jsCursor <= sourceText.length && lineNumber < startLine) {
const newlineIndex = sourceText.indexOf('\n', jsCursor)
const jsEnd = newlineIndex >= 0 ? newlineIndex : sourceText.length
sourceCursor = trackUnicodeOffsets
? sourceCursor + unicodeCodePointLength(sourceText, jsCursor, jsEnd) + (newlineIndex >= 0 ? 1 : 0)
: (newlineIndex >= 0 ? newlineIndex + 1 : sourceText.length + 1)
jsCursor = newlineIndex >= 0 ? newlineIndex + 1 : sourceText.length + 1
lineNumber += 1
}
while (jsCursor <= sourceText.length && lines.length < limit) {
const newlineIndex = sourceText.indexOf('\n', jsCursor)
const jsEnd = newlineIndex >= 0 ? newlineIndex : sourceText.length
const fullSourceEnd = trackUnicodeOffsets
? sourceCursor + unicodeCodePointLength(sourceText, jsCursor, jsEnd)
: jsEnd
const focusedStart = focusLine === lineNumber && focusOffset != null
? Math.max(sourceCursor, focusOffset - Math.floor(characterLimit / 3))
: sourceCursor
const segmentSourceStart = Math.min(
focusedStart,
Math.max(sourceCursor, fullSourceEnd - characterLimit),
)
const relativeSegmentStart = trackUnicodeOffsets
? segmentSourceStart - sourceCursor
: Math.max(0, segmentSourceStart - jsCursor)
const segmentJsStart = advanceCodePoints(
sourceText,
jsCursor,
jsEnd,
relativeSegmentStart,
)
const segmentJsEnd = advanceCodePoints(
sourceText,
segmentJsStart,
jsEnd,
characterLimit,
)
const segmentLength = trackUnicodeOffsets
? unicodeCodePointLength(sourceText, segmentJsStart, segmentJsEnd)
: segmentJsEnd - segmentJsStart
const start = trackUnicodeOffsets ? segmentSourceStart : segmentJsStart
const end = start + segmentLength
const content = `${segmentJsStart > jsCursor ? '… ' : ''}${sourceText.slice(segmentJsStart, segmentJsEnd)}${segmentJsEnd < jsEnd ? ' …' : ''}`
lines.push({ number: lineNumber, content, start, end })
sourceCursor = fullSourceEnd + (newlineIndex >= 0 ? 1 : 0)
jsCursor = newlineIndex >= 0 ? newlineIndex + 1 : sourceText.length + 1
lineNumber += 1
}
return {
lines,
startLine: lines[0]?.number ?? startLine,
endLine: lines[lines.length - 1]?.number ?? startLine,
hasPrevious: startLine > 1,
hasMore: jsCursor <= sourceText.length,
}
}
/** 根据后端 code point 偏移查找物理行号,不构建全文行数组。 */
export function sourceLineNumberAtOffset(sourceText: string, targetOffset: number) {
const normalizedOffset = Math.max(0, Math.trunc(targetOffset) || 0)
let offset = 0
let lineNumber = 1
for (const character of sourceText) {
if (offset >= normalizedOffset) break
if (character === '\n') lineNumber += 1
offset += 1
}
return lineNumber
}
/**
* 手动新增项可能先以空内容保存为 invalid编辑后又由后端标记为 modified
* 因此不能只依赖可变的 status空原文且完全没有来源定位才是稳定兜底。
*/
export function isManualPreviewItem(item: PreviewItem): boolean {
const hasSourceLocation = item.sourceStart != null
|| item.sourceEnd != null
|| item.sourceStartLine != null
|| item.sourceEndLine != null
|| Boolean(item.sourcePages?.length)
|| Boolean(item.sourceLocator)
return item.status === 'manual' || (!item.originalContent && !hasSourceLocation)
} }
/** 与后端预览 token 估算规则一致,仅用于编辑中的即时计数。 */ /** 与后端预览 token 估算规则一致,仅用于编辑中的即时计数。 */

View File

@@ -24,6 +24,7 @@ export type PreprocessOption =
| 'detect_structure' | 'detect_structure'
| 'deduplicate' | 'deduplicate'
| 'normalize_format' | 'normalize_format'
/** 仅用于恢复历史任务,新任务界面不再提供。 */
| 'filter_anomaly' | 'filter_anomaly'
| 'desensitize' | 'desensitize'
@@ -94,7 +95,6 @@ export interface ExternalDataSource {
export interface UploadedDataFile { export interface UploadedDataFile {
uid: string | number uid: string | number
sourceFileId?: string sourceFileId?: string
rawFile?: File
name: string name: string
size: number size: number
count: number count: number
@@ -118,6 +118,22 @@ export interface SourceLine {
end: number end: number
} }
export type PreviewSourceLocatorKind = 'json' | 'jsonl' | 'csv' | 'xlsx'
export interface PreviewSourceLocator {
kind: PreviewSourceLocatorKind
record_index?: number | null
start_line?: number | null
end_line?: number | null
source_start?: number | null
source_end?: number | null
json_pointer?: string | null
sheet_index?: number | null
sheet_name?: string | null
row_number?: number | null
sheet_record_index?: number | null
}
export interface PreviewItem { export interface PreviewItem {
id: string id: string
sourceFileId: string sourceFileId: string
@@ -129,6 +145,8 @@ export interface PreviewItem {
sourceStartLine: number | null sourceStartLine: number | null
sourceEndLine: number | null sourceEndLine: number | null
sourcePages?: number[] sourcePages?: number[]
sourceLocator?: PreviewSourceLocator
headingPath?: string[]
tokenCount: number tokenCount: number
status: 'original' | 'modified' | 'manual' | 'invalid' status: 'original' | 'modified' | 'manual' | 'invalid'
qualityScore?: number qualityScore?: number

View File

@@ -71,7 +71,11 @@ export function useDataProcessGeneration(bindings: GenerationBindings) {
let generationTimer: ReturnType<typeof setTimeout> | null = null let generationTimer: ReturnType<typeof setTimeout> | null = null
let generationRun = 0 let generationRun = 0
let pollFailureCount = 0 let pollFailureCount = 0
let generationStarting = false const generationStarting = ref(false)
const generationRestoring = ref(false)
const canReturnFromGeneration = computed(() => (
generation.status === 'idle' && !generationStarting.value && !generationRestoring.value
))
function stopGenerationTimer() { function stopGenerationTimer() {
generationRun += 1 generationRun += 1
@@ -170,14 +174,14 @@ export function useDataProcessGeneration(bindings: GenerationBindings) {
} }
async function startGeneration() { async function startGeneration() {
if (generationStarting || generation.status === 'running') return false if (generationStarting.value || generation.status === 'running') return false
const taskId = bindings.taskId.value const taskId = bindings.taskId.value
if (!taskId) { if (!taskId) {
ElMessage.error('任务尚未创建,请返回上一步重试') ElMessage.error('任务尚未创建,请返回上一步重试')
return false return false
} }
generationStarting = true generationStarting.value = true
let runId: number | null = null let runId: number | null = null
try { try {
const canStart = await bindings.beforeGenerate?.() const canStart = await bindings.beforeGenerate?.()
@@ -204,17 +208,18 @@ export function useDataProcessGeneration(bindings: GenerationBindings) {
generation.message = error instanceof Error ? error.message : '启动数据处理失败,请重试。' generation.message = error instanceof Error ? error.message : '启动数据处理失败,请重试。'
return false return false
} finally { } finally {
generationStarting = false generationStarting.value = false
} }
} }
async function resumeGeneration() { async function resumeGeneration() {
const taskId = bindings.taskId.value const taskId = bindings.taskId.value
if (!taskId) return if (!taskId) return
generationRestoring.value = true
try {
stopGenerationTimer() stopGenerationTimer()
const activeRunId = generationRun const activeRunId = generationRun
pollFailureCount = 0 pollFailureCount = 0
try {
const progress = await getDataProcessProgress(taskId) const progress = await getDataProcessProgress(taskId)
if (activeRunId !== generationRun) return if (activeRunId !== generationRun) return
if (progress.status === 'running') { if (progress.status === 'running') {
@@ -236,6 +241,8 @@ export function useDataProcessGeneration(bindings: GenerationBindings) {
} catch (error) { } catch (error) {
generation.status = 'failed' generation.status = 'failed'
generation.message = error instanceof Error ? error.message : '查询任务进度失败,请重试。' generation.message = error instanceof Error ? error.message : '查询任务进度失败,请重试。'
} finally {
generationRestoring.value = false
} }
} }
@@ -430,7 +437,9 @@ export function useDataProcessGeneration(bindings: GenerationBindings) {
return { return {
bulkRegeneration, bulkRegeneration,
canReturnFromGeneration,
generation, generation,
generationStarting,
regeneratingResultId, regeneratingResultId,
resultRegenerationBusy, resultRegenerationBusy,
results, results,

View File

@@ -2,7 +2,6 @@ import { computed, nextTick, ref, type Reactive, type Ref } from 'vue'
import { useRoute } from 'vue-router' import { useRoute } from 'vue-router'
import { import {
getDataProcessPreview, getDataProcessPreview,
getDataProcessSourceContent,
getDataProcessTask, getDataProcessTask,
regenerateDataProcessTask, regenerateDataProcessTask,
} from '@/api/modules/dataProcess' } from '@/api/modules/dataProcess'
@@ -15,7 +14,10 @@ import {
createStructuredOptionsFromConfig, createStructuredOptionsFromConfig,
createUnstructuredOptionsFromConfig, createUnstructuredOptionsFromConfig,
} from './dataProcessCreateState' } from './dataProcessCreateState'
import { mapDataProcessSourceFile } from './useDataProcessSourceUpload' import {
loadCanonicalSourceContent,
mapDataProcessSourceFile,
} from './useDataProcessSourceUpload'
import type { import type {
PreviewItem, PreviewItem,
ProcessType, ProcessType,
@@ -50,24 +52,6 @@ interface RegenerationBindings {
resetDownstream: () => void resetDownstream: () => void
} }
async function loadSourceContent(taskId: string, fileId: string | number) {
const chunks: string[] = []
let startLine = 1
while (true) {
const source = await getDataProcessSourceContent(taskId, fileId, {
start_line: startLine,
line_count: 10_000,
})
chunks.push(source.content || '')
if (!source.has_more) break
const nextLine = Number(source.end_line || startLine) + 1
if (nextLine <= startLine) break
startLine = nextLine
}
// source_content_lines 已保留原始换行;分页之间直接拼接,避免凭空增加空行并破坏偏移。
return chunks.join('')
}
async function loadAllPreviews(taskId: string, mapPreviewItem: RegenerationBindings['mapPreviewItem']) { async function loadAllPreviews(taskId: string, mapPreviewItem: RegenerationBindings['mapPreviewItem']) {
const first = await getDataProcessPreview(taskId, { page: 1, page_size: 500 }) const first = await getDataProcessPreview(taskId, { page: 1, page_size: 500 })
const items = [...first.items] const items = [...first.items]
@@ -97,7 +81,7 @@ export function useDataProcessRegeneration(bindings: RegenerationBindings) {
async function hydrateWorkspace(task: DataProcessTask, preservePreviews: boolean) { async function hydrateWorkspace(task: DataProcessTask, preservePreviews: boolean) {
const taskId = String(task.id) const taskId = String(task.id)
bindings.uploadedFiles.value = await Promise.all((task.source_files || []).map(async (file) => ( bindings.uploadedFiles.value = await Promise.all((task.source_files || []).map(async (file) => (
mapDataProcessSourceFile(file, await loadSourceContent(taskId, file.id)) mapDataProcessSourceFile(file, await loadCanonicalSourceContent(taskId, file.id))
))) )))
bindings.previewItems.value = preservePreviews bindings.previewItems.value = preservePreviews
? await loadAllPreviews(taskId, bindings.mapPreviewItem) ? await loadAllPreviews(taskId, bindings.mapPreviewItem)

View File

@@ -6,7 +6,6 @@ import {
} from '@/api/modules/dataProcess' } from '@/api/modules/dataProcess'
import type { ProcessType, UploadedDataFile } from './types' import type { ProcessType, UploadedDataFile } from './types'
const BINARY_FILE_EXTENSIONS = new Set(['xlsx', 'pdf', 'docx', 'pptx'])
const STRUCTURED_FILE_EXTENSIONS = new Set(['json', 'jsonl', 'ndjson', 'csv', 'tsv', 'xlsx']) const STRUCTURED_FILE_EXTENSIONS = new Set(['json', 'jsonl', 'ndjson', 'csv', 'tsv', 'xlsx'])
const UNSTRUCTURED_FILE_EXTENSIONS = new Set([ const UNSTRUCTURED_FILE_EXTENSIONS = new Set([
'txt', 'md', 'markdown', 'pdf', 'docx', 'pptx', 'json', 'jsonl', 'ndjson', 'txt', 'md', 'markdown', 'pdf', 'docx', 'pptx', 'json', 'jsonl', 'ndjson',
@@ -15,11 +14,11 @@ const LEGACY_OFFICE_EXTENSIONS = new Set(['doc', 'xls', 'ppt'])
const MAX_SOURCE_FILE_BYTES = 200 * 1024 * 1024 const MAX_SOURCE_FILE_BYTES = 200 * 1024 * 1024
const MAX_SOURCE_FILE_COUNT = 20 const MAX_SOURCE_FILE_COUNT = 20
const MAX_SOURCE_BATCH_BYTES = 500 * 1024 * 1024 const MAX_SOURCE_BATCH_BYTES = 500 * 1024 * 1024
const SOURCE_CONTENT_PAGE_CHARS = 1_000_000
interface SourceUploadJob { interface SourceUploadJob {
uid: string uid: string
file: File file: File
extension: string
} }
interface SourceUploadOptions { interface SourceUploadOptions {
@@ -60,9 +59,6 @@ export function validateSourceFileSelection(
: '结构化数据支持 JSON、JSONL、NDJSON、CSV、TSV、XLSX', : '结构化数据支持 JSON、JSONL、NDJSON、CSV、TSV、XLSX',
} }
} }
if (selectedFiles.some((file) => file.name === raw.name && file.size === raw.size)) {
return { valid: false, severity: 'warning', message: '同名且同大小的文件已经选择' }
}
if (selectedFiles.length >= MAX_SOURCE_FILE_COUNT) { if (selectedFiles.length >= MAX_SOURCE_FILE_COUNT) {
return { valid: false, severity: 'warning', message: `每个任务最多选择 ${MAX_SOURCE_FILE_COUNT} 个文件` } return { valid: false, severity: 'warning', message: `每个任务最多选择 ${MAX_SOURCE_FILE_COUNT} 个文件` }
} }
@@ -73,6 +69,34 @@ export function validateSourceFileSelection(
return { valid: true, extension } return { valid: true, extension }
} }
function unicodeCodePointLength(value: string) {
let length = 0
for (const _character of value) length += 1
return length
}
/** 分页读取服务端保存的规范化正文,避免重新使用浏览器本地解码结果。 */
export async function loadCanonicalSourceContent(
taskId: string | number,
fileId: string | number,
) {
const chunks: string[] = []
let offset = 0
while (true) {
const source = await getDataProcessSourceContent(taskId, fileId, {
offset,
limit: SOURCE_CONTENT_PAGE_CHARS,
})
const content = source.content || ''
chunks.push(content)
if (!source.has_more) break
const nextOffset = Number(source.offset ?? offset) + unicodeCodePointLength(content)
if (nextOffset <= offset) throw new Error('服务端规范化内容分页异常,请删除文件后重试')
offset = nextOffset
}
return chunks.join('')
}
export function mapDataProcessSourceFile( export function mapDataProcessSourceFile(
file: DataProcessSourceFile, file: DataProcessSourceFile,
content = '', content = '',
@@ -126,16 +150,6 @@ export function useDataProcessSourceUpload(options: SourceUploadOptions) {
pending.uploadError = undefined pending.uploadError = undefined
try { try {
let content = ''
if (!BINARY_FILE_EXTENSIONS.has(job.extension)) {
try {
content = new TextDecoder('utf-8', { fatal: true }).decode(await job.file.arrayBuffer())
} catch {
throw new Error('文本文件不是有效的 UTF-8 编码,请转换编码后重试')
}
if (!content.trim()) throw new Error('不能上传空文件')
}
const uploaded = await uploadDataProcessSourceFiles(currentTaskId, [job.file], (progress) => { const uploaded = await uploadDataProcessSourceFiles(currentTaskId, [job.file], (progress) => {
pending.uploadProgress = progress pending.uploadProgress = progress
}) })
@@ -144,22 +158,13 @@ export function useDataProcessSourceUpload(options: SourceUploadOptions) {
// 先登记后端 ID确保正文读取失败时仍可正确删除已落库的文件。 // 先登记后端 ID确保正文读取失败时仍可正确删除已落库的文件。
Object.assign(pending, mapDataProcessSourceFile(source), { Object.assign(pending, mapDataProcessSourceFile(source), {
rawFile: job.file,
status: 'uploading', status: 'uploading',
uploadProgress: 99, uploadProgress: 99,
}) })
if (BINARY_FILE_EXTENSIONS.has(job.extension)) {
try { try {
const parsed = await getDataProcessSourceContent(currentTaskId, source.id, { pending.content = await loadCanonicalSourceContent(currentTaskId, source.id)
start_line: 1,
line_count: 10_000,
})
pending.content = parsed.content
} catch { } catch {
// 原文件已经成功落库,正文稍后仍可由预览构建接口读取,不重复上传。 throw new Error('文件已上传,但服务端规范化内容读取失败,请删除文件后重试')
}
} else {
pending.content = content
} }
pending.status = 'ready' pending.status = 'ready'

View File

@@ -0,0 +1,129 @@
<script setup lang="ts">
import { onMounted, ref } from 'vue'
import { useRoute, useRouter } from 'vue-router'
import { ElMessage } from 'element-plus'
import DataTablePage from '@/components/DataTablePage.vue'
import AclDialog from '@/components/AclDialog.vue'
import { getProject, getProjectMembers, addProjectMember, removeProjectMember, type Project, type ProjectMember } from '@/api/modules/project'
import { getUsers, type SystemUser } from '@/api/modules/system'
const route = useRoute()
const router = useRouter()
const project = ref<Project | null>(null)
const members = ref<ProjectMember[]>([])
const users = ref<SystemUser[]>([])
const loading = ref(false)
const aclVisible = ref(false)
const showAddMember = ref(false)
const addForm = ref({ user_id: '', role: 'member' })
async function load() {
const id = route.params.id as string
loading.value = true
try {
project.value = await getProject(id)
members.value = await getProjectMembers(id)
} finally {
loading.value = false
}
}
async function loadUsers() {
try {
users.value = await getUsers()
} catch {
users.value = []
}
}
async function submitAddMember() {
if (!project.value) return
if (!addForm.value.user_id) {
ElMessage.warning('请选择用户')
return
}
await addProjectMember(project.value.id, { ...addForm.value })
ElMessage.success('成员已添加')
showAddMember.value = false
addForm.value = { user_id: '', role: 'member' }
load()
}
async function removeMember(userId: string) {
if (!project.value) return
await removeProjectMember(project.value.id, userId)
ElMessage.success('已移除成员')
load()
}
onMounted(() => {
loadUsers()
load()
})
</script>
<template>
<div class="page">
<el-page-header title="返回" @back="router.back()">
<template #content>
<span class="page-title">项目详情{{ project?.name }}</span>
</template>
</el-page-header>
<el-card class="section" v-loading="loading">
<template #header>基本信息</template>
<el-descriptions :column="2" border>
<el-descriptions-item label="名称">{{ project?.name }}</el-descriptions-item>
<el-descriptions-item label="编码">{{ project?.code }}</el-descriptions-item>
<el-descriptions-item label="状态">{{ project?.status }}</el-descriptions-item>
<el-descriptions-item label="租户">{{ project?.tenant_id }}</el-descriptions-item>
<el-descriptions-item label="描述" :span="2">{{ project?.description }}</el-descriptions-item>
</el-descriptions>
<el-divider />
<el-button @click="aclVisible = true">资源授权 (ACL)</el-button>
</el-card>
<el-card class="section">
<template #header>
项目成员
<el-button type="primary" size="small" style="float: right" @click="showAddMember = true">添加成员</el-button>
</template>
<DataTablePage title="项目成员" :data="members">
<template #columns>
<el-table-column prop="username" label="用户名" min-width="140" />
<el-table-column prop="display_name" label="显示名" min-width="120" />
<el-table-column prop="role" label="角色" min-width="100" />
<el-table-column prop="create_time" label="加入时间" min-width="180" />
</template>
<template #actions="{ row }">
<el-button link type="danger" @click="removeMember(row.user_id)">移除</el-button>
</template>
</DataTablePage>
</el-card>
<AclDialog v-model="aclVisible" resource-type="project" :resource-id="(route.params.id as string)" />
<el-dialog v-model="showAddMember" title="添加成员" width="420px">
<el-form label-width="80px">
<el-form-item label="用户" required>
<el-select v-model="addForm.user_id" filterable style="width: 100%">
<el-option v-for="u in users" :key="u.id" :label="u.username" :value="u.id" />
</el-select>
</el-form-item>
<el-form-item label="角色">
<el-select v-model="addForm.role" style="width: 100%">
<el-option label="member" value="member" />
<el-option label="admin" value="admin" />
<el-option label="viewer" value="viewer" />
</el-select>
</el-form-item>
</el-form>
<template #footer>
<el-button @click="showAddMember = false">取消</el-button>
<el-button type="primary" @click="submitAddMember">确定</el-button>
</template>
</el-dialog>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
.section { margin-top: 16px; }
.page-title { font-size: 16px; font-weight: 600; }
</style>

View File

@@ -0,0 +1,109 @@
<script setup lang="ts">
import { computed, onMounted, ref } from 'vue'
import { useRouter } from 'vue-router'
import { ElMessage } from 'element-plus'
import { Plus } from '@element-plus/icons-vue'
import DataTablePage from '@/components/DataTablePage.vue'
import { createProject, getProjects, type Project } from '@/api/modules/project'
import { getTenants, type Tenant } from '@/api/modules/tenant'
const router = useRouter()
const loading = ref(false)
const projects = ref<Project[]>([])
const tenants = ref<Tenant[]>([])
const tenantId = ref('default')
const showCreate = ref(false)
const form = ref({ name: '', code: '', description: '', tenant_id: 'default' })
const tenantOptions = computed(() => [
{ label: 'default', value: 'default' },
...tenants.value.map((t) => ({ label: t.name, value: t.id })),
])
async function load() {
loading.value = true
try {
projects.value = await getProjects(tenantId.value)
} finally {
loading.value = false
}
}
async function loadTenants() {
try {
tenants.value = await getTenants()
} catch {
tenants.value = []
}
}
function openDetail(id: string) {
router.push(`/projects/${id}`)
}
async function submitCreate() {
if (!form.value.name || !form.value.code) {
ElMessage.warning('请填写项目名与编码')
return
}
await createProject({ ...form.value })
ElMessage.success('项目创建成功')
showCreate.value = false
form.value = { name: '', code: '', description: '', tenant_id: 'default' }
load()
}
onMounted(() => {
loadTenants()
load()
})
</script>
<template>
<div class="page">
<DataTablePage title="项目空间" :data="projects" :loading="loading" searchable search-fields="name,code">
<template #toolbar-extra>
<el-select v-model="tenantId" placeholder="租户" style="width: 160px" @change="load">
<el-option v-for="t in tenantOptions" :key="t.value" :label="t.label" :value="t.value" />
</el-select>
<el-button type="primary" :icon="Plus" @click="showCreate = true">新建项目</el-button>
</template>
<template #columns>
<el-table-column prop="name" label="项目名" min-width="140" />
<el-table-column prop="code" label="编码" min-width="100" />
<el-table-column prop="status" label="状态" min-width="100" />
<el-table-column prop="description" label="描述" min-width="200" show-overflow-tooltip />
<el-table-column prop="create_time" label="创建时间" min-width="180" />
</template>
<template #actions="{ row }">
<el-button link type="primary" @click="openDetail(row.id)">详情</el-button>
</template>
</DataTablePage>
<el-dialog v-model="showCreate" title="新建项目" width="520px">
<el-form label-width="90px">
<el-form-item label="名称" required>
<el-input v-model="form.name" placeholder="项目名" />
</el-form-item>
<el-form-item label="编码" required>
<el-input v-model="form.code" placeholder="project code" />
</el-form-item>
<el-form-item label="租户">
<el-select v-model="form.tenant_id" style="width: 100%">
<el-option v-for="t in tenantOptions" :key="t.value" :label="t.label" :value="t.value" />
</el-select>
</el-form-item>
<el-form-item label="描述">
<el-input v-model="form.description" type="textarea" :rows="3" />
</el-form-item>
</el-form>
<template #footer>
<el-button @click="showCreate = false">取消</el-button>
<el-button type="primary" @click="submitCreate">创建</el-button>
</template>
</el-dialog>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
</style>

View File

@@ -26,6 +26,7 @@ const trainContent = ref('')
// 搜索 // 搜索
const keyword = ref('') const keyword = ref('')
const level = ref('') // 日志级别筛选INFO/WARN/ERROR/空=全部
const fullContent = ref('') const fullContent = ref('')
// 自动刷新 // 自动刷新
@@ -33,11 +34,16 @@ const refreshInterval = ref(10)
const { remaining, start: startCountdown, stop: stopCountdown } = useCountdown(10) const { remaining, start: startCountdown, stop: stopCountdown } = useCountdown(10)
const filteredLog = computed(() => { const filteredLog = computed(() => {
if (!keyword.value.trim()) return { content: fullContent.value, count: 0 } let lines = fullContent.value.split('\n')
// 级别筛选
if (level.value) {
lines = lines.filter((line) => line.toUpperCase().includes(level.value.toUpperCase()))
}
// 关键词筛选
if (keyword.value.trim()) {
const kw = keyword.value.toLowerCase().trim() const kw = keyword.value.toLowerCase().trim()
const lines = fullContent.value lines = lines.filter((line) => line.toLowerCase().includes(kw))
.split('\n') }
.filter((line) => line.toLowerCase().includes(kw))
return { content: lines.join('\n'), count: lines.length } return { content: lines.join('\n'), count: lines.length }
}) })
@@ -180,7 +186,14 @@ onMounted(() => {
<el-input v-model="keyword" placeholder="搜索日志..." size="small" clearable style="width: 240px"> <el-input v-model="keyword" placeholder="搜索日志..." size="small" clearable style="width: 240px">
<template #prefix><i class="fa fa-search" /></template> <template #prefix><i class="fa fa-search" /></template>
</el-input> </el-input>
<span v-if="keyword" class="match-count">{{ matchCount }} 条匹配</span> <el-select v-model="level" placeholder="日志级别" size="small" clearable style="width: 120px">
<el-option value="" label="全部级别" />
<el-option value="INFO" label="INFO" />
<el-option value="WARN" label="WARN" />
<el-option value="ERROR" label="ERROR" />
<el-option value="DEBUG" label="DEBUG" />
</el-select>
<span v-if="keyword || level" class="match-count">{{ matchCount }} 条匹配</span>
</div> </div>
<pre class="log-pre">{{ filteredContent || (activeTab === 'system' ? sysContent : trainContent) || '日志内容将在这里显示...' }}</pre> <pre class="log-pre">{{ filteredContent || (activeTab === 'system' ? sysContent : trainContent) || '日志内容将在这里显示...' }}</pre>
</div> </div>

View File

@@ -1,12 +1,36 @@
<script setup lang="ts"> <script setup lang="ts">
import { onMounted, ref } from 'vue' import { computed, onMounted, reactive, ref } from 'vue'
import { getUsers } from '@/api/modules/system' import { ElMessage, ElMessageBox } from 'element-plus'
import type { SystemUser } from '@/types' import {
deleteUser,
getUsers,
resetUserPassword,
updateUserAccess,
} from '@/api/modules/system'
import type { PermissionCode, SystemUser, UserStatus } from '@/types'
import { statusLabel, statusTagType } from '@/utils/status' import { statusLabel, statusTagType } from '@/utils/status'
const loading = ref(false) const loading = ref(false)
const users = ref<SystemUser[]>([]) const users = ref<SystemUser[]>([])
// 权限码 -> 中文名(与路由模块一一对应)
const PERMISSION_LABELS: Record<PermissionCode, string> = {
dashboard: '服务看板',
'fine-tune': '模型训练',
'model-eval': '模型评测',
'model-inference': '模型推理',
'model-manage': '模型管理',
dataset: '数据集管理',
'data-process': '数据处理',
'data-convert': '数据转换',
compute: '计算资源',
hardware: '硬件监控',
logs: '日志中心',
'user-settings': '用户与权限',
}
const ALL_PERMISSIONS = Object.keys(PERMISSION_LABELS) as PermissionCode[]
async function loadUsers() { async function loadUsers() {
loading.value = true loading.value = true
try { try {
@@ -17,6 +41,109 @@ async function loadUsers() {
} }
onMounted(loadUsers) onMounted(loadUsers)
// 当前登录用户,用于禁止操作自身(避免误锁自己)
const currentUsername = ref<string>('')
try {
currentUsername.value = JSON.parse(localStorage.getItem('currentUser') || '{}').username || ''
} catch {
currentUsername.value = ''
}
function isSelf(row: SystemUser) {
return row.username === currentUsername.value
}
// ---------- 启停 ----------
async function toggleStatus(row: SystemUser, next: boolean) {
const nextStatus: UserStatus = next ? 'active' : 'disabled'
const prev = row.status
row.status = nextStatus
try {
await updateUserAccess(row.id, { status: nextStatus })
ElMessage.success(`${row.display_name}${next ? '启用' : '停用'}`)
await loadUsers()
} catch {
row.status = prev
ElMessage.error('状态更新失败')
}
}
// ---------- 重置密码 ----------
const pwdDialog = reactive({ visible: false, id: '', name: '', password: '', saving: false })
function openResetPwd(row: SystemUser) {
pwdDialog.id = row.id
pwdDialog.name = row.display_name
pwdDialog.password = 'Platform@123'
pwdDialog.visible = true
}
async function confirmResetPwd() {
if (!pwdDialog.password.trim()) {
ElMessage.warning('请输入新密码')
return
}
pwdDialog.saving = true
try {
await resetUserPassword(pwdDialog.id, pwdDialog.password.trim())
ElMessage.success(`已重置 ${pwdDialog.name} 的密码`)
pwdDialog.visible = false
} catch {
ElMessage.error('重置密码失败')
} finally {
pwdDialog.saving = false
}
}
// ---------- 页面权限 ----------
const permDialog = reactive({
visible: false,
id: '',
name: '',
checked: [] as PermissionCode[],
saving: false,
})
function openPerms(row: SystemUser) {
permDialog.id = row.id
permDialog.name = row.display_name
permDialog.checked = [...(row.permissions || [])]
permDialog.visible = true
}
async function confirmPerms() {
permDialog.saving = true
try {
await updateUserAccess(permDialog.id, { permissions: permDialog.checked })
ElMessage.success(`已更新 ${permDialog.name} 的页面权限`)
permDialog.visible = false
await loadUsers()
} catch {
ElMessage.error('权限更新失败')
} finally {
permDialog.saving = false
}
}
const permColumns = computed(() => ALL_PERMISSIONS)
// ---------- 删除 ----------
async function removeUser(row: SystemUser) {
try {
await ElMessageBox.confirm(
`确定删除用户 “${row.display_name}${row.username})” 吗?该操作不可恢复。`,
'删除用户',
{ type: 'warning', confirmButtonText: '删除', cancelButtonText: '取消' },
)
} catch {
return
}
try {
await deleteUser(row.id)
ElMessage.success(`已删除 ${row.display_name}`)
await loadUsers()
} catch (err: any) {
const msg = err?.response?.data?.message || '删除失败'
ElMessage.error(msg)
}
}
</script> </script>
<template> <template>
@@ -24,25 +151,93 @@ onMounted(loadUsers)
<header class="page-header"> <header class="page-header">
<div> <div>
<h1>用户设置</h1> <h1>用户设置</h1>
<p>管理平台账号角色状态页面权限</p> <p>管理平台账号角色状态登录密码与页面权限</p>
</div> </div>
<el-button type="primary" @click="$router.push('/user-settings/create')">创建用户</el-button> <el-button type="primary" @click="$router.push('/user-settings/create')">创建用户</el-button>
</header> </header>
<el-table :data="users"> <el-table :data="users" border>
<el-table-column prop="username" label="账号" min-width="140" /> <el-table-column prop="username" label="账号" min-width="140" />
<el-table-column prop="display_name" label="显示名称" min-width="160" /> <el-table-column prop="display_name" label="显示名称" min-width="160" />
<el-table-column prop="role" label="角色" width="120" /> <el-table-column prop="role" label="角色" width="120" />
<el-table-column label="状态" width="120"> <el-table-column label="状态" width="130">
<template #default="{ row }"> <template #default="{ row }">
<el-tag :type="statusTagType(row.status)" size="small">{{ statusLabel(row.status) }}</el-tag> <el-tag :type="statusTagType(row.status)" size="small">{{ statusLabel(row.status) }}</el-tag>
</template> </template>
</el-table-column> </el-table-column>
<el-table-column label="权限" width="120"> <el-table-column label="页面权限" min-width="160">
<template #default="{ row }">{{ row.permissions?.length || 0 }}</template> <template #default="{ row }">
<el-tag
v-for="p in (row.permissions || []).slice(0, 3)"
:key="p"
size="small"
type="info"
class="perm-tag"
>{{ PERMISSION_LABELS[p] || p }}</el-tag>
<span v-if="(row.permissions || []).length > 3" class="perm-more">
+{{ (row.permissions || []).length - 3 }}
</span>
<span v-if="!(row.permissions || []).length" class="perm-more"></span>
</template>
</el-table-column> </el-table-column>
<el-table-column prop="create_time" label="创建时间" min-width="180" /> <el-table-column prop="create_time" label="创建时间" min-width="180" />
<el-table-column label="操作" width="260" fixed="right">
<template #default="{ row }">
<el-switch
:model-value="row.status === 'active'"
:disabled="row.protected || isSelf(row)"
@change="(v: any) => toggleStatus(row, v)"
inline-prompt
active-text="启用"
inactive-text="停用"
/>
<el-button
link
type="primary"
:disabled="row.protected"
@click="openResetPwd(row)"
>重置密码</el-button>
<el-button
link
type="primary"
@click="openPerms(row)"
>页面权限</el-button>
<el-button
link
type="danger"
:disabled="row.protected || isSelf(row)"
@click="removeUser(row)"
>删除</el-button>
</template>
</el-table-column>
</el-table> </el-table>
<!-- 重置密码 -->
<el-dialog v-model="pwdDialog.visible" title="重置密码" width="420px">
<p class="dlg-tip"> <b>{{ pwdDialog.name }}</b> 设置新密码</p>
<el-input v-model="pwdDialog.password" placeholder="请输入新密码" show-password />
<template #footer>
<el-button @click="pwdDialog.visible = false">取消</el-button>
<el-button type="primary" :loading="pwdDialog.saving" @click="confirmResetPwd">确定重置</el-button>
</template>
</el-dialog>
<!-- 页面权限 -->
<el-dialog v-model="permDialog.visible" title="页面权限" width="540px">
<p class="dlg-tip"> <b>{{ permDialog.name }}</b> 分配可访问的页面模块</p>
<el-checkbox-group v-model="permDialog.checked" class="perm-group">
<el-checkbox
v-for="code in permColumns"
:key="code"
:value="code"
:label="PERMISSION_LABELS[code]"
/>
</el-checkbox-group>
<template #footer>
<el-button @click="permDialog.visible = false">取消</el-button>
<el-button type="primary" :loading="permDialog.saving" @click="confirmPerms">保存</el-button>
</template>
</el-dialog>
</section> </section>
</template> </template>
@@ -67,4 +262,25 @@ onMounted(loadUsers)
margin: 8px 0 0; margin: 8px 0 0;
color: #64748b; color: #64748b;
} }
.perm-tag {
margin-right: 4px;
margin-bottom: 2px;
}
.perm-more {
color: #94a3b8;
font-size: 12px;
}
.dlg-tip {
margin: 0 0 12px;
color: #475569;
}
.perm-group {
display: grid;
grid-template-columns: repeat(3, 1fr);
gap: 8px 12px;
}
</style> </style>

View File

@@ -0,0 +1,92 @@
<script setup lang="ts">
import { onMounted, ref } from 'vue'
import { useRoute, useRouter } from 'vue-router'
import { ElMessage } from 'element-plus'
import DataTablePage from '@/components/DataTablePage.vue'
import { getTenant, setTenantQuota, type Tenant } from '@/api/modules/tenant'
import { getProjects, type Project } from '@/api/modules/project'
const route = useRoute()
const router = useRouter()
const tenant = ref<Tenant | null>(null)
const projects = ref<Project[]>([])
const loading = ref(false)
const quotaText = ref('')
async function load() {
const id = route.params.id as string
loading.value = true
try {
tenant.value = await getTenant(id)
projects.value = await getProjects(id)
quotaText.value = JSON.stringify(tenant.value?.quota || {})
} finally {
loading.value = false
}
}
async function saveQuota() {
if (!tenant.value) return
try {
const q = JSON.parse(quotaText.value || '{}')
await setTenantQuota(tenant.value.id, q)
ElMessage.success('配额已保存')
load()
} catch {
ElMessage.error('配额需为合法 JSON')
}
}
function openProject(id: string) {
router.push(`/projects/${id}`)
}
onMounted(load)
</script>
<template>
<div class="page">
<el-page-header title="返回" @back="router.back()">
<template #content>
<span class="page-title">租户详情{{ tenant?.name }}</span>
</template>
</el-page-header>
<el-card class="section" v-loading="loading">
<template #header>基本信息</template>
<el-descriptions :column="2" border>
<el-descriptions-item label="名称">{{ tenant?.name }}</el-descriptions-item>
<el-descriptions-item label="编码">{{ tenant?.code }}</el-descriptions-item>
<el-descriptions-item label="状态">{{ tenant?.status }}</el-descriptions-item>
<el-descriptions-item label="创建时间">{{ tenant?.create_time }}</el-descriptions-item>
</el-descriptions>
<el-divider />
<div class="quota-edit">
<span class="label">配额 JSON</span>
<el-input v-model="quotaText" type="textarea" :rows="3" />
<el-button type="primary" @click="saveQuota">保存配额</el-button>
</div>
</el-card>
<el-card class="section">
<template #header>项目空间</template>
<DataTablePage title="项目空间" :data="projects">
<template #columns>
<el-table-column prop="name" label="项目名" min-width="140" />
<el-table-column prop="code" label="编码" min-width="100" />
<el-table-column prop="status" label="状态" min-width="100" />
<el-table-column prop="create_time" label="创建时间" min-width="180" />
</template>
<template #actions="{ row }">
<el-button link type="primary" @click="openProject(row.id)">打开</el-button>
</template>
</DataTablePage>
</el-card>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
.section { margin-top: 16px; }
.page-title { font-size: 16px; font-weight: 600; }
.quota-edit { display: flex; flex-direction: column; gap: 12px; max-width: 480px; }
.label { font-size: 13px; color: #606266; }
</style>

View File

@@ -0,0 +1,111 @@
<script setup lang="ts">
import { onMounted, ref } from 'vue'
import { useRouter } from 'vue-router'
import { ElMessage, ElMessageBox } from 'element-plus'
import { Plus } from '@element-plus/icons-vue'
import DataTablePage from '@/components/DataTablePage.vue'
import { createTenant, getTenants, setTenantQuota, type Tenant } from '@/api/modules/tenant'
const router = useRouter()
const loading = ref(false)
const tenants = ref<Tenant[]>([])
const showCreate = ref(false)
const form = ref({ name: '', code: '', quota: '' as string })
async function load() {
loading.value = true
try {
tenants.value = await getTenants()
} finally {
loading.value = false
}
}
function openDetail(id: string) {
router.push(`/tenants/${id}`)
}
async function submitCreate() {
if (!form.value.name) {
ElMessage.warning('请填写租户名称')
return
}
let quota: Record<string, unknown> = {}
if (form.value.quota) {
try {
quota = JSON.parse(form.value.quota)
} catch {
ElMessage.error('配额需为合法 JSON')
return
}
}
await createTenant({ name: form.value.name, code: form.value.code, quota })
ElMessage.success('租户创建成功')
showCreate.value = false
form.value = { name: '', code: '', quota: '' }
load()
}
async function setQuota(row: Tenant) {
const input = await ElMessageBox.prompt('输入租户配额 JSON', '设置配额', {
inputValue: JSON.stringify(row.quota || {}),
}).catch(() => null)
if (!input) return
try {
const q = JSON.parse(input.value)
await setTenantQuota(row.id, q)
ElMessage.success('配额已更新')
load()
} catch {
ElMessage.error('无效的 JSON')
}
}
onMounted(load)
</script>
<template>
<div class="page">
<DataTablePage title="租户管理" :data="tenants" :loading="loading">
<template #toolbar-extra>
<el-button type="primary" :icon="Plus" @click="showCreate = true">新建租户</el-button>
</template>
<template #columns>
<el-table-column prop="name" label="租户名称" min-width="140" />
<el-table-column prop="code" label="编码" min-width="100" />
<el-table-column label="配额" min-width="160">
<template #default="{ row }">
{{ Object.keys(row.quota || {}).length ? JSON.stringify(row.quota) : '—' }}
</template>
</el-table-column>
<el-table-column prop="status" label="状态" min-width="100" />
<el-table-column prop="create_time" label="创建时间" min-width="180" />
</template>
<template #actions="{ row }">
<el-button link type="primary" @click="openDetail(row.id)">详情</el-button>
<el-button link type="primary" @click="setQuota(row)">配额</el-button>
</template>
</DataTablePage>
<el-dialog v-model="showCreate" title="新建租户" width="520px">
<el-form label-width="90px">
<el-form-item label="名称" required>
<el-input v-model="form.name" placeholder="租户名称" />
</el-form-item>
<el-form-item label="编码">
<el-input v-model="form.code" placeholder="tenant code" />
</el-form-item>
<el-form-item label="配额 JSON">
<el-input v-model="form.quota" type="textarea" :rows="3" placeholder='{"gpu": 8}' />
</el-form-item>
</el-form>
<template #footer>
<el-button @click="showCreate = false">取消</el-button>
<el-button type="primary" @click="submitCreate">创建</el-button>
</template>
</el-dialog>
</div>
</template>
<style scoped lang="scss">
.page { padding: 16px; }
</style>