2026-07-28 19:34:41 +08:00
|
|
|
|
"""
|
|
|
|
|
|
Evaluation runner — executes model evaluation as a subprocess job.
|
|
|
|
|
|
|
|
|
|
|
|
Usage:
|
|
|
|
|
|
python -m compute.engines.llama_factory.eval_runner --config <config_json_path>
|
|
|
|
|
|
|
|
|
|
|
|
The config JSON is written by the compute API before spawning this subprocess.
|
|
|
|
|
|
Results are written to ``output_dir/eval_results.json`` and progress is printed
|
|
|
|
|
|
to stdout (captured as job logs).
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
|
|
import json
|
|
|
|
|
|
import math
|
2026-08-03 15:49:21 +08:00
|
|
|
|
import re
|
2026-07-28 19:34:41 +08:00
|
|
|
|
import sys
|
|
|
|
|
|
import time
|
2026-08-21 09:49:48 +08:00
|
|
|
|
from datetime import datetime, timezone
|
2026-08-03 15:49:21 +08:00
|
|
|
|
from difflib import SequenceMatcher
|
2026-07-28 19:34:41 +08:00
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
from typing import Any
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-03 15:49:21 +08:00
|
|
|
|
def _load_dataset(path: str) -> list[dict[str, Any]]:
|
2026-08-07 09:24:35 +08:00
|
|
|
|
"""Load a JSON or JSONL dataset file (jsonl-compatible).
|
|
|
|
|
|
|
|
|
|
|
|
Content-sniffs instead of trusting the extension so jsonl files with a BOM,
|
|
|
|
|
|
a single JSON array on one line, or mislabeled extensions all load correctly.
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
Supports common field names used across the platform:
|
|
|
|
|
|
* ``instruction`` + ``input`` + ``output`` (Alpaca-style)
|
|
|
|
|
|
* ``question`` + ``answer``
|
|
|
|
|
|
* ``messages`` (ShareGPT-style – the last assistant message is treated as reference)
|
|
|
|
|
|
"""
|
2026-08-03 15:49:21 +08:00
|
|
|
|
file_path = Path(path)
|
2026-08-07 09:24:35 +08:00
|
|
|
|
text = file_path.read_text(encoding="utf-8-sig", errors="replace").strip()
|
2026-08-03 15:49:21 +08:00
|
|
|
|
if not text:
|
|
|
|
|
|
return []
|
2026-08-07 09:24:35 +08:00
|
|
|
|
try:
|
2026-08-03 15:49:21 +08:00
|
|
|
|
value = json.loads(text)
|
2026-08-07 09:24:35 +08:00
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
|
value = None
|
|
|
|
|
|
if isinstance(value, list):
|
|
|
|
|
|
return [item for item in value if isinstance(item, dict)]
|
|
|
|
|
|
if isinstance(value, dict):
|
|
|
|
|
|
return [value]
|
2026-08-03 15:49:21 +08:00
|
|
|
|
|
2026-07-28 19:34:41 +08:00
|
|
|
|
samples: list[dict[str, Any]] = []
|
2026-08-03 15:49:21 +08:00
|
|
|
|
for line in text.splitlines():
|
|
|
|
|
|
line = line.strip()
|
|
|
|
|
|
if not line:
|
|
|
|
|
|
continue
|
|
|
|
|
|
try:
|
|
|
|
|
|
obj = json.loads(line)
|
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
|
continue
|
|
|
|
|
|
if isinstance(obj, dict):
|
2026-07-28 19:34:41 +08:00
|
|
|
|
samples.append(obj)
|
|
|
|
|
|
return samples
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _sample_question(sample: dict[str, Any]) -> str:
|
|
|
|
|
|
"""Extract the user-facing question / instruction from a sample."""
|
|
|
|
|
|
if sample.get("instruction"):
|
|
|
|
|
|
text = sample["instruction"]
|
|
|
|
|
|
if sample.get("input"):
|
|
|
|
|
|
text += "\n" + sample["input"]
|
|
|
|
|
|
return text
|
|
|
|
|
|
if sample.get("question"):
|
|
|
|
|
|
return sample["question"]
|
|
|
|
|
|
# ShareGPT-style: use the last user message as question
|
|
|
|
|
|
messages = sample.get("messages") or []
|
|
|
|
|
|
user_msgs = [m["content"] for m in messages if m.get("role") == "user"]
|
|
|
|
|
|
return user_msgs[-1] if user_msgs else ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _sample_reference(sample: dict[str, Any]) -> str:
|
|
|
|
|
|
"""Extract the reference answer from a sample."""
|
|
|
|
|
|
if sample.get("output"):
|
|
|
|
|
|
return sample["output"]
|
|
|
|
|
|
if sample.get("answer"):
|
|
|
|
|
|
return sample["answer"]
|
|
|
|
|
|
messages = sample.get("messages") or []
|
|
|
|
|
|
assistant_msgs = [m["content"] for m in messages if m.get("role") == "assistant"]
|
|
|
|
|
|
return assistant_msgs[-1] if assistant_msgs else ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# Basic metrics
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
def _compute_bleu(references: list[str], predictions: list[str], ngram: int = 4) -> dict[str, Any]:
|
|
|
|
|
|
"""Compute BLEU score via sacrebleu (corpus-level)."""
|
|
|
|
|
|
try:
|
|
|
|
|
|
from sacrebleu.metrics import BLEU
|
|
|
|
|
|
except ImportError:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(None, len(predictions), "sacrebleu 未安装", available=False)
|
|
|
|
|
|
if not predictions:
|
|
|
|
|
|
return _metric_record(0, 0)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
bleu = BLEU(max_ngram_order=ngram)
|
|
|
|
|
|
# sacrebleu expects list-of-strings; we have one reference per prediction
|
|
|
|
|
|
score = bleu.corpus_score(predictions, [references])
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(score.score, len(predictions), bleu=round(score.score, 2))
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _compute_rouge(references: list[str], predictions: list[str], methods: list[str] | None = None) -> dict[str, Any]:
|
|
|
|
|
|
"""Compute ROUGE scores via rouge-score."""
|
|
|
|
|
|
try:
|
|
|
|
|
|
from rouge_score import rouge_scorer
|
|
|
|
|
|
except ImportError:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
rouge_scorer = None
|
2026-07-28 19:34:41 +08:00
|
|
|
|
methods = methods or ["rouge1", "rouge2", "rougeL"]
|
|
|
|
|
|
# Normalize: map "rouge_1"/"rouge1" → "rouge1", "rouge_l"/"rougeL" → "rougeL"
|
|
|
|
|
|
_rouge_aliases = {"rouge_1": "rouge1", "rouge_2": "rouge2", "rouge_l": "rougeL"}
|
|
|
|
|
|
methods = [_rouge_aliases.get(m, m.replace("_", "")) for m in methods]
|
2026-08-21 09:49:48 +08:00
|
|
|
|
if rouge_scorer is None:
|
|
|
|
|
|
values = [_pair_rouge(ref, pred) for ref, pred in zip(references, predictions)]
|
|
|
|
|
|
avg = {key: round(sum(item.get(key, 0.0) for item in values) / max(len(values), 1), 4) for key in methods}
|
|
|
|
|
|
else:
|
|
|
|
|
|
class _Tokenizer:
|
|
|
|
|
|
def tokenize(self, value: str) -> list[str]:
|
|
|
|
|
|
return value.split()
|
|
|
|
|
|
scorer = rouge_scorer.RougeScorer(methods, use_stemmer=False, tokenizer=_Tokenizer())
|
|
|
|
|
|
totals: dict[str, float] = {}
|
|
|
|
|
|
n = max(len(predictions), 1)
|
|
|
|
|
|
for ref, pred in zip(references, predictions):
|
|
|
|
|
|
result = scorer.score(_rouge_tokens(ref), _rouge_tokens(pred))
|
|
|
|
|
|
for key in methods:
|
|
|
|
|
|
totals[key] = totals.get(key, 0) + result[key].fmeasure
|
|
|
|
|
|
avg = {key: round(value / n, 4) for key, value in totals.items()}
|
|
|
|
|
|
return _metric_record(avg.get("rougeL", avg.get("rouge1", 0)) * 100, len(predictions), **avg)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _compute_cosine(references: list[str], predictions: list[str]) -> dict[str, Any]:
|
|
|
|
|
|
"""Compute average cosine similarity via sklearn."""
|
|
|
|
|
|
try:
|
|
|
|
|
|
from sklearn.feature_extraction.text import TfidfVectorizer
|
|
|
|
|
|
from sklearn.metrics.pairwise import cosine_similarity
|
|
|
|
|
|
except ImportError:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(None, len(predictions), "scikit-learn 未安装", available=False)
|
|
|
|
|
|
if not predictions:
|
|
|
|
|
|
return _metric_record(0, 0)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
try:
|
|
|
|
|
|
vectorizer = TfidfVectorizer()
|
|
|
|
|
|
tfidf = vectorizer.fit_transform(references + predictions)
|
|
|
|
|
|
n = len(references)
|
|
|
|
|
|
ref_vec = tfidf[:n]
|
|
|
|
|
|
pred_vec = tfidf[n:]
|
|
|
|
|
|
sims = cosine_similarity(ref_vec, pred_vec).diagonal()
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(float(sims.mean()) * 100, n)
|
|
|
|
|
|
except ValueError as exc:
|
|
|
|
|
|
return _metric_record(0, len(predictions), str(exc))
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-08-03 15:49:21 +08:00
|
|
|
|
def _normalize_text(value: str) -> str:
|
|
|
|
|
|
return re.sub(r"\s+", " ", str(value or "").strip().lower())
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
def _metric_record(score: float | None, sample_count: int, error: str = "", available: bool = True, **extra: Any) -> dict[str, Any]:
|
|
|
|
|
|
return {
|
|
|
|
|
|
"enabled": True,
|
|
|
|
|
|
"available": available,
|
|
|
|
|
|
"score": None if score is None else round(max(0.0, min(100.0, float(score))), 2),
|
|
|
|
|
|
"max_score": 100,
|
|
|
|
|
|
"unit": "percent",
|
|
|
|
|
|
"sample_count": sample_count,
|
|
|
|
|
|
"error": error,
|
|
|
|
|
|
**extra,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 10:48:58 +08:00
|
|
|
|
def _metric_dimension_summary(metrics: dict[str, Any]) -> list[dict[str, Any]]:
|
|
|
|
|
|
labels = {
|
|
|
|
|
|
"bleu": "BLEU",
|
|
|
|
|
|
"rouge": "ROUGE-L",
|
|
|
|
|
|
"cosine": "Cosine 相似度",
|
|
|
|
|
|
"exact_match": "精确匹配",
|
|
|
|
|
|
"text_similarity": "文本相似度",
|
|
|
|
|
|
}
|
|
|
|
|
|
result: list[dict[str, Any]] = []
|
|
|
|
|
|
for name, item in metrics.items():
|
|
|
|
|
|
if not isinstance(item, dict) or item.get("score") is None:
|
|
|
|
|
|
continue
|
|
|
|
|
|
result.append({
|
|
|
|
|
|
"name": labels.get(name, name),
|
|
|
|
|
|
"score": float(item.get("score") or 0),
|
|
|
|
|
|
"max_score": float(item.get("max_score") or 100),
|
|
|
|
|
|
"pass_rate": float(item.get("score") or 0),
|
|
|
|
|
|
"sample_count": int(item.get("sample_count") or 0),
|
|
|
|
|
|
"available": item.get("available", True),
|
|
|
|
|
|
"error": item.get("error", ""),
|
|
|
|
|
|
})
|
|
|
|
|
|
return result
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
def _rouge_tokens(text: str) -> str:
|
|
|
|
|
|
text = str(text or "").strip().lower()
|
|
|
|
|
|
tokens: list[str] = []
|
|
|
|
|
|
for segment in re.findall(r"[一-鿿]+|[^\s一-鿿]+", text):
|
|
|
|
|
|
tokens.extend(segment if "一" <= segment[0] <= "鿿" else [segment])
|
|
|
|
|
|
return " ".join(tokens)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _pair_rouge(reference: str, prediction: str) -> dict[str, float]:
|
|
|
|
|
|
try:
|
|
|
|
|
|
from rouge_score import rouge_scorer
|
|
|
|
|
|
except ImportError:
|
|
|
|
|
|
reference_tokens = _rouge_tokens(reference).split()
|
|
|
|
|
|
prediction_tokens = _rouge_tokens(prediction).split()
|
|
|
|
|
|
if not reference_tokens or not prediction_tokens:
|
|
|
|
|
|
return {"rouge1": 0.0, "rouge2": 0.0, "rougeL": 0.0}
|
|
|
|
|
|
|
|
|
|
|
|
def f1(overlap: int, reference_count: int, prediction_count: int) -> float:
|
|
|
|
|
|
if not reference_count or not prediction_count or not overlap:
|
|
|
|
|
|
return 0.0
|
|
|
|
|
|
precision = overlap / prediction_count
|
|
|
|
|
|
recall = overlap / reference_count
|
|
|
|
|
|
return 2 * precision * recall / (precision + recall)
|
|
|
|
|
|
|
|
|
|
|
|
from collections import Counter
|
|
|
|
|
|
reference_unigrams = Counter(reference_tokens)
|
|
|
|
|
|
prediction_unigrams = Counter(prediction_tokens)
|
|
|
|
|
|
unigram_overlap = sum((reference_unigrams & prediction_unigrams).values())
|
|
|
|
|
|
reference_bigrams = Counter(zip(reference_tokens, reference_tokens[1:]))
|
|
|
|
|
|
prediction_bigrams = Counter(zip(prediction_tokens, prediction_tokens[1:]))
|
|
|
|
|
|
bigram_overlap = sum((reference_bigrams & prediction_bigrams).values())
|
|
|
|
|
|
matrix = [[0] * (len(prediction_tokens) + 1) for _ in range(len(reference_tokens) + 1)]
|
|
|
|
|
|
for row, reference_token in enumerate(reference_tokens, start=1):
|
|
|
|
|
|
for column, prediction_token in enumerate(prediction_tokens, start=1):
|
|
|
|
|
|
matrix[row][column] = matrix[row - 1][column - 1] + 1 if reference_token == prediction_token else max(matrix[row - 1][column], matrix[row][column - 1])
|
|
|
|
|
|
return {
|
|
|
|
|
|
"rouge1": f1(unigram_overlap, len(reference_tokens), len(prediction_tokens)),
|
|
|
|
|
|
"rouge2": f1(bigram_overlap, max(len(reference_tokens) - 1, 0), max(len(prediction_tokens) - 1, 0)),
|
|
|
|
|
|
"rougeL": f1(matrix[-1][-1], len(reference_tokens), len(prediction_tokens)),
|
|
|
|
|
|
}
|
|
|
|
|
|
class _Tokenizer:
|
|
|
|
|
|
def tokenize(self, value: str) -> list[str]:
|
|
|
|
|
|
return value.split()
|
|
|
|
|
|
if not reference.strip() or not prediction.strip():
|
|
|
|
|
|
return {"rouge1": 0.0, "rouge2": 0.0, "rougeL": 0.0}
|
|
|
|
|
|
scorer = rouge_scorer.RougeScorer(("rouge1", "rouge2", "rougeL"), use_stemmer=False, tokenizer=_Tokenizer())
|
|
|
|
|
|
scores = scorer.score(_rouge_tokens(reference), _rouge_tokens(prediction))
|
|
|
|
|
|
return {key: float(value.fmeasure) for key, value in scores.items()}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-03 15:49:21 +08:00
|
|
|
|
def _compute_exact_match(references: list[str], predictions: list[str]) -> dict[str, Any]:
|
|
|
|
|
|
total = len(predictions)
|
|
|
|
|
|
if not total:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(0, 0, matched=0, total=0)
|
2026-08-03 15:49:21 +08:00
|
|
|
|
matched = sum(
|
|
|
|
|
|
1
|
|
|
|
|
|
for ref, pred in zip(references, predictions)
|
|
|
|
|
|
if _normalize_text(ref) == _normalize_text(pred)
|
|
|
|
|
|
)
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(matched / total * 100, total, matched=matched, total=total)
|
2026-08-03 15:49:21 +08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _compute_text_similarity(references: list[str], predictions: list[str]) -> dict[str, Any]:
|
|
|
|
|
|
if not predictions:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(0, 0)
|
2026-08-03 15:49:21 +08:00
|
|
|
|
scores = [
|
|
|
|
|
|
SequenceMatcher(None, _normalize_text(ref), _normalize_text(pred)).ratio()
|
|
|
|
|
|
for ref, pred in zip(references, predictions)
|
|
|
|
|
|
]
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return _metric_record(sum(scores) / max(len(scores), 1) * 100, len(predictions))
|
2026-08-03 15:49:21 +08:00
|
|
|
|
|
|
|
|
|
|
|
2026-07-28 19:34:41 +08:00
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# LLM Judge
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
def _normalise_api_url(value: str) -> str:
|
|
|
|
|
|
url = str(value or "").strip().rstrip("/")
|
|
|
|
|
|
return url + "/chat/completions" if url.endswith("/v1") else url + "/v1/chat/completions"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _parse_judge_reply(reply: str, score_min: float, score_max: float) -> tuple[float | None, dict[str, Any], str]:
|
|
|
|
|
|
payload: dict[str, Any] = {}
|
|
|
|
|
|
match = re.search(r"\{[\s\S]*\}", str(reply or ""))
|
|
|
|
|
|
if match:
|
|
|
|
|
|
try:
|
|
|
|
|
|
value = json.loads(match.group(0))
|
|
|
|
|
|
if isinstance(value, dict):
|
|
|
|
|
|
payload = value
|
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
|
pass
|
|
|
|
|
|
reason = str(payload.get("综合评价") or payload.get("reason") or reply or "")
|
|
|
|
|
|
raw = payload.get("score", payload.get("overall_score"))
|
|
|
|
|
|
if raw is None:
|
|
|
|
|
|
matches = re.findall(r"(?:得分|分数|评分|score|分)[^\d]*(\d+(?:\.\d+)?)", reason, re.IGNORECASE)
|
|
|
|
|
|
raw = matches[-1] if matches else None
|
|
|
|
|
|
try:
|
|
|
|
|
|
numeric = float(raw)
|
|
|
|
|
|
except (TypeError, ValueError):
|
|
|
|
|
|
dimensions = payload.get("dimensions") if isinstance(payload.get("dimensions"), dict) else {}
|
|
|
|
|
|
if not dimensions:
|
|
|
|
|
|
dimensions = {
|
|
|
|
|
|
key: value
|
|
|
|
|
|
for key, value in payload.items()
|
|
|
|
|
|
if key not in {"score", "overall_score", "综合评价", "reason"}
|
|
|
|
|
|
}
|
|
|
|
|
|
values = [float(value) for value in dimensions.values() if isinstance(value, (int, float))]
|
|
|
|
|
|
numeric = sum(values) / len(values) if values else None
|
|
|
|
|
|
if numeric is None:
|
|
|
|
|
|
return None, payload, reason
|
|
|
|
|
|
# The judge prompt uses the configured score range. Do not treat a
|
|
|
|
|
|
# decimal such as 0.5/5 as a 0.5/1 score, otherwise low scores are
|
|
|
|
|
|
# incorrectly inflated (0.5/5 would become 50 instead of 10).
|
|
|
|
|
|
normalized = (numeric - score_min) / max(score_max - score_min, 1e-9) * 100
|
|
|
|
|
|
return max(0.0, min(100.0, normalized)), payload, reason
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-07-28 19:34:41 +08:00
|
|
|
|
def _judge_sample(
|
|
|
|
|
|
question: str,
|
|
|
|
|
|
reference: str,
|
|
|
|
|
|
prediction: str,
|
|
|
|
|
|
config: dict[str, Any],
|
|
|
|
|
|
) -> dict[str, Any]:
|
|
|
|
|
|
"""Call an OpenAI-compatible LLM to judge a single sample.
|
|
|
|
|
|
|
|
|
|
|
|
Returns a dict with keys:
|
|
|
|
|
|
score, max_score, passed, judgement, evaluation_reason, error_type
|
|
|
|
|
|
"""
|
|
|
|
|
|
api_url = (config.get("api_url") or "").strip().rstrip("/")
|
|
|
|
|
|
api_key = (config.get("api_key") or "").strip()
|
|
|
|
|
|
eval_model = (config.get("eval_model") or "").strip()
|
2026-08-04 18:21:16 +08:00
|
|
|
|
# 优先使用模型记录里配置的真实 API 模型名(如 deepseek-chat),
|
|
|
|
|
|
# 否则回退到平台内部模型名
|
|
|
|
|
|
api_model = (config.get("api_model") or "").strip() or eval_model
|
2026-07-28 19:34:41 +08:00
|
|
|
|
eval_prompt = (config.get("eval_prompt") or "").strip()
|
|
|
|
|
|
score_min = float(config.get("score_min", 0))
|
|
|
|
|
|
score_max = float(config.get("score_max", 5))
|
|
|
|
|
|
pass_threshold = float(config.get("pass_threshold", 3))
|
|
|
|
|
|
|
|
|
|
|
|
if not api_url or not eval_model:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return {"available": False, "score": None, "max_score": 100, "passed": False, "judgement": "未配置",
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"evaluation_reason": "未配置评测模型", "error_type": "其他"}
|
|
|
|
|
|
|
|
|
|
|
|
system_msg = (
|
|
|
|
|
|
eval_prompt
|
|
|
|
|
|
or "你是一个专业的评测专家。请根据参考答-案对被测模型的输出进行评分。"
|
|
|
|
|
|
)
|
|
|
|
|
|
user_msg = (
|
|
|
|
|
|
f"## 问题\n{question}\n\n"
|
|
|
|
|
|
f"## 参考答案\n{reference}\n\n"
|
|
|
|
|
|
f"## 模型输出\n{prediction}\n\n"
|
|
|
|
|
|
f"请给出 {score_min}-{score_max} 分的评分,并说明理由。"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
try:
|
|
|
|
|
|
import urllib.request
|
|
|
|
|
|
import urllib.error
|
|
|
|
|
|
|
|
|
|
|
|
body = json.dumps({
|
2026-08-04 18:21:16 +08:00
|
|
|
|
"model": api_model,
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"messages": [
|
|
|
|
|
|
{"role": "system", "content": system_msg},
|
|
|
|
|
|
{"role": "user", "content": user_msg},
|
|
|
|
|
|
],
|
|
|
|
|
|
"temperature": 0.3,
|
|
|
|
|
|
"max_tokens": 512,
|
|
|
|
|
|
}).encode("utf-8")
|
|
|
|
|
|
|
|
|
|
|
|
req = urllib.request.Request(
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_normalise_api_url(api_url),
|
2026-07-28 19:34:41 +08:00
|
|
|
|
data=body,
|
|
|
|
|
|
headers={
|
|
|
|
|
|
"Content-Type": "application/json",
|
|
|
|
|
|
"Authorization": f"Bearer {api_key}",
|
|
|
|
|
|
},
|
|
|
|
|
|
)
|
|
|
|
|
|
resp = urllib.request.urlopen(req, timeout=120)
|
|
|
|
|
|
data = json.loads(resp.read().decode("utf-8"))
|
|
|
|
|
|
reply = data["choices"][0]["message"]["content"]
|
|
|
|
|
|
except Exception as exc:
|
2026-08-21 09:49:48 +08:00
|
|
|
|
return {"available": False, "score": None, "max_score": 100, "passed": False,
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"judgement": "错误", "evaluation_reason": f"评测模型调用失败: {exc}",
|
|
|
|
|
|
"error_type": "其他"}
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
parsed: dict[str, Any] = {}
|
|
|
|
|
|
match = re.search(r"\{[\s\S]*\}", reply)
|
|
|
|
|
|
if match:
|
|
|
|
|
|
try:
|
|
|
|
|
|
value = json.loads(match.group(0))
|
|
|
|
|
|
if isinstance(value, dict):
|
|
|
|
|
|
parsed = value
|
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
|
pass
|
|
|
|
|
|
raw_score = parsed.get("score", parsed.get("overall_score"))
|
|
|
|
|
|
reason = str(parsed.get("综合评价") or parsed.get("reason") or reply)
|
|
|
|
|
|
if raw_score is None:
|
|
|
|
|
|
matches = re.findall(r'(?:得分|分数|评分|score|分)[^\d]*(\d+(?:\.\d+)?)', reason, re.IGNORECASE)
|
|
|
|
|
|
raw_score = matches[-1] if matches else None
|
|
|
|
|
|
try:
|
|
|
|
|
|
numeric_score = float(raw_score)
|
|
|
|
|
|
except (TypeError, ValueError):
|
|
|
|
|
|
dimensions = parsed.get("dimensions") if isinstance(parsed.get("dimensions"), dict) else {}
|
|
|
|
|
|
if not dimensions:
|
|
|
|
|
|
dimensions = {
|
|
|
|
|
|
key: value
|
|
|
|
|
|
for key, value in parsed.items()
|
|
|
|
|
|
if key not in {"score", "overall_score", "综合评价", "reason"}
|
|
|
|
|
|
}
|
|
|
|
|
|
values = [float(value) for value in dimensions.values() if isinstance(value, (int, float))]
|
|
|
|
|
|
numeric_score = sum(values) / len(values) if values else None
|
|
|
|
|
|
if numeric_score is None:
|
|
|
|
|
|
return {"available": False, "score": None, "max_score": 100, "passed": False, "judgement": "错误",
|
|
|
|
|
|
"evaluation_reason": "评测模型未返回可解析分数:" + reason[:1800], "error_type": "其他"}
|
|
|
|
|
|
normalized_score = (numeric_score - score_min) / max(score_max - score_min, 1e-9) * 100
|
|
|
|
|
|
normalized_score = max(0, min(100, normalized_score))
|
|
|
|
|
|
normalized_threshold = (pass_threshold - score_min) / max(score_max - score_min, 1e-9) * 100
|
|
|
|
|
|
passed = normalized_score >= normalized_threshold
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
# Determine judgement label
|
2026-08-21 09:49:48 +08:00
|
|
|
|
if normalized_score >= min(100, normalized_threshold + 20):
|
2026-07-28 19:34:41 +08:00
|
|
|
|
judgement = "正确"
|
2026-08-21 09:49:48 +08:00
|
|
|
|
elif passed:
|
2026-07-28 19:34:41 +08:00
|
|
|
|
judgement = "部分正确"
|
|
|
|
|
|
else:
|
|
|
|
|
|
judgement = "错误"
|
|
|
|
|
|
|
|
|
|
|
|
# Guess error type from reply
|
2026-08-21 09:49:48 +08:00
|
|
|
|
reply_lower = (reply + " " + reason).lower()
|
2026-07-28 19:34:41 +08:00
|
|
|
|
if any(w in reply_lower for w in ["幻觉", "hallucination", "编造"]):
|
|
|
|
|
|
error_type = "幻觉"
|
|
|
|
|
|
elif any(w in reply_lower for w in ["不完整", "incomplete", "遗漏"]):
|
|
|
|
|
|
error_type = "不完整"
|
|
|
|
|
|
elif any(w in reply_lower for w in ["格式", "format"]):
|
|
|
|
|
|
error_type = "格式偏差"
|
|
|
|
|
|
elif any(w in reply_lower for w in ["混淆", "confusion", "错误"]):
|
|
|
|
|
|
error_type = "混淆"
|
|
|
|
|
|
else:
|
|
|
|
|
|
error_type = "其他"
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
parsed_dimensions = parsed.get("dimensions") if isinstance(parsed.get("dimensions"), dict) else {
|
|
|
|
|
|
key: value
|
|
|
|
|
|
for key, value in parsed.items()
|
|
|
|
|
|
if key not in {"score", "overall_score", "综合评价", "reason"}
|
|
|
|
|
|
}
|
2026-07-28 19:34:41 +08:00
|
|
|
|
return {
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"available": True,
|
|
|
|
|
|
"score": round(normalized_score, 2),
|
|
|
|
|
|
"max_score": 100,
|
|
|
|
|
|
"raw_score": numeric_score,
|
|
|
|
|
|
"raw_max_score": score_max,
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"passed": passed,
|
|
|
|
|
|
"judgement": judgement,
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"evaluation_reason": reason[:2000],
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"error_type": error_type,
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"dimension_scores": [
|
|
|
|
|
|
{
|
|
|
|
|
|
"name": str(name),
|
|
|
|
|
|
"score": round(
|
|
|
|
|
|
max(
|
|
|
|
|
|
0,
|
|
|
|
|
|
min(
|
|
|
|
|
|
100,
|
|
|
|
|
|
(float(value) - score_min)
|
|
|
|
|
|
/ max(score_max - score_min, 1e-9)
|
|
|
|
|
|
* 100,
|
|
|
|
|
|
),
|
|
|
|
|
|
),
|
|
|
|
|
|
2,
|
|
|
|
|
|
),
|
|
|
|
|
|
"max_score": 100,
|
|
|
|
|
|
}
|
|
|
|
|
|
for name, value in parsed_dimensions.items()
|
|
|
|
|
|
if isinstance(value, (int, float))
|
|
|
|
|
|
],
|
2026-07-28 19:34:41 +08:00
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 09:49:48 +08:00
|
|
|
|
def _write_json(path: Path, payload: dict[str, Any]) -> None:
|
|
|
|
|
|
temporary = path.with_suffix(path.suffix + ".part")
|
|
|
|
|
|
temporary.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
|
|
|
|
temporary.replace(path)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _write_eval_progress(
|
|
|
|
|
|
output_dir: Path,
|
|
|
|
|
|
status: str,
|
|
|
|
|
|
stage: str,
|
|
|
|
|
|
total: int,
|
|
|
|
|
|
completed: int,
|
|
|
|
|
|
message: str = "",
|
|
|
|
|
|
current_index: int = 0,
|
|
|
|
|
|
) -> None:
|
|
|
|
|
|
_write_json(
|
|
|
|
|
|
output_dir / "eval_progress.json",
|
|
|
|
|
|
{
|
|
|
|
|
|
"status": status,
|
|
|
|
|
|
"stage": stage,
|
|
|
|
|
|
"total": total,
|
|
|
|
|
|
"completed": completed,
|
|
|
|
|
|
"percentage": round(completed / total * 100, 1) if total else 100,
|
|
|
|
|
|
"current_index": current_index,
|
|
|
|
|
|
"message": message,
|
|
|
|
|
|
"updated_at": datetime.now(timezone.utc).isoformat(),
|
|
|
|
|
|
},
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _deterministic_sample_score(reference: str, prediction: str) -> float:
|
|
|
|
|
|
values = [SequenceMatcher(None, _normalize_text(reference), _normalize_text(prediction)).ratio()]
|
|
|
|
|
|
if _normalize_text(reference) == _normalize_text(prediction):
|
|
|
|
|
|
values.append(1.0)
|
|
|
|
|
|
rouge = _pair_rouge(reference, prediction)
|
|
|
|
|
|
if rouge.get("rougeL") is not None:
|
|
|
|
|
|
values.append(float(rouge["rougeL"]))
|
|
|
|
|
|
return round(sum(values) / len(values) * 100, 2)
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-07-28 19:34:41 +08:00
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# Main entry point
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
def run_eval(config: dict[str, Any]) -> dict[str, Any]:
|
|
|
|
|
|
"""Execute a full evaluation run. Returns the result dict (also written to file)."""
|
|
|
|
|
|
model_path = config["model_name_or_path"]
|
|
|
|
|
|
adapter_path = config.get("adapter_name_or_path", "")
|
|
|
|
|
|
template = config.get("template", "qwen")
|
|
|
|
|
|
dataset_path = config["dataset_path"]
|
|
|
|
|
|
output_dir = Path(config["output_dir"])
|
|
|
|
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
|
basic_cfg = config.get("basic_metrics", {})
|
|
|
|
|
|
dimension_cfg = config.get("dimension", {}) or {}
|
|
|
|
|
|
output_precision = int(basic_cfg.get("output_precision", 2))
|
|
|
|
|
|
|
|
|
|
|
|
# ---- 1. Load dataset ----
|
|
|
|
|
|
print(f"[eval] loading dataset: {dataset_path}")
|
2026-08-03 15:49:21 +08:00
|
|
|
|
raw_samples = _load_dataset(dataset_path)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
print(f"[eval] loaded {len(raw_samples)} samples")
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "running", "dataset", len(raw_samples), 0, f"已加载 {len(raw_samples)} 条样本")
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
# ---- 2. Load model ----
|
|
|
|
|
|
print(f"[eval] loading model: {model_path}")
|
|
|
|
|
|
from compute.engines.llama_factory.inference import InferenceSession
|
|
|
|
|
|
session = InferenceSession()
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "running", "model_loading", len(raw_samples), 0, "正在加载评测模型")
|
2026-08-04 18:21:16 +08:00
|
|
|
|
session.load(
|
2026-07-28 19:34:41 +08:00
|
|
|
|
model_name_or_path=model_path,
|
|
|
|
|
|
adapter_name_or_path=adapter_path,
|
|
|
|
|
|
template=template,
|
|
|
|
|
|
infer_backend=config.get("infer_backend", "huggingface"),
|
|
|
|
|
|
infer_dtype=config.get("infer_dtype", "auto"),
|
|
|
|
|
|
)
|
2026-08-04 18:21:16 +08:00
|
|
|
|
# load() 为异步加载(立即返回 loading),必须等待后台线程完成后再进行推理
|
|
|
|
|
|
load_result = session.wait_until_loaded(timeout=float(config.get("load_timeout", 1800)))
|
2026-07-28 19:34:41 +08:00
|
|
|
|
if not load_result.get("loaded"):
|
|
|
|
|
|
raise RuntimeError(f"model load failed: {load_result.get('error', 'unknown')}")
|
|
|
|
|
|
print(f"[eval] model loaded OK")
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "running", "inference", len(raw_samples), 0, "开始生成模型回答")
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
# ---- 3. Run inference on each sample ----
|
|
|
|
|
|
samples: list[dict[str, Any]] = []
|
|
|
|
|
|
predictions: list[str] = []
|
|
|
|
|
|
references: list[str] = []
|
|
|
|
|
|
questions: list[str] = []
|
|
|
|
|
|
|
|
|
|
|
|
total = len(raw_samples)
|
|
|
|
|
|
judge_enabled = bool(dimension_cfg.get("eval_model") and dimension_cfg.get("api_url"))
|
|
|
|
|
|
print(f"[eval] starting inference on {total} samples, judge={'enabled' if judge_enabled else 'disabled'}")
|
|
|
|
|
|
|
|
|
|
|
|
for idx, raw in enumerate(raw_samples, start=1):
|
|
|
|
|
|
question = _sample_question(raw)
|
|
|
|
|
|
reference = _sample_reference(raw)
|
|
|
|
|
|
if not question:
|
|
|
|
|
|
print(f"[eval] sample {idx}/{total}: skipped (no question)")
|
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
|
|
# Inference
|
|
|
|
|
|
chat_msgs = [{"role": "user", "content": question}]
|
|
|
|
|
|
result = session.chat(
|
|
|
|
|
|
chat_msgs,
|
|
|
|
|
|
temperature=float(config.get("temperature", 0.1)),
|
|
|
|
|
|
top_p=float(config.get("top_p", 0.95)),
|
|
|
|
|
|
max_new_tokens=int(config.get("max_new_tokens", 512)),
|
|
|
|
|
|
do_sample=False,
|
|
|
|
|
|
)
|
|
|
|
|
|
prediction = result.get("response", "") if not result.get("error") else f"[ERROR] {result['error']}"
|
|
|
|
|
|
|
|
|
|
|
|
predictions.append(prediction)
|
|
|
|
|
|
references.append(reference)
|
|
|
|
|
|
questions.append(question)
|
|
|
|
|
|
|
|
|
|
|
|
# LLM Judge
|
|
|
|
|
|
judge_result: dict[str, Any] = {}
|
|
|
|
|
|
if judge_enabled:
|
|
|
|
|
|
judge_result = _judge_sample(question, reference, prediction, dimension_cfg)
|
|
|
|
|
|
|
|
|
|
|
|
samples.append({
|
|
|
|
|
|
"index": idx,
|
|
|
|
|
|
"input": question,
|
|
|
|
|
|
"reference_answer": reference,
|
|
|
|
|
|
"model_output": prediction,
|
|
|
|
|
|
"score": judge_result.get("score"),
|
|
|
|
|
|
"max_score": judge_result.get("max_score", dimension_cfg.get("score_max", 5)),
|
|
|
|
|
|
"passed": judge_result.get("passed"),
|
|
|
|
|
|
"judgement": judge_result.get("judgement"),
|
|
|
|
|
|
"evaluation_reason": judge_result.get("evaluation_reason", ""),
|
|
|
|
|
|
"error_type": judge_result.get("error_type"),
|
|
|
|
|
|
"dimension_scores": [
|
|
|
|
|
|
{"name": "judge_score", "score": judge_result.get("score", 0),
|
|
|
|
|
|
"max_score": judge_result.get("max_score", dimension_cfg.get("score_max", 5))},
|
|
|
|
|
|
] if judge_result else [],
|
|
|
|
|
|
"status": "completed",
|
|
|
|
|
|
})
|
2026-08-21 09:49:48 +08:00
|
|
|
|
if not judge_enabled:
|
|
|
|
|
|
deterministic_score = _deterministic_sample_score(reference, prediction)
|
|
|
|
|
|
samples[-1].update({
|
|
|
|
|
|
"score": deterministic_score,
|
|
|
|
|
|
"max_score": 100,
|
|
|
|
|
|
"raw_score": deterministic_score,
|
|
|
|
|
|
"raw_max_score": 100,
|
|
|
|
|
|
"passed": deterministic_score >= 60,
|
|
|
|
|
|
"judgement": "正确" if deterministic_score >= 80 else "部分正确" if deterministic_score >= 60 else "错误",
|
|
|
|
|
|
"evaluation_reason": "基于精确匹配、文本相似度和 ROUGE-L 的确定性评分",
|
|
|
|
|
|
})
|
|
|
|
|
|
_write_json(
|
|
|
|
|
|
output_dir / "eval_results.json",
|
|
|
|
|
|
{
|
|
|
|
|
|
"status": "running",
|
|
|
|
|
|
"overall_score": round(
|
|
|
|
|
|
sum(float(item["score"]) for item in samples if item.get("score") is not None)
|
|
|
|
|
|
/ max(len([item for item in samples if item.get("score") is not None]), 1),
|
|
|
|
|
|
output_precision,
|
|
|
|
|
|
),
|
|
|
|
|
|
"overall_score_max": 100,
|
|
|
|
|
|
"overall_evaluation": f"已完成 {len(samples)}/{total} 条样本",
|
|
|
|
|
|
"dimension_summary": [],
|
|
|
|
|
|
"samples": samples,
|
|
|
|
|
|
"sample_count": total,
|
|
|
|
|
|
"completed_count": len(samples),
|
|
|
|
|
|
"passed_count": sum(bool(item.get("passed")) for item in samples),
|
|
|
|
|
|
"basic_metrics": {},
|
|
|
|
|
|
"metric_summary_version": 2,
|
|
|
|
|
|
},
|
|
|
|
|
|
)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
progress_pct = int(idx / max(total, 1) * 100)
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "running", "inference", total, len(samples), f"已完成第 {idx} 条样本", idx)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
print(f"[eval] sample {idx}/{total} ({progress_pct}%) done")
|
|
|
|
|
|
|
|
|
|
|
|
# ---- 4. Compute basic metrics ----
|
|
|
|
|
|
print(f"[eval] computing basic metrics on {len(predictions)} predictions")
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "running", "metrics", total, len(samples), "正在计算评测指标", total)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
metrics_result: dict[str, Any] = {}
|
|
|
|
|
|
|
|
|
|
|
|
bleu_cfg = basic_cfg.get("bleu", {})
|
|
|
|
|
|
if bleu_cfg.get("enabled"):
|
|
|
|
|
|
metrics_result["bleu"] = _compute_bleu(references, predictions, int(bleu_cfg.get("ngram", 4)))
|
|
|
|
|
|
|
|
|
|
|
|
rouge_cfg = basic_cfg.get("rouge", {})
|
2026-08-21 09:49:48 +08:00
|
|
|
|
if rouge_cfg.get("enabled") or not judge_enabled:
|
2026-07-28 19:34:41 +08:00
|
|
|
|
metrics_result["rouge"] = _compute_rouge(references, predictions, rouge_cfg.get("methods"))
|
|
|
|
|
|
|
|
|
|
|
|
cosine_cfg = basic_cfg.get("cosine", {})
|
2026-08-21 09:49:48 +08:00
|
|
|
|
if cosine_cfg.get("enabled") or not judge_enabled:
|
2026-07-28 19:34:41 +08:00
|
|
|
|
metrics_result["cosine"] = _compute_cosine(references, predictions)
|
2026-08-03 15:49:21 +08:00
|
|
|
|
metrics_result["exact_match"] = _compute_exact_match(references, predictions)
|
|
|
|
|
|
metrics_result["text_similarity"] = _compute_text_similarity(references, predictions)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
|
|
|
|
|
|
# ---- 5. Summarise ----
|
|
|
|
|
|
completed = len(samples)
|
|
|
|
|
|
if judge_enabled:
|
|
|
|
|
|
scored = [s for s in samples if s.get("score") is not None]
|
|
|
|
|
|
passed_count = len([s for s in scored if s.get("passed")])
|
|
|
|
|
|
avg_score = round(sum(s["score"] for s in scored) / max(len(scored), 1), output_precision)
|
2026-08-21 09:49:48 +08:00
|
|
|
|
overall_score = avg_score
|
2026-07-28 19:34:41 +08:00
|
|
|
|
overall_score_max = 100
|
|
|
|
|
|
dimension_summary = [{
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"name": "LLM Judge",
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"score": overall_score,
|
|
|
|
|
|
"max_score": 100,
|
|
|
|
|
|
"pass_rate": round(passed_count / max(completed, 1) * 100, 1),
|
2026-08-21 10:48:58 +08:00
|
|
|
|
"sample_count": completed,
|
|
|
|
|
|
"available": bool(scored),
|
|
|
|
|
|
"error": "部分样本未返回可解析评分" if len(scored) < completed else "",
|
|
|
|
|
|
}] + _metric_dimension_summary(metrics_result)
|
2026-08-21 09:49:48 +08:00
|
|
|
|
overall_evaluation = f"评测完成:{completed} 样本,{passed_count} 通过,平均 {avg_score}/100 分"
|
2026-07-28 19:34:41 +08:00
|
|
|
|
else:
|
|
|
|
|
|
passed_count = 0
|
2026-08-03 15:49:21 +08:00
|
|
|
|
enabled_scores = [
|
|
|
|
|
|
float(item.get("score") or 0)
|
|
|
|
|
|
for item in metrics_result.values()
|
|
|
|
|
|
if isinstance(item, dict) and item.get("enabled", True) and item.get("score") is not None
|
|
|
|
|
|
]
|
|
|
|
|
|
overall_score = round(sum(enabled_scores) / len(enabled_scores), output_precision) if enabled_scores else 0
|
2026-07-28 19:34:41 +08:00
|
|
|
|
overall_score_max = 100
|
2026-08-21 10:48:58 +08:00
|
|
|
|
dimension_summary = _metric_dimension_summary(metrics_result)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
overall_evaluation = f"评测完成:{completed} 样本(未配置 LLM 评委)"
|
|
|
|
|
|
|
|
|
|
|
|
result = {
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"status": "completed",
|
2026-07-28 19:34:41 +08:00
|
|
|
|
"overall_score": overall_score,
|
|
|
|
|
|
"overall_score_max": overall_score_max,
|
|
|
|
|
|
"overall_evaluation": overall_evaluation,
|
|
|
|
|
|
"improvement_suggestions": [],
|
|
|
|
|
|
"dimension_summary": dimension_summary,
|
|
|
|
|
|
"samples": samples,
|
|
|
|
|
|
"sample_count": total,
|
|
|
|
|
|
"completed_count": completed,
|
|
|
|
|
|
"passed_count": passed_count,
|
|
|
|
|
|
"basic_metrics": metrics_result,
|
2026-08-21 09:49:48 +08:00
|
|
|
|
"metric_summary_version": 2,
|
2026-07-28 19:34:41 +08:00
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
# ---- 6. Write results ----
|
|
|
|
|
|
result_path = output_dir / "eval_results.json"
|
|
|
|
|
|
result_path.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
|
2026-08-21 09:49:48 +08:00
|
|
|
|
_write_eval_progress(output_dir, "completed", "completed", total, completed, "评测完成", total)
|
2026-07-28 19:34:41 +08:00
|
|
|
|
print(f"[eval] results written to {result_path}")
|
|
|
|
|
|
return result
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def main() -> None:
|
|
|
|
|
|
import argparse
|
|
|
|
|
|
parser = argparse.ArgumentParser(description="YG-FT Evaluation Runner")
|
|
|
|
|
|
parser.add_argument("--config", required=True, help="Path to eval config JSON file")
|
|
|
|
|
|
args = parser.parse_args()
|
|
|
|
|
|
|
|
|
|
|
|
config_path = Path(args.config)
|
|
|
|
|
|
if not config_path.exists():
|
|
|
|
|
|
print(f"FATAL: config file not found: {args.config}", file=sys.stderr)
|
|
|
|
|
|
sys.exit(1)
|
|
|
|
|
|
|
|
|
|
|
|
config = json.loads(config_path.read_text(encoding="utf-8"))
|
|
|
|
|
|
start = time.time()
|
|
|
|
|
|
try:
|
|
|
|
|
|
run_eval(config)
|
|
|
|
|
|
elapsed = time.time() - start
|
|
|
|
|
|
print(f"[eval] DONE in {elapsed:.1f}s")
|
|
|
|
|
|
except Exception as exc:
|
|
|
|
|
|
print(f"[eval] FAILED: {exc}", file=sys.stderr)
|
|
|
|
|
|
import traceback
|
|
|
|
|
|
traceback.print_exc()
|
|
|
|
|
|
sys.exit(1)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
|
main()
|