"""数据处理 - 生成结果的多层质量评测。 三层体系:规则层(确定性规则分)+ 语义层(本地嵌入向量)+ 评审层 (复用生成模型按 rubric 打分的 LLM-as-judge)。任一层失败自动降级, 评测永远返回可用结果,不阻断调用方流程。 """ from __future__ import annotations import logging from collections.abc import Mapping from dataclasses import asdict from datetime import UTC, datetime from typing import Any import httpx from .algorithms import normalize_text, score_quality from .algorithms.quality import composite_overall, semantic_quality_scores from .generation import ( ModelGenerationError, _is_retryable_generation_error, _json_payload, _message_content, chat_completions_url, ) logger = logging.getLogger(__name__) # 送入评审提示词的来源正文上限,避免超长切片挤占评分输出空间。 _MAX_JUDGE_SOURCE_CHARS = 6000 _JUDGE_DIMENSIONS: dict[str, tuple[str, ...]] = { "standard": ( "faithfulness", "correctness", "clarity", "completeness", "alignment", ), "reasoning": ( "faithfulness", "correctness", "clarity", "completeness", "alignment", "reasoning_validity", ), "dpo": ( "clarity", "chosen_quality", "rejected_quality", "preference_reasonableness", "faithfulness", ), } _DIMENSION_LABELS: dict[str, str] = { "faithfulness": "忠实度", "correctness": "正确性", "clarity": "问题清晰度", "completeness": "回答完整性", "alignment": "指令对齐", "reasoning_validity": "推理有效性", "chosen_quality": "chosen 回答质量", "rejected_quality": "rejected 回答质量", "preference_reasonableness": "偏好区分合理性", } _DIMENSION_RULES: dict[str, str] = { "faithfulness": "忠实度:答案的全部陈述是否被参考资料支持,没有编造、没有引入资料之外的信息;未提供参考资料时按答案内部自洽性评估", "correctness": "正确性:答案中的事实、概念与计算是否正确", "clarity": "问题清晰度:问题是否清晰、自包含、无歧义,脱离上下文也能理解", "completeness": "回答完整性:答案是否充分、直接地回应了问题的全部要点", "alignment": "指令对齐:答案的形式与范围是否符合问题的要求(如格式、语言、范围限定)", "reasoning_validity": "推理有效性:思维链步骤是否逻辑连贯、无跳步或循环论证,结论是否由推理过程自然得出", "chosen_quality": "chosen 回答质量:更优回答的正确性、完整性与表述质量", "rejected_quality": "rejected 回答质量:较差回答是否仍具备基本可读性,使对比训练有意义", "preference_reasonableness": "偏好区分合理性:chosen 是否明显优于 rejected,且优劣差异与问题直接相关", } def _judge_system_prompt(output_type: str) -> str: dimensions = _JUDGE_DIMENSIONS[output_type] rules = "\n".join(f"- {_DIMENSION_RULES[name]}" for name in dimensions) scores_schema = ", ".join(f'"{name}": 1-5' for name in dimensions) return ( "你是大模型训练数据质量评审员。严格依据用户消息中的【参考资料】评审这条训练数据,逐维度按 1-5 分打分:\n" f"{rules}\n" "评分锚点:5 分=完全符合维度描述;3 分=基本符合但有明显不足;1 分=严重不符合。\n" "忠实度只依据参考资料与公认常识判断,无法得到支持的陈述必须扣分;不要因为答案冗长而加分。\n" "只输出一个 JSON 对象,不要输出 JSON 之外的任何文字。\n" '输出格式:{"scores": {' + scores_schema + '}, "reason": "一句话总评", "issues": ["具体问题,没有则为空数组"]}' ) def _judge_user_prompt(record: Mapping[str, Any], source_content: str) -> str: source = normalize_text(source_content)[:_MAX_JUDGE_SOURCE_CHARS] or "(无参考资料)" instruction = normalize_text(str(record.get("instruction") or "")) or "(空)" input_text = normalize_text(str(record.get("input") or "")) sections = [f"【参考资料】\n{source}", f"【问题】\n{instruction}"] if input_text: sections.append(f"【输入】\n{input_text}") if record.get("chosen") or record.get("rejected"): sections.append(f"【更优回答 chosen】\n{normalize_text(str(record.get('chosen') or '')) or '(空)'}") sections.append(f"【较差回答 rejected】\n{normalize_text(str(record.get('rejected') or '')) or '(空)'}") else: output = normalize_text(str(record.get("output") or "")) sections.append(f"【回答】\n{output or '(空)'}") return "\n\n".join(sections) def _validated_judge_payload(payload: Any, output_type: str) -> dict[str, Any]: if not isinstance(payload, Mapping): raise ModelGenerationError("评审响应不是 JSON 对象") raw_scores = payload.get("scores") if not isinstance(raw_scores, Mapping): raise ModelGenerationError("评审响应缺少 scores 对象") expected = _JUDGE_DIMENSIONS[output_type] scores: dict[str, float] = {} for name in expected: value = raw_scores.get(name) if isinstance(value, bool) or not isinstance(value, (int, float)): raise ModelGenerationError(f"评审响应缺少维度 {name} 的有效分数") scores[name] = round(max(1.0, min(5.0, float(value))), 1) issues = payload.get("issues") if not isinstance(issues, list): issues = [] issues = [str(item)[:200] for item in issues if str(item).strip()][:8] reason = normalize_text(str(payload.get("reason") or ""))[:300] return { "scores": scores, "overall": round(sum(scores.values()) / len(scores) * 20, 2), "reason": reason, "issues": issues, } def _judge_record( record: Mapping[str, Any], source_content: str, *, model: Mapping[str, Any], config: Mapping[str, Any], client: httpx.Client | None, ) -> dict[str, Any] | None: output_type = str(config.get("output_type") or "standard").strip().lower() if output_type not in _JUDGE_DIMENSIONS: output_type = "standard" endpoint = chat_completions_url(str(model.get("api_url") or "")) model_name = str(model.get("online_model_name") or model.get("name") or "").strip() if not model_name: raise ModelGenerationError("generation model name is required") temperature = 0.1 max_tokens = max(256, min(2048, int(config.get("max_tokens", 1024) or 1024))) timeout = max(1.0, min(120.0, float(config.get("request_timeout_seconds", 60) or 60))) retries = max(0, min(5, int(config.get("generation_retries", 2) or 2))) headers = {"Content-Type": "application/json"} api_key = str(model.get("api_key") or "").strip() if api_key: headers["Authorization"] = f"Bearer {api_key}" request_payload: dict[str, Any] = { "model": model_name, "messages": [ {"role": "system", "content": _judge_system_prompt(output_type)}, {"role": "user", "content": _judge_user_prompt(record, source_content)}, ], "temperature": temperature, "max_tokens": max_tokens, } if bool(config.get("json_mode", False)): request_payload["response_format"] = {"type": "json_object"} owns_client = client is None http_client = client or httpx.Client(timeout=timeout) try: last_error: Exception | None = None for _ in range(retries + 1): try: response = http_client.post(endpoint, headers=headers, json=request_payload) response.raise_for_status() body = response.json() if not isinstance(body, Mapping): raise ModelGenerationError("model response body must be a JSON object") judged = _validated_judge_payload( _json_payload(_message_content(body)), output_type, ) judged["model"] = model_name judged["output_type"] = output_type return judged except Exception as exc: last_error = exc if not _is_retryable_generation_error(exc): break raise ModelGenerationError(f"质量评审调用失败: {last_error}") finally: if owns_client: http_client.close() def evaluate_result_record( record: Mapping[str, Any], *, source_content: str = "", model: Mapping[str, Any] | None = None, config: Mapping[str, Any] | None = None, client: httpx.Client | None = None, embed_model: Any = None, min_output_length: int = 20, ) -> dict[str, Any]: """对一条生成结果执行三层评测,返回可直接落库的 quality_score 字典。 规则层字段保持原样平铺(向后兼容既有读取方);新增 ``semantic``、 ``judge``、``layers``、``evaluated`` 与组合 ``overall``。 """ config_dict = dict(config or {}) rule = score_quality( record, min_output_length=min_output_length, source_content=source_content, ) quality: dict[str, Any] = asdict(rule) semantic = semantic_quality_scores( record, source_content=source_content, embed_model=embed_model, ) judge: dict[str, Any] | None = None if model is not None: try: judge = _judge_record( record, source_content, model=model, config=config_dict, client=client, ) except Exception as exc: logger.warning( "data process judge evaluation degraded: %s", exc, ) layers = { "rule": rule.overall, "semantic": semantic.get("overall") if semantic else None, "judge": judge.get("overall") if judge else None, } quality.update( semantic=semantic, judge=judge, layers=layers, evaluated=True, evaluated_at=datetime.now(UTC).isoformat(), overall=composite_overall( rule=layers["rule"], semantic=layers["semantic"], judge=layers["judge"], ), ) return quality def reevaluate_edited_record( record: Mapping[str, Any], *, source_content: str = "", previous_quality: Mapping[str, Any] | None = None, embed_model: Any = None, min_output_length: int = 20, ) -> dict[str, Any]: """手动编辑/恢复后重算规则与语义层,丢弃已过期的评审层。 编辑会改变内容,旧的评审分不再可信;规则与语义层本地重算零成本。 ``evaluated`` 标记沿用原值,保证已评测过的结果编辑后仍有可用分数。 """ rule = score_quality( record, min_output_length=min_output_length, source_content=source_content, ) quality: dict[str, Any] = asdict(rule) semantic = semantic_quality_scores( record, source_content=source_content, embed_model=embed_model, ) previous = dict(previous_quality or {}) evaluated = bool(previous.get("evaluated")) layers = { "rule": rule.overall, "semantic": semantic.get("overall") if semantic else None, "judge": None, } quality.update( semantic=semantic, judge=None, layers=layers, evaluated=evaluated, evaluated_at=( datetime.now(UTC).isoformat() if evaluated else None ), overall=composite_overall( rule=layers["rule"], semantic=layers["semantic"], ), ) return quality