147 lines
5.1 KiB
Python
147 lines
5.1 KiB
Python
#!/usr/bin/env python3
|
|
"""细纲 rubric 的离线回归测试。"""
|
|
|
|
import inspect
|
|
import pathlib
|
|
import sys
|
|
import unittest
|
|
|
|
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[3]
|
|
SCRIPT_DIR = PROJECT_ROOT / "muse" / "lifecycle" / "quality" / "skills" / "judge" / "评估内容质量" / "scripts"
|
|
if str(SCRIPT_DIR) not in sys.path:
|
|
sys.path.insert(0, str(SCRIPT_DIR))
|
|
from fine_outline_rubric import ( # noqa: E402
|
|
DIMENSIONS,
|
|
RUBRIC_PROFILE,
|
|
stability_warning,
|
|
validate_report,
|
|
validate_scores,
|
|
)
|
|
|
|
|
|
def valid_scores():
|
|
return {
|
|
dimension: {"score": 4, "evidence": f"证据-{dimension}"}
|
|
for dimension in DIMENSIONS
|
|
}
|
|
|
|
|
|
class FineOutlineRubricTest(unittest.TestCase):
|
|
def test_every_dimension_requires_evidence(self):
|
|
self.assertEqual(validate_scores(valid_scores()), [])
|
|
missing_evidence = valid_scores()
|
|
missing_evidence[DIMENSIONS[0]] = {"score": 4}
|
|
self.assertTrue(any("缺少证据" in error for error in validate_scores(missing_evidence)))
|
|
|
|
def test_prose_dimensions_are_rejected(self):
|
|
scores = valid_scores()
|
|
scores["style_fit"] = {"score": 5, "evidence": "不应出现"}
|
|
self.assertTrue(any("禁止正文质量维度" in error for error in validate_scores(scores)))
|
|
|
|
def test_profile_and_score_range_are_checked(self):
|
|
report = {"profile": RUBRIC_PROFILE, "scores": valid_scores()}
|
|
self.assertEqual(validate_report(report), [])
|
|
bad = {"profile": "quality_gate", "scores": valid_scores()}
|
|
bad["scores"][DIMENSIONS[1]] = {"score": 6, "evidence": "超范围"}
|
|
self.assertEqual(len(validate_report(bad)), 2)
|
|
|
|
def test_large_reviewer_gap_warns(self):
|
|
first = {dimension: 4 for dimension in DIMENSIONS}
|
|
second = {dimension: 4 for dimension in DIMENSIONS}
|
|
second[DIMENSIONS[2]] = 5
|
|
result = stability_warning(first, second)
|
|
self.assertFalse(result["stable"])
|
|
self.assertEqual(result["gaps"][DIMENSIONS[2]], 1.0)
|
|
|
|
def test_missing_dimension_fails_stability_closed(self):
|
|
first = {dimension: 4 for dimension in DIMENSIONS}
|
|
second = {dimension: 4 for dimension in DIMENSIONS[:-1]}
|
|
result = stability_warning(first, second)
|
|
self.assertFalse(result["stable"])
|
|
self.assertIn(DIMENSIONS[-1], result["missingDimensions"])
|
|
|
|
def test_batch_report_requires_exact_candidates_and_distinct_judge_identity(self):
|
|
self.assertIn("expected_judge_id", inspect.signature(validate_report).parameters)
|
|
report = {
|
|
"profile": RUBRIC_PROFILE,
|
|
"judgeId": "judge-primary",
|
|
"evaluations": [
|
|
{"candidateId": "blind-1", "scores": valid_scores(), "summary": "摘要一"},
|
|
{"candidateId": "blind-2", "scores": valid_scores(), "summary": "摘要二"},
|
|
],
|
|
}
|
|
self.assertEqual(
|
|
validate_report(
|
|
report,
|
|
expected_judge_id="judge-primary",
|
|
expected_candidate_ids=("blind-1", "blind-2"),
|
|
),
|
|
[],
|
|
)
|
|
duplicate = {**report, "evaluations": [report["evaluations"][0], report["evaluations"][0]]}
|
|
self.assertTrue(
|
|
validate_report(
|
|
duplicate,
|
|
expected_judge_id="judge-primary",
|
|
expected_candidate_ids=("blind-1", "blind-2"),
|
|
)
|
|
)
|
|
|
|
invalid_id = {
|
|
**report,
|
|
"evaluations": [{**report["evaluations"][0], "candidateId": ["blind-1"]}],
|
|
}
|
|
self.assertTrue(validate_report(invalid_id, expected_judge_id="judge-primary"))
|
|
|
|
arm_leak = {**report, "arm": "outline_only"}
|
|
self.assertTrue(
|
|
validate_report(
|
|
arm_leak,
|
|
expected_judge_id="judge-primary",
|
|
expected_candidate_ids=("blind-1", "blind-2"),
|
|
)
|
|
)
|
|
|
|
def test_nested_evaluation_and_score_reject_arm_card_manifest_leakage(self):
|
|
evaluation = {
|
|
"candidateId": "blind-1",
|
|
"scores": valid_scores(),
|
|
"summary": "短安全摘要",
|
|
}
|
|
report = {
|
|
"profile": RUBRIC_PROFILE,
|
|
"judgeId": "judge-primary",
|
|
"evaluations": [evaluation],
|
|
}
|
|
leaking_evaluation = {
|
|
**report,
|
|
"evaluations": [{**evaluation, "arm": "outline_only"}],
|
|
}
|
|
leaking_scores = valid_scores()
|
|
leaking_scores[DIMENSIONS[0]] = {
|
|
**leaking_scores[DIMENSIONS[0]],
|
|
"cardManifest": {"count": 1},
|
|
}
|
|
leaking_score = {
|
|
**report,
|
|
"evaluations": [{**evaluation, "scores": leaking_scores}],
|
|
}
|
|
self.assertTrue(
|
|
validate_report(
|
|
leaking_evaluation,
|
|
expected_judge_id="judge-primary",
|
|
expected_candidate_ids=("blind-1",),
|
|
)
|
|
)
|
|
self.assertTrue(
|
|
validate_report(
|
|
leaking_score,
|
|
expected_judge_id="judge-primary",
|
|
expected_candidate_ids=("blind-1",),
|
|
)
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|