muse-agent-example/tests/契约/test_语义检测合同.py
zizi d909d1bd1b 后端实现与用例身份:19 包集成落地并修复收尾缺陷
实现侧:
- 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。
- 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。
- 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。
- 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。
- 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。
- 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。
- 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。
- 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。

用例侧:
- 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存;
- 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
2026-09-18 01:15:00 +08:00

166 lines
6.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""B06核原字和完整声明;纯合同验证不认证模型执行或语义正确性。"""
import copy
import pytest
from muse.审校修订.接口 import 审校错误, 核对语义检测, 组装语义检测材料
def 材料():
return 组装语义检测材料(
"旅者等在城门。",
{
"sources": [{"source_id": "outline", "kind": "fine_outline", "text": "旅者守门。"}],
"assertions": [{"statement_id": "guard", "text": "旅者是守卫。"}],
"constraints": [{"statement_id": "stay", "text": "不得离开城门。"}],
},
)
def 输出(material, script="ok"):
quote = material["candidate"]
result = {
"claims": [
{
"claim_id": "claim-1",
"text": "候选中人物采取行动。",
"candidate_quote": quote,
"state": "supported",
"evidence_refs": ["source:outline"],
"reason": "参照细纲核对。",
}
],
"findings": [],
"assertion_verdicts": [
{
"statement_id": id_,
"verdict": "pass",
"candidate_quote": quote,
"evidence_refs": ["assertion:" + id_],
"reason": "与给定命题相符。",
}
for id_ in material["required_assertions"]
],
"constraint_verdicts": [
{
"statement_id": id_,
"verdict": "pass",
"candidate_quote": quote,
"evidence_refs": ["constraint:" + id_],
"reason": "符合给定约束。",
}
for id_ in material["required_constraints"]
],
"new_setting_candidates": [],
}
if script == "high":
result["findings"] = [
{
"finding_id": "high-1",
"category": "fact",
"severity": "high",
"candidate_quote": quote,
"evidence_refs": ["source:outline"],
"message": "合成事实冲突。",
}
]
elif script == "unknown":
result["claims"][0].update(state="unknown", evidence_refs=[], reason="给定来源不足以认定。")
result["constraint_verdicts"][0].update(verdict="unknown", reason="约束无法确定。")
result["new_setting_candidates"] = [
{"claim_id": "claim-1", "proposal": "待作者核对的新设定。"}
]
elif script == "missing":
result["assertion_verdicts"] = []
return result
@pytest.mark.case_id(
"NC-w25-25f101",
environment="离线合同",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["原字定位及完整命题来自实际材料"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_原字定位与命题覆盖来自实际材料__25f101():
material = 材料()
result = 核对语义检测(material, 输出(material))
assert result["claims"][0]["start"] == 0
assert result["claims"][0]["end"] == len(material["candidate"])
assert result["constraint_counts"] == {"pass": 1, "fail": 0, "unknown": 0}
assert result["status"] == "passed"
@pytest.mark.case_id(
"NC-w25-25f102",
environment="离线合同",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["缺项、越权字段、伪位置、重复及无外部依据拒绝"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
@pytest.mark.parametrize(
"variant", ["passed", "quote", "missing", "reference", "duplicate", "position", "external"]
)
def test_缺项陌生依据及自报通过不能伪装检测__25f102(variant):
material = 材料()
output = 输出(material)
if variant == "passed":
output["passed"] = True
elif variant == "quote":
output["claims"][0]["candidate_quote"] = "原文没有这句话"
elif variant == "missing":
output["assertion_verdicts"] = []
elif variant == "reference":
output["claims"][0]["evidence_refs"] = ["source:stranger"]
elif variant == "duplicate":
output["claims"].append(copy.deepcopy(output["claims"][0]))
elif variant == "position":
output["claims"][0]["start"] = 0
else:
output = 输出(material, "high")
output["findings"][0]["evidence_refs"] = ["candidate"]
with pytest.raises(审校错误):
核对语义检测(material, output)
@pytest.mark.case_id(
"NC-w25-25f103",
environment="离线合同",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["高严重度失败与未知未决分别保留,不能冒充零缺陷"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_高严重度与未知不得被零发现覆盖__25f103():
material = 材料()
high = 核对语义检测(material, 输出(material, "high"))
unknown = 核对语义检测(material, 输出(material, "unknown"))
assert high["status"] == "failed" and high["high_severity_count"] == 1
assert unknown["status"] == "inconclusive" and unknown["unknown_count"] == 2
assert unknown["new_setting_candidates"][0]["claim_id"] == "claim-1"
@pytest.mark.case_id(
"NC-w25-25f104",
environment="离线合同",
given="明确候选、共同依据与本例异常",
when="运行B06核验或隔离S02实际检测与读回",
then=["声明新事实须有细纲,新建议只关联未决或声明陈述"],
contract="docs/系统架构/新版设计/模块设计/B10-效果评测.md",
)
def test_声明新事实须有细纲且建议不能借用已确认陈述__25f104():
material = 材料()
output = 输出(material)
output["claims"][0].update(state="declared_new", evidence_refs=["assertion:guard"])
with pytest.raises(审校错误, match="细纲"):
核对语义检测(material, output)
output["claims"][0]["evidence_refs"] = ["source:outline"]
output["new_setting_candidates"] = [{"claim_id": "claim-1", "proposal": "新设定待确认。"}]
assert 核对语义检测(material, output)["new_setting_candidates"]
output["claims"][0]["state"] = "supported"
with pytest.raises(审校错误, match="未决"):
核对语义检测(material, output)