实现侧: - 上下文:任务范围拆分为 范围校验/范围授权;索引按可发现口径重建、索引新鲜度改对称差;依赖校验统一快照漂移说明。 - 知识方法:方法与材料读取口径统一;超限方法材料按可选省略,核对路径不再二次计费;删除无合同的读时重算。 - 任务运行:新增 context.usage/tool.denied 事件类型;连接池常驻并在装配生命周期内开关;调用结算与核对分列。 - 效果评测/审校修订/交付连载/作者经验/作品规划:凭据冻结、标定消费、导出补证、事实引文核对等收尾修复。 - 资源加载:能力正文不再夹带索引用的导航注记(该注记此前进入角色与技能的模型提示)。 - 元数据:受保护骨架与代码保护属性对齐;字段校验与内置结构口径同步。 - 基础设施:环境预检进入装配生命周期;数据库连接运行期字段不参与相等比较;索引指纹归一化 jsonb 浮点。 - 删除被替代实现:7 份旧提示词模板与空壳 资料来源 读取器。 用例侧: - 用例身份与导航元信息迁移;夹具补生命周期、同库暴露与模板封存; - 本轮定向修复:方法材料省略、事实引文、迁移回执、额度与暂停用例、慢用例超时预算等。
709 lines
27 KiB
Python
709 lines
27 KiB
Python
"""固定B06返修候选经B10双评委比较;合成HTTP不认证真实外部模型或文学标定。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import copy
|
||
import json
|
||
import os
|
||
import socket
|
||
import subprocess
|
||
import sys
|
||
import time
|
||
from dataclasses import replace
|
||
from datetime import UTC, datetime, timedelta
|
||
from decimal import Decimal
|
||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||
from pathlib import Path
|
||
from threading import Thread
|
||
from uuid import uuid4
|
||
|
||
import httpx
|
||
import pytest
|
||
import test_持久返修会话 as 返修测试
|
||
import uvicorn
|
||
from fastapi.testclient import TestClient
|
||
|
||
from muse.任务运行.接口 import (
|
||
任务状态,
|
||
内容哈希,
|
||
凭据引用,
|
||
提供方配置,
|
||
角色策略目录,
|
||
运行配置内容,
|
||
配置版本管理,
|
||
配置验证证据,
|
||
预算管理,
|
||
额度策略,
|
||
)
|
||
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
|
||
from muse.共享.错误 import Muse错误
|
||
from muse.启动 import 构建
|
||
from muse.审校修订.接口 import 返修比较定位
|
||
from muse.接入.http.应用 import 创建应用
|
||
from muse.效果评测.接口 import (
|
||
实验执行请求,
|
||
评测服务,
|
||
返修比较实验请求,
|
||
返修比较数据集发布,
|
||
配置选择,
|
||
)
|
||
from muse.编排.执行评测 import 登记评测执行
|
||
from muse.配置 import 应用配置, 服务配置
|
||
|
||
pytestmark = pytest.mark.数据库
|
||
生成环境 = 返修测试.生成环境
|
||
修订环境 = 返修测试.修订环境
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w23-239007",
|
||
environment="隔离PG/合成HTTP及浏览器",
|
||
given="具名作者、固定来源、隔离测试配置与真实模块存储",
|
||
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
||
then=["双评委记录经真实浏览器展示后作者可保持原文"],
|
||
contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md",
|
||
)
|
||
@pytest.mark.浏览器
|
||
def test_双评委记录经真实浏览器展示后作者可保持原文__239007(比较环境, tmp_path):
|
||
if os.environ.get("MUSE_RUN_BROWSER") != "1":
|
||
pytest.skip("需显式启用浏览器")
|
||
env = 比较环境
|
||
session, _, _, _, exp = _准备实验(env, "browser-pair")
|
||
_启动并运行(env, exp)
|
||
proof = (
|
||
env["eval_app"]
|
||
.要求评测()
|
||
.导出返修比较证明(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
env["pools"][用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
)
|
||
with socket.socket() as sock:
|
||
sock.bind(("127.0.0.1", 0))
|
||
port = sock.getsockname()[1]
|
||
base = f"http://127.0.0.1:{port}"
|
||
secret = tmp_path / "browser-password"
|
||
secret.write_text("synthetic-comparison-browser")
|
||
secret.chmod(0o600)
|
||
config = replace(
|
||
env["装配"].配置,
|
||
HTTP=服务配置(
|
||
str(secret),
|
||
作者ID="gen-author",
|
||
公开地址=base,
|
||
允许来源=(base,),
|
||
),
|
||
)
|
||
server = uvicorn.Server(
|
||
uvicorn.Config(
|
||
创建应用(config),
|
||
host="127.0.0.1",
|
||
port=port,
|
||
log_level="warning",
|
||
)
|
||
)
|
||
thread = Thread(target=server.run, daemon=True)
|
||
thread.start()
|
||
output = Path(os.environ.get("MUSE_BROWSER_ARTIFACTS", str(tmp_path / "browser"))).resolve()
|
||
output.mkdir(parents=True, exist_ok=True)
|
||
try:
|
||
for _ in range(100):
|
||
if server.started:
|
||
break
|
||
time.sleep(0.05)
|
||
assert server.started and httpx.get(base + "/ready", trust_env=False).status_code == 200
|
||
environment = {
|
||
**os.environ,
|
||
"MUSE_WORKBENCH_URL": base,
|
||
"MUSE_AUTHOR_PASSWORD_FILE": str(secret),
|
||
"MUSE_REVISION_COMPARISON_RECEIPT": proof["receipt_id"],
|
||
"MUSE_REVISION_SESSION_URL": (
|
||
"/works/gen-work/review?chapter=ch-4"
|
||
f"&diagnosis={env['授权'].diagnosis_id}&revision_session={session['session_id']}"
|
||
),
|
||
"MUSE_BROWSER_EXECUTABLE": (
|
||
"/Applications/Google Chrome.app/Contents/MacOS/Google Chrome"
|
||
),
|
||
}
|
||
with (output / "浏览器.json").open("w") as report:
|
||
result = subprocess.run(
|
||
[
|
||
"pnpm",
|
||
"exec",
|
||
"playwright",
|
||
"test",
|
||
"返修独立比较.spec.ts",
|
||
"--reporter=json",
|
||
"--output",
|
||
str(output / "产物"),
|
||
],
|
||
cwd=Path(__file__).resolve().parents[2] / "web",
|
||
env=environment,
|
||
stdout=report,
|
||
stderr=subprocess.STDOUT,
|
||
timeout=90,
|
||
)
|
||
assert result.returncode == 0, "浏览器未通过,见浏览器.json"
|
||
final = (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修会话(
|
||
env["作者"],
|
||
env["装配"].任务运行,
|
||
session["session_id"],
|
||
)
|
||
)
|
||
assert final["state"] == "closed" and final["author_choice"] == "original"
|
||
assert len(env["received_judges"]) == 2
|
||
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
|
||
finally:
|
||
server.should_exit = True
|
||
thread.join(10)
|
||
assert not thread.is_alive()
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w23-239006",
|
||
environment="隔离PG/合成HTTP及浏览器",
|
||
given="具名作者、固定来源、隔离测试配置与真实模块存储",
|
||
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
||
then=["返修比较经真实CLI封存默认HTTP执行及生产只读回查"],
|
||
contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md",
|
||
)
|
||
def test_返修比较经真实CLI封存默认HTTP执行及生产只读回查__239006(比较环境, tmp_path):
|
||
env = 比较环境
|
||
session, _, publication, data, exp = _准备实验(env, "entry-pair")
|
||
secret = tmp_path / "author-password"
|
||
secret.write_text("synthetic-pair-entry")
|
||
secret.chmod(0o600)
|
||
|
||
def 配置文件(purpose):
|
||
config = tmp_path / (purpose.value + ".toml")
|
||
config.write_text(
|
||
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
|
||
+ json.dumps(env["pools"][purpose].引用.位置)
|
||
+ '\n["资源"]\n"发布身份"='
|
||
+ json.dumps(env["装配"].配置.资源发布身份)
|
||
+ '\n["运行"]\n"用途"='
|
||
+ json.dumps(purpose.value)
|
||
+ '\n[HTTP]\n"作者ID"="gen-author"\n"口令文件"='
|
||
+ json.dumps(str(secret))
|
||
+ '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n'
|
||
)
|
||
return config
|
||
|
||
production, evaluation, maintenance = (配置文件(p) for p in (用途.生产, 用途.评测, 用途.维护))
|
||
|
||
def cli(group, config, action, target, *extra):
|
||
result = subprocess.run(
|
||
[
|
||
sys.executable,
|
||
"-I",
|
||
"-m",
|
||
"muse",
|
||
group,
|
||
str(config),
|
||
action,
|
||
str(target),
|
||
*map(str, extra),
|
||
],
|
||
cwd=tmp_path,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=30,
|
||
)
|
||
assert result.returncode == 0, result.stderr
|
||
return json.loads(result.stdout)
|
||
|
||
request_file = tmp_path / "pair.json"
|
||
request_file.write_text(publication.model_dump_json())
|
||
sealed = cli("评测", maintenance, "发布返修数据集", request_file, "--来源配置", production)
|
||
assert sealed["version_id"] == data["version_id"] and sealed["duplicate"]
|
||
cfg = replace(
|
||
env["eval_app"].配置,
|
||
HTTP=服务配置(
|
||
str(secret),
|
||
作者ID="gen-author",
|
||
公开地址="http://testserver",
|
||
允许来源=("http://testserver",),
|
||
),
|
||
)
|
||
creator = {
|
||
"command_id": "compare-entry-pair",
|
||
"request": {
|
||
"dataset_version_id": data["version_id"],
|
||
"dataset_hash": data["public_hash"],
|
||
"judges": [
|
||
{"config_id": "revision-judge-1", "version": "1"},
|
||
{"config_id": "revision-judge-2", "version": "1"},
|
||
],
|
||
"dimensions": ["事实保真", "表达有效性"],
|
||
"max_cost_usd": "4",
|
||
"max_calls_per_sample": 6,
|
||
},
|
||
}
|
||
app = 创建应用(cfg, 计价=_合成计价())
|
||
with TestClient(app, headers={"Origin": "http://testserver"}) as client:
|
||
assert (
|
||
client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200
|
||
)
|
||
created = client.post("/api/v1/evaluation/revision-experiments", json=creator)
|
||
assert created.status_code == 201, created.text
|
||
assert created.json()["experiment_id"] == exp["experiment_id"]
|
||
started = client.post(
|
||
f"/api/v1/evaluation/experiments/{exp['experiment_id']}/execution",
|
||
json={
|
||
"approval_ref": "synthetic-pair-budget",
|
||
"deadline": (datetime.now(UTC) + timedelta(minutes=10)).isoformat(),
|
||
},
|
||
)
|
||
assert started.status_code == 201, started.text
|
||
tasks = [unit["task_id"] for unit in started.json()["units"]]
|
||
assert len(tasks) == 2 and env["received_judges"] == []
|
||
for tid in tasks:
|
||
for _ in range(3):
|
||
step = client.post(f"/api/v1/tasks/{tid}/advance")
|
||
assert step.status_code == 200, step.text
|
||
if step.json()["state"] == "completed":
|
||
break
|
||
assert step.json()["state"] == "completed"
|
||
assert len(env["received_judges"]) == 2
|
||
exported = cli(
|
||
"评测", evaluation, "导出返修证明", exp["experiment_id"], "--维护配置", maintenance
|
||
)
|
||
receipt_id = exported["receipt_id"]
|
||
read = cli("审校", production, "返修比较证明", receipt_id)
|
||
assert read["current"]["selection"] == "candidate"
|
||
assert not read["current"]["automatic_adoption"]
|
||
from muse.配置 import 读取配置
|
||
|
||
with TestClient(
|
||
创建应用(读取配置(production)), headers={"Origin": "http://testserver"}
|
||
) as client:
|
||
assert (
|
||
client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200
|
||
)
|
||
path = f"/api/v1/revision-sessions/{session['session_id']}/comparisons/{receipt_id}"
|
||
returned = client.get(path)
|
||
assert returned.status_code == 200 and returned.json() == read
|
||
wrong = client.get(f"/api/v1/revision-sessions/{uuid4()}/comparisons/{receipt_id}")
|
||
assert wrong.status_code == 403 and wrong.json()["code"] == "SCOPE_DENIED"
|
||
assert len(env["received_judges"]) == 2
|
||
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
|
||
|
||
|
||
class _合成计价:
|
||
版本 = "revision-pair-synthetic-price-v1"
|
||
|
||
def 金额(self, 结果):
|
||
return Decimal("0.125") if 结果.用量 else None
|
||
|
||
|
||
@pytest.fixture
|
||
def 比较环境(修订环境, tmp_path, monkeypatch):
|
||
# 返修来源与评测维护必须同库:复用修订环境所在库,另借新库会被身份守卫拒绝。
|
||
应用测试库 = 修订环境["库组"]
|
||
judges = ("synthetic-revision-judge-a", "synthetic-revision-judge-b")
|
||
policy = 角色策略目录.从发布包()
|
||
definition = copy.deepcopy(policy.定义)
|
||
definition["models"]["fixed"].extend(judges)
|
||
definition["models"]["judge-fixed"].extend(judges)
|
||
isolated = 角色策略目录(
|
||
definition,
|
||
资源发布身份=policy.资源发布身份,
|
||
角色资源=policy.角色资源,
|
||
)
|
||
monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda cls: isolated))
|
||
received: list[dict] = []
|
||
scripted: list[str] = []
|
||
|
||
class Handler(BaseHTTPRequestHandler):
|
||
def do_POST(self):
|
||
request = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
||
received.append(request)
|
||
action = scripted.pop(0) if scripted else "candidate"
|
||
material = json.loads(request["input"])
|
||
if action == "api_error":
|
||
response = {
|
||
"id": f"revision-pair-{len(received)}",
|
||
"model": None,
|
||
"status": "failed",
|
||
"output": [],
|
||
"usage": {"input_tokens": 0, "output_tokens": 0},
|
||
"error": {"code": "synthetic_failure", "message": "隔离评委失败"},
|
||
}
|
||
else:
|
||
candidate_side = next(
|
||
side
|
||
for side in ("left", "right")
|
||
if "值得注意的是" not in material[side]["text"]
|
||
)
|
||
evidence = [
|
||
{"side": side, "quote": material[side]["text"]} for side in ("left", "right")
|
||
]
|
||
output = {
|
||
"choice": candidate_side,
|
||
"rationale": "隔离评委选择删去无功能元话语的一侧",
|
||
"dimensions": [
|
||
{
|
||
"dimension": dimension,
|
||
"choice": candidate_side,
|
||
"rationale": "两侧原字可核,候选更直接且事实保持",
|
||
"evidence": evidence,
|
||
}
|
||
for dimension in material["dimensions"]
|
||
],
|
||
}
|
||
response = {
|
||
"id": f"revision-pair-{len(received)}",
|
||
"model": request["model"],
|
||
"status": "completed",
|
||
"output": [
|
||
{
|
||
"type": "message",
|
||
"content": [
|
||
{
|
||
"type": "output_text",
|
||
"text": json.dumps(output, ensure_ascii=False),
|
||
}
|
||
],
|
||
}
|
||
],
|
||
"usage": {"input_tokens": 5, "output_tokens": 7},
|
||
}
|
||
payload = (
|
||
"data: "
|
||
+ json.dumps({"type": f"response.{response['status']}", "response": response})
|
||
+ "\n\n"
|
||
).encode()
|
||
self.send_response(200)
|
||
self.send_header("Content-Type", "text/event-stream")
|
||
self.send_header("Content-Length", str(len(payload)))
|
||
self.end_headers()
|
||
self.wfile.write(payload)
|
||
|
||
def log_message(self, format, *args):
|
||
pass
|
||
|
||
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
||
thread = Thread(target=server.serve_forever, daemon=True)
|
||
thread.start()
|
||
try:
|
||
pool = 应用测试库[用途.评测]
|
||
app = 构建(
|
||
应用配置(
|
||
pool.引用,
|
||
isolated.资源发布身份,
|
||
运行用途=用途.评测,
|
||
原文暂存=str(tmp_path / "evaluation-raw"),
|
||
)
|
||
)
|
||
with app.生命周期():
|
||
登记评测执行(app, _合成计价())
|
||
key = tmp_path / "judge-key"
|
||
key.write_text("synthetic-only")
|
||
key.chmod(0o600)
|
||
|
||
class Validator:
|
||
身份 = "synthetic-revision-comparison-offline"
|
||
|
||
def 验证(self, 内容, 执行用途):
|
||
return 配置验证证据(
|
||
内容哈希(内容.冻结()),
|
||
内容.角色策略版本,
|
||
内容.资源发布身份,
|
||
执行用途,
|
||
("synthetic:isolated-revision-pair-http",),
|
||
"offline_contract",
|
||
)
|
||
|
||
manager = 配置版本管理(pool, Validator())
|
||
for index, model in enumerate(judges, start=1):
|
||
config_id = f"revision-judge-{index}"
|
||
content = 运行配置内容(
|
||
"direct",
|
||
"1",
|
||
definition["version"],
|
||
isolated.资源发布身份,
|
||
"revision-pair-budget",
|
||
{
|
||
"judge": {
|
||
"provider": "synthetic",
|
||
"model": model,
|
||
"thinking": "high",
|
||
"tools": [],
|
||
}
|
||
},
|
||
(凭据引用("key", "受控存储", str(key)),),
|
||
(
|
||
提供方配置(
|
||
"synthetic",
|
||
"responses",
|
||
f"http://127.0.0.1:{server.server_port}/v1/responses",
|
||
"key",
|
||
),
|
||
),
|
||
_合成计价.版本,
|
||
)
|
||
manager.保存草案(config_id, "1", content)
|
||
receipt = manager.验证版本(config_id, "1")
|
||
manager.启用(
|
||
config_id,
|
||
"1",
|
||
验证回执=receipt,
|
||
批准引用="synthetic-revision-pair",
|
||
预期代次=0,
|
||
)
|
||
预算管理(pool, "revision-pair-budget").登记策略(
|
||
额度策略("revision-pair-budget", "1", Decimal("20"), 20)
|
||
)
|
||
yield {
|
||
**修订环境,
|
||
"pools": 应用测试库,
|
||
"eval_app": app,
|
||
"eval_actor": 调用身份("gen-author", None, 用途.评测, 内容用途.检测),
|
||
"maint_actor": 调用身份("gen-author", None, 用途.维护, 内容用途.检测),
|
||
"received_judges": received,
|
||
"scripted_judges": scripted,
|
||
}
|
||
finally:
|
||
server.shutdown()
|
||
server.server_close()
|
||
thread.join(timeout=5)
|
||
|
||
|
||
def _准备实验(env, command: str):
|
||
session = 返修测试.发起并产出候选(env, command=command)
|
||
return _从既有返修会话准备实验(env, command, session)
|
||
|
||
|
||
def _从既有返修会话准备实验(env, command: str, session: dict):
|
||
candidate_id = session["rounds"][0]["candidate_id"]
|
||
publication = 返修比较数据集发布(
|
||
dataset_id="revision-pair-" + command,
|
||
revision=1,
|
||
source=返修比较定位(
|
||
session_id=session["session_id"],
|
||
round=1,
|
||
candidate_id=candidate_id,
|
||
),
|
||
approval_ref="author-approved-fixed-pair:" + command,
|
||
expires_at=datetime.now(UTC) + timedelta(minutes=30),
|
||
)
|
||
maint = 评测服务(env["pools"][用途.维护])
|
||
data = maint.发布返修比较数据集(
|
||
env["maint_actor"], publication, env["库"], env["装配"].任务运行
|
||
)
|
||
request = 返修比较实验请求(
|
||
dataset_version_id=data["version_id"],
|
||
dataset_hash=data["public_hash"],
|
||
judges=(
|
||
配置选择(config_id="revision-judge-1", version="1"),
|
||
配置选择(config_id="revision-judge-2", version="1"),
|
||
),
|
||
dimensions=("事实保真", "表达有效性"),
|
||
max_cost_usd=Decimal("4"),
|
||
max_calls_per_sample=6,
|
||
)
|
||
svc = env["eval_app"].要求评测()
|
||
exp = svc.创建返修比较实验(env["eval_actor"], "compare-" + command, request)
|
||
return session, candidate_id, publication, data, exp
|
||
|
||
|
||
def _启动并运行(env, exp, *, allow_failure=False):
|
||
svc = env["eval_app"].要求评测()
|
||
face = svc.启动实验(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
实验执行请求(
|
||
approval_ref="author-approved-judge-budget",
|
||
deadline=datetime.now(UTC) + timedelta(minutes=10),
|
||
),
|
||
)
|
||
_运行就绪任务(env, allow_failure=allow_failure)
|
||
return face
|
||
|
||
|
||
def _运行就绪任务(env, *, allow_failure=False):
|
||
for _ in range(20):
|
||
claim = env["eval_app"].任务运行.领取步骤(
|
||
"revision-pair-worker", ["eval.prepare", "eval.call.judge"]
|
||
)
|
||
if claim is None:
|
||
break
|
||
try:
|
||
env["eval_app"].任务运行.执行一步(claim)
|
||
except Muse错误:
|
||
if not allow_failure:
|
||
raise
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w23-239001",
|
||
environment="隔离PG/合成HTTP及浏览器",
|
||
given="具名作者、固定来源、隔离测试配置与真实模块存储",
|
||
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
||
then=["确切返修候选只经双盲评委比较并以只读证明返回B06"],
|
||
contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md",
|
||
)
|
||
def test_确切返修候选只经双盲评委比较并以只读证明返回B06__239001(比较环境):
|
||
env = 比较环境
|
||
session, candidate_id, publication, data, exp = _准备实验(env, "fixed-pair")
|
||
visible = env["eval_app"].要求评测().读取数据集(env["eval_actor"], data["version_id"])
|
||
encoded = json.dumps(visible, ensure_ascii=False)
|
||
assert env["原稿"]["visible_text"] not in encoded
|
||
assert env["正文"].读取候选(env["作者"], candidate_id)["visible_text"] not in encoded
|
||
with pytest.raises(Muse错误, match="不能重新生成"):
|
||
env["eval_app"].要求评测().读取生成输入(
|
||
env["eval_actor"], exp["experiment_id"], data["sample_id"]
|
||
)
|
||
|
||
duplicate = 评测服务(env["pools"][用途.维护]).发布返修比较数据集(
|
||
env["maint_actor"], publication, env["库"], env["装配"].任务运行
|
||
)
|
||
assert duplicate["duplicate"] is True and duplicate["source_hash"] == data["source_hash"]
|
||
face = _启动并运行(env, exp)
|
||
assert len(face["units"]) == 2
|
||
assert {u["kind"] for u in face["units"]} == {"comparison"}
|
||
report = env["eval_app"].要求评测().读取实验报告(env["eval_actor"], exp["experiment_id"])
|
||
sample = report["samples"][0]
|
||
assert report["adapter"] == "fixed_revision_pair_v1"
|
||
assert sample["generation_complete"] and sample["comparison_complete"]
|
||
assert sample["outcome"] == "candidate" and len(sample["decisions"]) == 2
|
||
assert len(env["received_judges"]) == 2
|
||
materials = [json.loads(r["input"]) for r in env["received_judges"]]
|
||
assert all(set(m) == {"left", "right", "dimensions"} for m in materials)
|
||
assert materials[0]["left"]["text"] == materials[1]["right"]["text"]
|
||
assert materials[0]["right"]["text"] == materials[1]["left"]["text"]
|
||
|
||
proof = (
|
||
env["eval_app"]
|
||
.要求评测()
|
||
.导出返修比较证明(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
env["pools"][用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
)
|
||
calls = len(env["received_judges"])
|
||
read = (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, proof["receipt_id"])
|
||
)
|
||
assert read["proof"]["outcome"] == "candidate"
|
||
assert read["proof"]["validation_modes"] == ["offline_contract"]
|
||
assert "uncalibrated_pairwise_preference" in read["proof"]["reasons"]
|
||
assert read["current"] == {
|
||
"stopped": False,
|
||
"selection": "candidate",
|
||
"reasons": read["proof"]["reasons"],
|
||
"automatic_adoption": False,
|
||
"production_activation": "not_evaluated",
|
||
}
|
||
assert (
|
||
env["eval_app"]
|
||
.要求评测()
|
||
.导出返修比较证明(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
env["pools"][用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
== proof
|
||
)
|
||
assert len(env["received_judges"]) == calls
|
||
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
|
||
assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided"
|
||
|
||
wrong = publication.model_copy(
|
||
update={
|
||
"dataset_id": "wrong-candidate",
|
||
"source": publication.source.model_copy(update={"candidate_id": uuid4()}),
|
||
}
|
||
)
|
||
with pytest.raises(Muse错误, match="同一模型候选"):
|
||
评测服务(env["pools"][用途.维护]).发布返修比较数据集(
|
||
env["maint_actor"], wrong, env["库"], env["装配"].任务运行
|
||
)
|
||
with pytest.raises(Muse错误):
|
||
评测服务(env["pools"][用途.维护]).发布返修比较数据集(
|
||
replace(env["maint_actor"], 作者="other-author"),
|
||
publication.model_copy(update={"dataset_id": "wrong-author"}),
|
||
env["库"],
|
||
env["装配"].任务运行,
|
||
)
|
||
assert session["rounds"][0]["candidate_id"] == candidate_id
|
||
|
||
|
||
@pytest.mark.case_id(
|
||
"NC-w23-239002",
|
||
environment="隔离PG/合成HTTP及浏览器",
|
||
given="具名作者、固定来源、隔离测试配置与真实模块存储",
|
||
when="通过实际公开入口执行并核对拒绝、状态及持久结果",
|
||
then=["评委失败时旧证明保留原文且恢复只补原比较任务"],
|
||
contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md",
|
||
)
|
||
def test_评委失败时旧证明保留原文且恢复只补原比较任务__239002(比较环境):
|
||
env = 比较环境
|
||
_, candidate_id, _, _, exp = _准备实验(env, "recover-pair")
|
||
env["scripted_judges"].extend(["candidate", "api_error"])
|
||
_启动并运行(env, exp, allow_failure=True)
|
||
svc = env["eval_app"].要求评测()
|
||
failed = svc.读取执行工作面(env["eval_actor"], exp["experiment_id"])
|
||
assert sorted(u["state"] for u in failed["units"]) == ["completed", "failed"]
|
||
old = svc.导出返修比较证明(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
env["pools"][用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
old_read = (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"])
|
||
)
|
||
assert old_read["proof"]["comparison_complete"] is False
|
||
assert old_read["current"]["selection"] == "original"
|
||
failed_task = next(u["task_id"] for u in failed["units"] if u["state"] == "failed")
|
||
env["eval_app"].任务运行.控制任务(
|
||
failed_task,
|
||
env["eval_actor"].作者,
|
||
任务状态.已失败,
|
||
"恢复",
|
||
命令ID="recover-fixed-pair-judge",
|
||
)
|
||
_运行就绪任务(env)
|
||
final = svc.读取实验报告(env["eval_actor"], exp["experiment_id"])
|
||
assert final["state"] == "completed"
|
||
assert final["samples"][0]["outcome"] == "candidate"
|
||
assert len(env["received_judges"]) == 3
|
||
new = svc.导出返修比较证明(
|
||
env["eval_actor"],
|
||
exp["experiment_id"],
|
||
env["pools"][用途.维护],
|
||
env["maint_actor"],
|
||
)
|
||
assert new["receipt_id"] != old["receipt_id"]
|
||
assert (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"])["current"][
|
||
"selection"
|
||
]
|
||
== "original"
|
||
)
|
||
assert (
|
||
env["装配"]
|
||
.要求审校()
|
||
.读取返修比较证明(env["作者"], env["装配"].任务运行, new["receipt_id"])["current"][
|
||
"selection"
|
||
]
|
||
== "candidate"
|
||
)
|
||
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
|
||
assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided"
|