"""固定B06返修候选经B10双评委比较;合成HTTP不认证真实外部模型或文学标定。""" from __future__ import annotations import copy import json import os import socket import subprocess import sys import time from dataclasses import replace from datetime import UTC, datetime, timedelta from decimal import Decimal from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path from threading import Thread from uuid import uuid4 import httpx import pytest import test_持久返修会话 as 返修测试 import uvicorn from fastapi.testclient import TestClient from muse.任务运行.接口 import ( 任务状态, 内容哈希, 凭据引用, 提供方配置, 角色策略目录, 运行配置内容, 配置版本管理, 配置验证证据, 预算管理, 额度策略, ) from muse.共享.调用身份 import 内容用途, 用途, 调用身份 from muse.共享.错误 import Muse错误 from muse.启动 import 构建 from muse.审校修订.接口 import 返修比较定位 from muse.接入.http.应用 import 创建应用 from muse.效果评测.接口 import ( 实验执行请求, 评测服务, 返修比较实验请求, 返修比较数据集发布, 配置选择, ) from muse.编排.执行评测 import 登记评测执行 from muse.配置 import 应用配置, 服务配置 pytestmark = pytest.mark.数据库 生成环境 = 返修测试.生成环境 修订环境 = 返修测试.修订环境 @pytest.mark.case_id( "NC-w23-239007", environment="隔离PG/合成HTTP及浏览器", given="具名作者、固定来源、隔离测试配置与真实模块存储", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["双评委记录经真实浏览器展示后作者可保持原文"], contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md", ) @pytest.mark.浏览器 def test_双评委记录经真实浏览器展示后作者可保持原文__239007(比较环境, tmp_path): if os.environ.get("MUSE_RUN_BROWSER") != "1": pytest.skip("需显式启用浏览器") env = 比较环境 session, _, _, _, exp = _准备实验(env, "browser-pair") _启动并运行(env, exp) proof = ( env["eval_app"] .要求评测() .导出返修比较证明( env["eval_actor"], exp["experiment_id"], env["pools"][用途.维护], env["maint_actor"], ) ) with socket.socket() as sock: sock.bind(("127.0.0.1", 0)) port = sock.getsockname()[1] base = f"http://127.0.0.1:{port}" secret = tmp_path / "browser-password" secret.write_text("synthetic-comparison-browser") secret.chmod(0o600) config = replace( env["装配"].配置, HTTP=服务配置( str(secret), 作者ID="gen-author", 公开地址=base, 允许来源=(base,), ), ) server = uvicorn.Server( uvicorn.Config( 创建应用(config), host="127.0.0.1", port=port, log_level="warning", ) ) thread = Thread(target=server.run, daemon=True) thread.start() output = Path(os.environ.get("MUSE_BROWSER_ARTIFACTS", str(tmp_path / "browser"))).resolve() output.mkdir(parents=True, exist_ok=True) try: for _ in range(100): if server.started: break time.sleep(0.05) assert server.started and httpx.get(base + "/ready", trust_env=False).status_code == 200 environment = { **os.environ, "MUSE_WORKBENCH_URL": base, "MUSE_AUTHOR_PASSWORD_FILE": str(secret), "MUSE_REVISION_COMPARISON_RECEIPT": proof["receipt_id"], "MUSE_REVISION_SESSION_URL": ( "/works/gen-work/review?chapter=ch-4" f"&diagnosis={env['授权'].diagnosis_id}&revision_session={session['session_id']}" ), "MUSE_BROWSER_EXECUTABLE": ( "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome" ), } with (output / "浏览器.json").open("w") as report: result = subprocess.run( [ "pnpm", "exec", "playwright", "test", "返修独立比较.spec.ts", "--reporter=json", "--output", str(output / "产物"), ], cwd=Path(__file__).resolve().parents[2] / "web", env=environment, stdout=report, stderr=subprocess.STDOUT, timeout=90, ) assert result.returncode == 0, "浏览器未通过,见浏览器.json" final = ( env["装配"] .要求审校() .读取返修会话( env["作者"], env["装配"].任务运行, session["session_id"], ) ) assert final["state"] == "closed" and final["author_choice"] == "original" assert len(env["received_judges"]) == 2 assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"] finally: server.should_exit = True thread.join(10) assert not thread.is_alive() @pytest.mark.case_id( "NC-w23-239006", environment="隔离PG/合成HTTP及浏览器", given="具名作者、固定来源、隔离测试配置与真实模块存储", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["返修比较经真实CLI封存默认HTTP执行及生产只读回查"], contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md", ) def test_返修比较经真实CLI封存默认HTTP执行及生产只读回查__239006(比较环境, tmp_path): env = 比较环境 session, _, publication, data, exp = _准备实验(env, "entry-pair") secret = tmp_path / "author-password" secret.write_text("synthetic-pair-entry") secret.chmod(0o600) def 配置文件(purpose): config = tmp_path / (purpose.value + ".toml") config.write_text( '["数据库"]\n"取值方式"="受控存储"\n"位置"=' + json.dumps(env["pools"][purpose].引用.位置) + '\n["资源"]\n"发布身份"=' + json.dumps(env["装配"].配置.资源发布身份) + '\n["运行"]\n"用途"=' + json.dumps(purpose.value) + '\n[HTTP]\n"作者ID"="gen-author"\n"口令文件"=' + json.dumps(str(secret)) + '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n' ) return config production, evaluation, maintenance = (配置文件(p) for p in (用途.生产, 用途.评测, 用途.维护)) def cli(group, config, action, target, *extra): result = subprocess.run( [ sys.executable, "-I", "-m", "muse", group, str(config), action, str(target), *map(str, extra), ], cwd=tmp_path, capture_output=True, text=True, timeout=30, ) assert result.returncode == 0, result.stderr return json.loads(result.stdout) request_file = tmp_path / "pair.json" request_file.write_text(publication.model_dump_json()) sealed = cli("评测", maintenance, "发布返修数据集", request_file, "--来源配置", production) assert sealed["version_id"] == data["version_id"] and sealed["duplicate"] cfg = replace( env["eval_app"].配置, HTTP=服务配置( str(secret), 作者ID="gen-author", 公开地址="http://testserver", 允许来源=("http://testserver",), ), ) creator = { "command_id": "compare-entry-pair", "request": { "dataset_version_id": data["version_id"], "dataset_hash": data["public_hash"], "judges": [ {"config_id": "revision-judge-1", "version": "1"}, {"config_id": "revision-judge-2", "version": "1"}, ], "dimensions": ["事实保真", "表达有效性"], "max_cost_usd": "4", "max_calls_per_sample": 6, }, } app = 创建应用(cfg, 计价=_合成计价()) with TestClient(app, headers={"Origin": "http://testserver"}) as client: assert ( client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200 ) created = client.post("/api/v1/evaluation/revision-experiments", json=creator) assert created.status_code == 201, created.text assert created.json()["experiment_id"] == exp["experiment_id"] started = client.post( f"/api/v1/evaluation/experiments/{exp['experiment_id']}/execution", json={ "approval_ref": "synthetic-pair-budget", "deadline": (datetime.now(UTC) + timedelta(minutes=10)).isoformat(), }, ) assert started.status_code == 201, started.text tasks = [unit["task_id"] for unit in started.json()["units"]] assert len(tasks) == 2 and env["received_judges"] == [] for tid in tasks: for _ in range(3): step = client.post(f"/api/v1/tasks/{tid}/advance") assert step.status_code == 200, step.text if step.json()["state"] == "completed": break assert step.json()["state"] == "completed" assert len(env["received_judges"]) == 2 exported = cli( "评测", evaluation, "导出返修证明", exp["experiment_id"], "--维护配置", maintenance ) receipt_id = exported["receipt_id"] read = cli("审校", production, "返修比较证明", receipt_id) assert read["current"]["selection"] == "candidate" assert not read["current"]["automatic_adoption"] from muse.配置 import 读取配置 with TestClient( 创建应用(读取配置(production)), headers={"Origin": "http://testserver"} ) as client: assert ( client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200 ) path = f"/api/v1/revision-sessions/{session['session_id']}/comparisons/{receipt_id}" returned = client.get(path) assert returned.status_code == 200 and returned.json() == read wrong = client.get(f"/api/v1/revision-sessions/{uuid4()}/comparisons/{receipt_id}") assert wrong.status_code == 403 and wrong.json()["code"] == "SCOPE_DENIED" assert len(env["received_judges"]) == 2 assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"] class _合成计价: 版本 = "revision-pair-synthetic-price-v1" def 金额(self, 结果): return Decimal("0.125") if 结果.用量 else None @pytest.fixture def 比较环境(修订环境, tmp_path, monkeypatch): # 返修来源与评测维护必须同库:复用修订环境所在库,另借新库会被身份守卫拒绝。 应用测试库 = 修订环境["库组"] judges = ("synthetic-revision-judge-a", "synthetic-revision-judge-b") policy = 角色策略目录.从发布包() definition = copy.deepcopy(policy.定义) definition["models"]["fixed"].extend(judges) definition["models"]["judge-fixed"].extend(judges) isolated = 角色策略目录( definition, 资源发布身份=policy.资源发布身份, 角色资源=policy.角色资源, ) monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda cls: isolated)) received: list[dict] = [] scripted: list[str] = [] class Handler(BaseHTTPRequestHandler): def do_POST(self): request = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) received.append(request) action = scripted.pop(0) if scripted else "candidate" material = json.loads(request["input"]) if action == "api_error": response = { "id": f"revision-pair-{len(received)}", "model": None, "status": "failed", "output": [], "usage": {"input_tokens": 0, "output_tokens": 0}, "error": {"code": "synthetic_failure", "message": "隔离评委失败"}, } else: candidate_side = next( side for side in ("left", "right") if "值得注意的是" not in material[side]["text"] ) evidence = [ {"side": side, "quote": material[side]["text"]} for side in ("left", "right") ] output = { "choice": candidate_side, "rationale": "隔离评委选择删去无功能元话语的一侧", "dimensions": [ { "dimension": dimension, "choice": candidate_side, "rationale": "两侧原字可核,候选更直接且事实保持", "evidence": evidence, } for dimension in material["dimensions"] ], } response = { "id": f"revision-pair-{len(received)}", "model": request["model"], "status": "completed", "output": [ { "type": "message", "content": [ { "type": "output_text", "text": json.dumps(output, ensure_ascii=False), } ], } ], "usage": {"input_tokens": 5, "output_tokens": 7}, } payload = ( "data: " + json.dumps({"type": f"response.{response['status']}", "response": response}) + "\n\n" ).encode() self.send_response(200) self.send_header("Content-Type", "text/event-stream") self.send_header("Content-Length", str(len(payload))) self.end_headers() self.wfile.write(payload) def log_message(self, format, *args): pass server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) thread = Thread(target=server.serve_forever, daemon=True) thread.start() try: pool = 应用测试库[用途.评测] app = 构建( 应用配置( pool.引用, isolated.资源发布身份, 运行用途=用途.评测, 原文暂存=str(tmp_path / "evaluation-raw"), ) ) with app.生命周期(): 登记评测执行(app, _合成计价()) key = tmp_path / "judge-key" key.write_text("synthetic-only") key.chmod(0o600) class Validator: 身份 = "synthetic-revision-comparison-offline" def 验证(self, 内容, 执行用途): return 配置验证证据( 内容哈希(内容.冻结()), 内容.角色策略版本, 内容.资源发布身份, 执行用途, ("synthetic:isolated-revision-pair-http",), "offline_contract", ) manager = 配置版本管理(pool, Validator()) for index, model in enumerate(judges, start=1): config_id = f"revision-judge-{index}" content = 运行配置内容( "direct", "1", definition["version"], isolated.资源发布身份, "revision-pair-budget", { "judge": { "provider": "synthetic", "model": model, "thinking": "high", "tools": [], } }, (凭据引用("key", "受控存储", str(key)),), ( 提供方配置( "synthetic", "responses", f"http://127.0.0.1:{server.server_port}/v1/responses", "key", ), ), _合成计价.版本, ) manager.保存草案(config_id, "1", content) receipt = manager.验证版本(config_id, "1") manager.启用( config_id, "1", 验证回执=receipt, 批准引用="synthetic-revision-pair", 预期代次=0, ) 预算管理(pool, "revision-pair-budget").登记策略( 额度策略("revision-pair-budget", "1", Decimal("20"), 20) ) yield { **修订环境, "pools": 应用测试库, "eval_app": app, "eval_actor": 调用身份("gen-author", None, 用途.评测, 内容用途.检测), "maint_actor": 调用身份("gen-author", None, 用途.维护, 内容用途.检测), "received_judges": received, "scripted_judges": scripted, } finally: server.shutdown() server.server_close() thread.join(timeout=5) def _准备实验(env, command: str): session = 返修测试.发起并产出候选(env, command=command) return _从既有返修会话准备实验(env, command, session) def _从既有返修会话准备实验(env, command: str, session: dict): candidate_id = session["rounds"][0]["candidate_id"] publication = 返修比较数据集发布( dataset_id="revision-pair-" + command, revision=1, source=返修比较定位( session_id=session["session_id"], round=1, candidate_id=candidate_id, ), approval_ref="author-approved-fixed-pair:" + command, expires_at=datetime.now(UTC) + timedelta(minutes=30), ) maint = 评测服务(env["pools"][用途.维护]) data = maint.发布返修比较数据集( env["maint_actor"], publication, env["库"], env["装配"].任务运行 ) request = 返修比较实验请求( dataset_version_id=data["version_id"], dataset_hash=data["public_hash"], judges=( 配置选择(config_id="revision-judge-1", version="1"), 配置选择(config_id="revision-judge-2", version="1"), ), dimensions=("事实保真", "表达有效性"), max_cost_usd=Decimal("4"), max_calls_per_sample=6, ) svc = env["eval_app"].要求评测() exp = svc.创建返修比较实验(env["eval_actor"], "compare-" + command, request) return session, candidate_id, publication, data, exp def _启动并运行(env, exp, *, allow_failure=False): svc = env["eval_app"].要求评测() face = svc.启动实验( env["eval_actor"], exp["experiment_id"], 实验执行请求( approval_ref="author-approved-judge-budget", deadline=datetime.now(UTC) + timedelta(minutes=10), ), ) _运行就绪任务(env, allow_failure=allow_failure) return face def _运行就绪任务(env, *, allow_failure=False): for _ in range(20): claim = env["eval_app"].任务运行.领取步骤( "revision-pair-worker", ["eval.prepare", "eval.call.judge"] ) if claim is None: break try: env["eval_app"].任务运行.执行一步(claim) except Muse错误: if not allow_failure: raise @pytest.mark.case_id( "NC-w23-239001", environment="隔离PG/合成HTTP及浏览器", given="具名作者、固定来源、隔离测试配置与真实模块存储", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["确切返修候选只经双盲评委比较并以只读证明返回B06"], contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md", ) def test_确切返修候选只经双盲评委比较并以只读证明返回B06__239001(比较环境): env = 比较环境 session, candidate_id, publication, data, exp = _准备实验(env, "fixed-pair") visible = env["eval_app"].要求评测().读取数据集(env["eval_actor"], data["version_id"]) encoded = json.dumps(visible, ensure_ascii=False) assert env["原稿"]["visible_text"] not in encoded assert env["正文"].读取候选(env["作者"], candidate_id)["visible_text"] not in encoded with pytest.raises(Muse错误, match="不能重新生成"): env["eval_app"].要求评测().读取生成输入( env["eval_actor"], exp["experiment_id"], data["sample_id"] ) duplicate = 评测服务(env["pools"][用途.维护]).发布返修比较数据集( env["maint_actor"], publication, env["库"], env["装配"].任务运行 ) assert duplicate["duplicate"] is True and duplicate["source_hash"] == data["source_hash"] face = _启动并运行(env, exp) assert len(face["units"]) == 2 assert {u["kind"] for u in face["units"]} == {"comparison"} report = env["eval_app"].要求评测().读取实验报告(env["eval_actor"], exp["experiment_id"]) sample = report["samples"][0] assert report["adapter"] == "fixed_revision_pair_v1" assert sample["generation_complete"] and sample["comparison_complete"] assert sample["outcome"] == "candidate" and len(sample["decisions"]) == 2 assert len(env["received_judges"]) == 2 materials = [json.loads(r["input"]) for r in env["received_judges"]] assert all(set(m) == {"left", "right", "dimensions"} for m in materials) assert materials[0]["left"]["text"] == materials[1]["right"]["text"] assert materials[0]["right"]["text"] == materials[1]["left"]["text"] proof = ( env["eval_app"] .要求评测() .导出返修比较证明( env["eval_actor"], exp["experiment_id"], env["pools"][用途.维护], env["maint_actor"], ) ) calls = len(env["received_judges"]) read = ( env["装配"] .要求审校() .读取返修比较证明(env["作者"], env["装配"].任务运行, proof["receipt_id"]) ) assert read["proof"]["outcome"] == "candidate" assert read["proof"]["validation_modes"] == ["offline_contract"] assert "uncalibrated_pairwise_preference" in read["proof"]["reasons"] assert read["current"] == { "stopped": False, "selection": "candidate", "reasons": read["proof"]["reasons"], "automatic_adoption": False, "production_activation": "not_evaluated", } assert ( env["eval_app"] .要求评测() .导出返修比较证明( env["eval_actor"], exp["experiment_id"], env["pools"][用途.维护], env["maint_actor"], ) == proof ) assert len(env["received_judges"]) == calls assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"] assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided" wrong = publication.model_copy( update={ "dataset_id": "wrong-candidate", "source": publication.source.model_copy(update={"candidate_id": uuid4()}), } ) with pytest.raises(Muse错误, match="同一模型候选"): 评测服务(env["pools"][用途.维护]).发布返修比较数据集( env["maint_actor"], wrong, env["库"], env["装配"].任务运行 ) with pytest.raises(Muse错误): 评测服务(env["pools"][用途.维护]).发布返修比较数据集( replace(env["maint_actor"], 作者="other-author"), publication.model_copy(update={"dataset_id": "wrong-author"}), env["库"], env["装配"].任务运行, ) assert session["rounds"][0]["candidate_id"] == candidate_id @pytest.mark.case_id( "NC-w23-239002", environment="隔离PG/合成HTTP及浏览器", given="具名作者、固定来源、隔离测试配置与真实模块存储", when="通过实际公开入口执行并核对拒绝、状态及持久结果", then=["评委失败时旧证明保留原文且恢复只补原比较任务"], contract="docs/系统架构/新版设计/模块设计/B06-审校修订.md", ) def test_评委失败时旧证明保留原文且恢复只补原比较任务__239002(比较环境): env = 比较环境 _, candidate_id, _, _, exp = _准备实验(env, "recover-pair") env["scripted_judges"].extend(["candidate", "api_error"]) _启动并运行(env, exp, allow_failure=True) svc = env["eval_app"].要求评测() failed = svc.读取执行工作面(env["eval_actor"], exp["experiment_id"]) assert sorted(u["state"] for u in failed["units"]) == ["completed", "failed"] old = svc.导出返修比较证明( env["eval_actor"], exp["experiment_id"], env["pools"][用途.维护], env["maint_actor"], ) old_read = ( env["装配"] .要求审校() .读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"]) ) assert old_read["proof"]["comparison_complete"] is False assert old_read["current"]["selection"] == "original" failed_task = next(u["task_id"] for u in failed["units"] if u["state"] == "failed") env["eval_app"].任务运行.控制任务( failed_task, env["eval_actor"].作者, 任务状态.已失败, "恢复", 命令ID="recover-fixed-pair-judge", ) _运行就绪任务(env) final = svc.读取实验报告(env["eval_actor"], exp["experiment_id"]) assert final["state"] == "completed" assert final["samples"][0]["outcome"] == "candidate" assert len(env["received_judges"]) == 3 new = svc.导出返修比较证明( env["eval_actor"], exp["experiment_id"], env["pools"][用途.维护], env["maint_actor"], ) assert new["receipt_id"] != old["receipt_id"] assert ( env["装配"] .要求审校() .读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"])["current"][ "selection" ] == "original" ) assert ( env["装配"] .要求审校() .读取返修比较证明(env["作者"], env["装配"].任务运行, new["receipt_id"])["current"][ "selection" ] == "candidate" ) assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"] assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided"