muse-agent-example/tests/集成/test_返修独立比较.py
zizi 9e6f1c4481 R2 改造交付:新版模块化单体全量成果
- src/muse 新版全模块(装配/共享/上下文/任务运行/作品规划/故事世界/正文写作/审校修订/知识方法/作者经验/效果评测/交付连载/资料研究/正式变更/元数据/接入/基础设施/编排)+ 测试树(单元/契约/集成/架构/迁移/端到端/夹具)
- 129 项功能全部实现与自动验证(功能覆盖.json/矩阵),含 W31 补齐的规则与代价/节奏安排/伏笔与承诺
- 旧实现按处置清单退出(702 条中 324 删,保护合同与未迁移条目留存有据);web/app.py 旧工作台退役,新工作台为唯一写入口
- 数据库/旧库迁移:真实旧库内容批次迁移链(端点守卫/PG作品正文映射/质量资产缺省投影)
- 运行手册 docs/运行手册.md;W30 本机服务阶段一已运行(infra PG 为正式内容权威)
- R2 执行证据与私有运行材料在 .agents.local/改造/R2-20260909/(不入库)
2026-09-15 12:47:42 +08:00

673 lines
25 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""固定B06返修候选经B10双评委比较;合成HTTP不认证真实外部模型或文学标定。"""
from __future__ import annotations
import copy
import json
import os
import socket
import subprocess
import sys
import time
from dataclasses import replace
from datetime import UTC, datetime, timedelta
from decimal import Decimal
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from threading import Thread
from uuid import uuid4
import httpx
import pytest
import test_持久返修会话 as 返修测试
import uvicorn
from fastapi.testclient import TestClient
from muse.任务运行.接口 import (
任务状态,
内容哈希,
凭据引用,
提供方配置,
角色策略目录,
运行配置内容,
配置版本管理,
配置验证证据,
预算管理,
额度策略,
)
from muse.共享.调用身份 import 内容用途, 用途, 调用身份
from muse.共享.错误 import Muse错误
from muse.启动 import 构建
from muse.审校修订.接口 import 返修比较定位
from muse.接入.http.应用 import 创建应用
from muse.效果评测.接口 import (
实验执行请求,
评测服务,
返修比较实验请求,
返修比较数据集发布,
配置选择,
)
from muse.编排.执行评测 import 登记评测执行
from muse.配置 import 应用配置, 服务配置
pytestmark = pytest.mark.数据库
生成环境 = 返修测试.生成环境
修订环境 = 返修测试.修订环境
@pytest.mark.浏览器
def test_双评委记录经真实浏览器展示后作者可保持原文__239007(比较环境, tmp_path):
if os.environ.get("MUSE_RUN_BROWSER") != "1":
pytest.skip("需显式启用浏览器")
env = 比较环境
session, _, _, _, exp = _准备实验(env, "browser-pair")
_启动并运行(env, exp)
proof = (
env["eval_app"]
.要求评测()
.导出返修比较证明(
env["eval_actor"],
exp["experiment_id"],
env["pools"][用途.维护],
env["maint_actor"],
)
)
with socket.socket() as sock:
sock.bind(("127.0.0.1", 0))
port = sock.getsockname()[1]
base = f"http://127.0.0.1:{port}"
secret = tmp_path / "browser-password"
secret.write_text("synthetic-comparison-browser")
secret.chmod(0o600)
config = replace(
env["装配"].配置,
HTTP=服务配置(
str(secret),
作者ID="gen-author",
公开地址=base,
允许来源=(base,),
),
)
server = uvicorn.Server(
uvicorn.Config(
创建应用(config),
host="127.0.0.1",
port=port,
log_level="warning",
)
)
thread = Thread(target=server.run, daemon=True)
thread.start()
output = Path(os.environ.get("MUSE_BROWSER_ARTIFACTS", str(tmp_path / "browser"))).resolve()
output.mkdir(parents=True, exist_ok=True)
try:
for _ in range(100):
if server.started:
break
time.sleep(0.05)
assert server.started and httpx.get(base + "/ready", trust_env=False).status_code == 200
environment = {
**os.environ,
"MUSE_WORKBENCH_URL": base,
"MUSE_AUTHOR_PASSWORD_FILE": str(secret),
"MUSE_REVISION_COMPARISON_RECEIPT": proof["receipt_id"],
"MUSE_REVISION_SESSION_URL": (
"/works/gen-work/review?chapter=ch-4"
f"&diagnosis={env['授权'].diagnosis_id}&revision_session={session['session_id']}"
),
"MUSE_BROWSER_EXECUTABLE": (
"/Applications/Google Chrome.app/Contents/MacOS/Google Chrome"
),
}
with (output / "浏览器.json").open("w") as report:
result = subprocess.run(
[
"pnpm",
"exec",
"playwright",
"test",
"返修独立比较.spec.ts",
"--reporter=json",
"--output",
str(output / "产物"),
],
cwd=Path(__file__).resolve().parents[2] / "web",
env=environment,
stdout=report,
stderr=subprocess.STDOUT,
timeout=90,
)
assert result.returncode == 0, "浏览器未通过,见浏览器.json"
final = (
env["装配"]
.要求审校()
.读取返修会话(
env["作者"],
env["装配"].任务运行,
session["session_id"],
)
)
assert final["state"] == "closed" and final["author_choice"] == "original"
assert len(env["received_judges"]) == 2
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
finally:
server.should_exit = True
thread.join(10)
assert not thread.is_alive()
def test_返修比较经真实CLI封存默认HTTP执行及生产只读回查__239006(比较环境, tmp_path):
env = 比较环境
session, _, publication, data, exp = _准备实验(env, "entry-pair")
secret = tmp_path / "author-password"
secret.write_text("synthetic-pair-entry")
secret.chmod(0o600)
def 配置文件(purpose):
config = tmp_path / (purpose.value + ".toml")
config.write_text(
'["数据库"]\n"取值方式"="受控存储"\n"位置"='
+ json.dumps(env["pools"][purpose].引用.位置)
+ '\n["资源"]\n"发布身份"='
+ json.dumps(env["装配"].配置.资源发布身份)
+ '\n["运行"]\n"用途"='
+ json.dumps(purpose.value)
+ '\n[HTTP]\n"作者ID"="gen-author"\n"口令文件"='
+ json.dumps(str(secret))
+ '\n"公开地址"="http://testserver"\n"允许来源"=["http://testserver"]\n'
)
return config
production, evaluation, maintenance = (配置文件(p) for p in (用途.生产, 用途.评测, 用途.维护))
def cli(group, config, action, target, *extra):
result = subprocess.run(
[
sys.executable,
"-I",
"-m",
"muse",
group,
str(config),
action,
str(target),
*map(str, extra),
],
cwd=tmp_path,
capture_output=True,
text=True,
timeout=30,
)
assert result.returncode == 0, result.stderr
return json.loads(result.stdout)
request_file = tmp_path / "pair.json"
request_file.write_text(publication.model_dump_json())
sealed = cli("评测", maintenance, "发布返修数据集", request_file, "--来源配置", production)
assert sealed["version_id"] == data["version_id"] and sealed["duplicate"]
cfg = replace(
env["eval_app"].配置,
HTTP=服务配置(
str(secret),
作者ID="gen-author",
公开地址="http://testserver",
允许来源=("http://testserver",),
),
)
creator = {
"command_id": "compare-entry-pair",
"request": {
"dataset_version_id": data["version_id"],
"dataset_hash": data["public_hash"],
"judges": [
{"config_id": "revision-judge-1", "version": "1"},
{"config_id": "revision-judge-2", "version": "1"},
],
"dimensions": ["事实保真", "表达有效性"],
"max_cost_usd": "4",
"max_calls_per_sample": 6,
},
}
app = 创建应用(cfg, 计价=_合成计价())
with TestClient(app, headers={"Origin": "http://testserver"}) as client:
assert (
client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200
)
created = client.post("/api/v1/evaluation/revision-experiments", json=creator)
assert created.status_code == 201, created.text
assert created.json()["experiment_id"] == exp["experiment_id"]
started = client.post(
f"/api/v1/evaluation/experiments/{exp['experiment_id']}/execution",
json={
"approval_ref": "synthetic-pair-budget",
"deadline": (datetime.now(UTC) + timedelta(minutes=10)).isoformat(),
},
)
assert started.status_code == 201, started.text
tasks = [unit["task_id"] for unit in started.json()["units"]]
assert len(tasks) == 2 and env["received_judges"] == []
for tid in tasks:
for _ in range(3):
step = client.post(f"/api/v1/tasks/{tid}/advance")
assert step.status_code == 200, step.text
if step.json()["state"] == "completed":
break
assert step.json()["state"] == "completed"
assert len(env["received_judges"]) == 2
exported = cli(
"评测", evaluation, "导出返修证明", exp["experiment_id"], "--维护配置", maintenance
)
receipt_id = exported["receipt_id"]
read = cli("审校", production, "返修比较证明", receipt_id)
assert read["current"]["selection"] == "candidate"
assert not read["current"]["automatic_adoption"]
from muse.配置 import 读取配置
with TestClient(
创建应用(读取配置(production)), headers={"Origin": "http://testserver"}
) as client:
assert (
client.post("/api/v1/session", json={"password": secret.read_text()}).status_code == 200
)
path = f"/api/v1/revision-sessions/{session['session_id']}/comparisons/{receipt_id}"
returned = client.get(path)
assert returned.status_code == 200 and returned.json() == read
wrong = client.get(f"/api/v1/revision-sessions/{uuid4()}/comparisons/{receipt_id}")
assert wrong.status_code == 403 and wrong.json()["code"] == "SCOPE_DENIED"
assert len(env["received_judges"]) == 2
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
class _合成计价:
版本 = "revision-pair-synthetic-price-v1"
def 金额(self, 结果):
return Decimal("0.125") if 结果.用量 else None
@pytest.fixture
def 比较环境(修订环境, 应用测试库, tmp_path, monkeypatch):
judges = ("synthetic-revision-judge-a", "synthetic-revision-judge-b")
policy = 角色策略目录.从发布包()
definition = copy.deepcopy(policy.定义)
definition["models"]["fixed"].extend(judges)
isolated = 角色策略目录(
definition,
资源发布身份=policy.资源发布身份,
角色资源=policy.角色资源,
)
monkeypatch.setattr(角色策略目录, "从发布包", classmethod(lambda cls: isolated))
received: list[dict] = []
scripted: list[str] = []
class Handler(BaseHTTPRequestHandler):
def do_POST(self):
request = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
received.append(request)
action = scripted.pop(0) if scripted else "candidate"
material = json.loads(request["input"])
if action == "api_error":
response = {
"id": f"revision-pair-{len(received)}",
"model": None,
"status": "failed",
"output": [],
"usage": {"input_tokens": 0, "output_tokens": 0},
"error": {"code": "synthetic_failure", "message": "隔离评委失败"},
}
else:
candidate_side = next(
side
for side in ("left", "right")
if "值得注意的是" not in material[side]["text"]
)
evidence = [
{"side": side, "quote": material[side]["text"]} for side in ("left", "right")
]
output = {
"choice": candidate_side,
"rationale": "隔离评委选择删去无功能元话语的一侧",
"dimensions": [
{
"dimension": dimension,
"choice": candidate_side,
"rationale": "两侧原字可核,候选更直接且事实保持",
"evidence": evidence,
}
for dimension in material["dimensions"]
],
}
response = {
"id": f"revision-pair-{len(received)}",
"model": request["model"],
"status": "completed",
"output": [
{
"type": "message",
"content": [
{
"type": "output_text",
"text": json.dumps(output, ensure_ascii=False),
}
],
}
],
"usage": {"input_tokens": 5, "output_tokens": 7},
}
payload = (
"data: "
+ json.dumps({"type": f"response.{response['status']}", "response": response})
+ "\n\n"
).encode()
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(payload)))
self.end_headers()
self.wfile.write(payload)
def log_message(self, format, *args):
pass
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
thread = Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
pool = 应用测试库[用途.评测]
app = 构建(
应用配置(
pool.引用,
isolated.资源发布身份,
运行用途=用途.评测,
原文暂存=str(tmp_path / "evaluation-raw"),
)
)
登记评测执行(app, _合成计价())
key = tmp_path / "judge-key"
key.write_text("synthetic-only")
key.chmod(0o600)
class Validator:
身份 = "synthetic-revision-comparison-offline"
def 验证(self, 内容, 执行用途):
return 配置验证证据(
内容哈希(内容.冻结()),
内容.角色策略版本,
内容.资源发布身份,
执行用途,
("synthetic:isolated-revision-pair-http",),
"offline_contract",
)
manager = 配置版本管理(pool, Validator())
for index, model in enumerate(judges, start=1):
config_id = f"revision-judge-{index}"
content = 运行配置内容(
"direct",
"1",
definition["version"],
isolated.资源发布身份,
"revision-pair-budget",
{
"judge": {
"provider": "synthetic",
"model": model,
"thinking": "high",
"tools": [],
}
},
(凭据引用("key", "受控存储", str(key)),),
(
提供方配置(
"synthetic",
"responses",
f"http://127.0.0.1:{server.server_port}/v1/responses",
"key",
),
),
_合成计价.版本,
)
manager.保存草案(config_id, "1", content)
receipt = manager.验证版本(config_id, "1")
manager.启用(
config_id,
"1",
验证回执=receipt,
批准引用="synthetic-revision-pair",
预期代次=0,
)
预算管理(pool, "revision-pair-budget").登记策略(
额度策略("revision-pair-budget", "1", Decimal("20"), 20)
)
yield {
**修订环境,
"pools": 应用测试库,
"eval_app": app,
"eval_actor": 调用身份("gen-author", None, 用途.评测, 内容用途.检测),
"maint_actor": 调用身份("gen-author", None, 用途.维护, 内容用途.检测),
"received_judges": received,
"scripted_judges": scripted,
}
finally:
server.shutdown()
server.server_close()
thread.join(timeout=5)
def _准备实验(env, command: str):
session = 返修测试.发起并产出候选(env, command=command)
return _从既有返修会话准备实验(env, command, session)
def _从既有返修会话准备实验(env, command: str, session: dict):
candidate_id = session["rounds"][0]["candidate_id"]
publication = 返修比较数据集发布(
dataset_id="revision-pair-" + command,
revision=1,
source=返修比较定位(
session_id=session["session_id"],
round=1,
candidate_id=candidate_id,
),
approval_ref="author-approved-fixed-pair:" + command,
expires_at=datetime.now(UTC) + timedelta(minutes=30),
)
maint = 评测服务(env["pools"][用途.维护])
data = maint.发布返修比较数据集(
env["maint_actor"], publication, env["库"], env["装配"].任务运行
)
request = 返修比较实验请求(
dataset_version_id=data["version_id"],
dataset_hash=data["public_hash"],
judges=(
配置选择(config_id="revision-judge-1", version="1"),
配置选择(config_id="revision-judge-2", version="1"),
),
dimensions=("事实保真", "表达有效性"),
max_cost_usd=Decimal("4"),
max_calls_per_sample=6,
)
svc = env["eval_app"].要求评测()
exp = svc.创建返修比较实验(env["eval_actor"], "compare-" + command, request)
return session, candidate_id, publication, data, exp
def _启动并运行(env, exp, *, allow_failure=False):
svc = env["eval_app"].要求评测()
face = svc.启动实验(
env["eval_actor"],
exp["experiment_id"],
实验执行请求(
approval_ref="author-approved-judge-budget",
deadline=datetime.now(UTC) + timedelta(minutes=10),
),
)
_运行就绪任务(env, allow_failure=allow_failure)
return face
def _运行就绪任务(env, *, allow_failure=False):
for _ in range(20):
claim = env["eval_app"].任务运行.领取步骤(
"revision-pair-worker", ["eval.prepare", "eval.call.judge"]
)
if claim is None:
break
try:
env["eval_app"].任务运行.执行一步(claim)
except Muse错误:
if not allow_failure:
raise
def test_确切返修候选只经双盲评委比较并以只读证明返回B06__239001(比较环境):
env = 比较环境
session, candidate_id, publication, data, exp = _准备实验(env, "fixed-pair")
visible = env["eval_app"].要求评测().读取数据集(env["eval_actor"], data["version_id"])
encoded = json.dumps(visible, ensure_ascii=False)
assert env["原稿"]["visible_text"] not in encoded
assert env["正文"].读取候选(env["作者"], candidate_id)["visible_text"] not in encoded
with pytest.raises(Muse错误, match="不能重新生成"):
env["eval_app"].要求评测().读取生成输入(
env["eval_actor"], exp["experiment_id"], data["sample_id"]
)
duplicate = 评测服务(env["pools"][用途.维护]).发布返修比较数据集(
env["maint_actor"], publication, env["库"], env["装配"].任务运行
)
assert duplicate["duplicate"] is True and duplicate["source_hash"] == data["source_hash"]
face = _启动并运行(env, exp)
assert len(face["units"]) == 2
assert {u["kind"] for u in face["units"]} == {"comparison"}
report = env["eval_app"].要求评测().读取实验报告(env["eval_actor"], exp["experiment_id"])
sample = report["samples"][0]
assert report["adapter"] == "fixed_revision_pair_v1"
assert sample["generation_complete"] and sample["comparison_complete"]
assert sample["outcome"] == "candidate" and len(sample["decisions"]) == 2
assert len(env["received_judges"]) == 2
materials = [json.loads(r["input"]) for r in env["received_judges"]]
assert all(set(m) == {"left", "right", "dimensions"} for m in materials)
assert materials[0]["left"]["text"] == materials[1]["right"]["text"]
assert materials[0]["right"]["text"] == materials[1]["left"]["text"]
proof = (
env["eval_app"]
.要求评测()
.导出返修比较证明(
env["eval_actor"],
exp["experiment_id"],
env["pools"][用途.维护],
env["maint_actor"],
)
)
calls = len(env["received_judges"])
read = (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, proof["receipt_id"])
)
assert read["proof"]["outcome"] == "candidate"
assert read["proof"]["validation_modes"] == ["offline_contract"]
assert "uncalibrated_pairwise_preference" in read["proof"]["reasons"]
assert read["current"] == {
"stopped": False,
"selection": "candidate",
"reasons": read["proof"]["reasons"],
"automatic_adoption": False,
"production_activation": "not_evaluated",
}
assert (
env["eval_app"]
.要求评测()
.导出返修比较证明(
env["eval_actor"],
exp["experiment_id"],
env["pools"][用途.维护],
env["maint_actor"],
)
== proof
)
assert len(env["received_judges"]) == calls
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided"
wrong = publication.model_copy(
update={
"dataset_id": "wrong-candidate",
"source": publication.source.model_copy(update={"candidate_id": uuid4()}),
}
)
with pytest.raises(Muse错误, match="同一模型候选"):
评测服务(env["pools"][用途.维护]).发布返修比较数据集(
env["maint_actor"], wrong, env["库"], env["装配"].任务运行
)
with pytest.raises(Muse错误):
评测服务(env["pools"][用途.维护]).发布返修比较数据集(
replace(env["maint_actor"], 作者="other-author"),
publication.model_copy(update={"dataset_id": "wrong-author"}),
env["库"],
env["装配"].任务运行,
)
assert session["rounds"][0]["candidate_id"] == candidate_id
def test_评委失败时旧证明保留原文且恢复只补原比较任务__239002(比较环境):
env = 比较环境
_, candidate_id, _, _, exp = _准备实验(env, "recover-pair")
env["scripted_judges"].extend(["candidate", "api_error"])
_启动并运行(env, exp, allow_failure=True)
svc = env["eval_app"].要求评测()
failed = svc.读取执行工作面(env["eval_actor"], exp["experiment_id"])
assert sorted(u["state"] for u in failed["units"]) == ["completed", "failed"]
old = svc.导出返修比较证明(
env["eval_actor"],
exp["experiment_id"],
env["pools"][用途.维护],
env["maint_actor"],
)
old_read = (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"])
)
assert old_read["proof"]["comparison_complete"] is False
assert old_read["current"]["selection"] == "original"
failed_task = next(u["task_id"] for u in failed["units"] if u["state"] == "failed")
env["eval_app"].任务运行.控制任务(
failed_task,
env["eval_actor"].作者,
任务状态.已失败,
"恢复",
命令ID="recover-fixed-pair-judge",
)
_运行就绪任务(env)
final = svc.读取实验报告(env["eval_actor"], exp["experiment_id"])
assert final["state"] == "completed"
assert final["samples"][0]["outcome"] == "candidate"
assert len(env["received_judges"]) == 3
new = svc.导出返修比较证明(
env["eval_actor"],
exp["experiment_id"],
env["pools"][用途.维护],
env["maint_actor"],
)
assert new["receipt_id"] != old["receipt_id"]
assert (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, old["receipt_id"])["current"][
"selection"
]
== "original"
)
assert (
env["装配"]
.要求审校()
.读取返修比较证明(env["作者"], env["装配"].任务运行, new["receipt_id"])["current"][
"selection"
]
== "candidate"
)
assert env["正文"].读取正文(env["作者"], "ch-4") == env["原稿"]
assert env["正文"].读取候选(env["作者"], candidate_id)["decision"] == "undecided"