zizi d80df3ed7c 治理: 剩余问题收口——humanization 规则/样例数据库权威 + PG 集成全通过 + 行为评测脚手架 + raw 冲突备忘录
范围(不含 design-story-foundation、docs/design、docs/write-chapter、
craft/、humanization/README.md 等进行中改动):

1. humanization 规则/样例运行时数据库权威
   - db/ddl/111:example_ai_flavor_rule / example_ai_flavor_sample /
     example_ai_flavor_rule_event(append-only 生命周期留痕),已应用到 muse-example
   - deai/load_db.py:数据库装载器,激活门/重复检测/指纹与文件装载器同源;
     数据库失败关闭,不静默回退 Git 文件资产
   - humanization/tools/seed_rules_db.py:YAML 种子单事务同步,幂等、
     变化留痕、--strict 对 db-only 行失败关闭;真实库已种入 26 规则/107 样例
   - prevent/diagnose/revise 生产路径切到数据库读取(--offline 显式读文件),
     落库前新鲜度检查与合同声明来源一致;四个 SKILL.md 数据库合同同步
   - 真实库验证:规则库指纹与文件种子一致(v-609bc40e21d0b5db),
     三个生产脚本端到端从库装载通过

2. PostgreSQL 集成:显式授权后 9/9 通过
   - 此前被依赖门阻断的 6 个 _db/smoke 测试全部通过
   - extract rollback 冒烟改为回滚事务内自给夹具(pending 窗/草稿缺失时自建),
     不再依赖瞬时生产状态;夹具残留核验为 0

3. Skill 行为评测脚手架(真实执行数量仍为 0)
   - harness/evals/skill_eval.py:场景合同、六类评测范畴、适配器和结构化裁决报告
   - diagnose-ai-flavor 参考场景 4 条 + 管道自测 7 项通过
   - 真实模型适配器未授权时以稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭;
     清单登记 skill_behavior_eval 条目,默认被依赖门阻断

4. evaluate-frozen-replay raw 存储边界冲突
   - docs/2026-08-19 备忘录:平台 DB-first 合同(创始人批准)与回放链
     仓外 vault 强制的冲突事实、两个选项和裁决前约束;运行时合同未单方面改写

5. harness 自身修复
   - runner 对账语义:行为评测入口不参与测试资产双向等值,但登记文件必须存在;
     manifest 保留 skill_behavior_eval 布尔字段并校验类型
   - 新增 2 条对账回归用例

验证证据: harness 三组自测 15+15+7 通过;静态审计 32 Skill / 0 问题;
76 个非数据库条目通过;9 个 PostgreSQL 集成条目显式授权后通过;
行为评测条目默认阻断;py_compile 与 git diff --check 通过。
未调用真实模型、embedding 或额度;真实行为评测执行数量仍为 0。
2026-08-19 02:41:25 +08:00

242 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""技能 4「修订」确定性脚本层(专题-09 §5.4 / §6 / §7)。
铁律:没有诊断产物,修订拒绝启动;没有事实快照,自动降回 Audit。
语义级动作(仲裁五问、改写文本、成对选择判定)在本脚本之外产生;
本脚本只强制合同与机械检查:产物头 → 唯一匹配 patch → 硬门 → 复扫 →
成对选择校验 → 审计报告。人不点头,候选永远是候选——本脚本不写任何正文。
落库合同:example_run + example_quality_result(judge_kind=review,
dimension=ai_flavor_revision,绑候选稿 sha256);--offline 才不写库。
"""
import argparse
import hashlib
import json
import sys
from pathlib import Path
SCRIPT_DIR = Path(__file__).resolve().parent
AGENT_ROOT = SCRIPT_DIR.parents[3]
for _p in (AGENT_ROOT / "humanization" / "src",
AGENT_ROOT / ".claude" / "skills" / "access-database" / "scripts",
AGENT_ROOT / ".claude" / "skills" / "establish-voice-baseline" / "scripts"):
if str(_p) not in sys.path:
sys.path.insert(0, str(_p))
from deai import load, report as report_mod # noqa: E402
from establish_voice_baseline import load_current_baseline # noqa: E402
from deai.pairwise import PairwiseNotExecuted # noqa: E402
from deai.patch import PatchError # noqa: E402
from deai.pipeline import DowngradedToAudit, RevisionNotAuthorized, run_patch # noqa: E402
TENANT_ID = 1
CREATOR = "1"
class ReviseContractError(ValueError):
"""修订合同失败:缺诊断、缺授权或门禁失败关闭。"""
def _load_json(path: Path, what: str) -> dict:
try:
data = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
raise ReviseContractError(f"{what}不可读或不合法: {path} ({exc})") from exc
if not isinstance(data, dict):
raise ReviseContractError(f"{what}必须是 JSON 对象: {path}")
return data
def load_revision_library(from_db: bool) -> tuple[dict, dict, str]:
"""修订用规则库:生产读数据库(失败关闭,不回退 Git),离线读文件。"""
if from_db:
from db import connect
from deai import load_db
with connect(readonly=True) as conn:
samples = load_db.load_samples_from_db(conn)
rules = load_db.load_rules_from_db(conn, samples=samples)
else:
samples = load.load_samples()
rules = load.load_rules(samples=samples)
return samples, rules, load.rule_library_version(rules)
def run_revision(text: str, *, artifact: dict, patches: list, task_contract: dict,
fact_snapshot: dict | None = None, voice_ledger: dict | None = None,
pairwise: dict | None = None, rewrite_model: str = "claude",
from_db: bool = False) -> tuple[dict, dict]:
"""Patch 全链:应用 → 硬门 → 复扫 → 成对选择校验 → 审计报告。"""
if not artifact:
raise ReviseContractError("没有诊断产物,修订拒绝启动(专题-09 铁律)")
if not isinstance(patches, list) or not patches:
raise ReviseContractError("没有 patch 清单,修订无事可做")
samples, rules, lib_version = load_revision_library(from_db)
record = run_patch(text, rules, artifact, patches, task_contract,
fact_snapshot, voice_ledger, rewrite_model, pairwise, lib_version)
audit = report_mod.assemble(
artifact, patches, record["candidate_text"], record["hard_gate"],
record["voice_gate"], record["regression_gate"], record["pairwise_choice"],
unresolved_risks=(
record["hard_gate"]["unverified"]
+ list(record["voice_gate"].get("unknown", []))
),
)
return record, audit
def _run_id(*parts: str) -> str:
return "rev-" + hashlib.sha256("|".join(parts).encode("utf-8")).hexdigest()[:40]
def persist_revision(*, work_ref: str, text_hash: str, candidate_text: str | None,
audit: dict | None, conclusion: str, detail: dict,
creator: str = CREATOR, tenant_id: int = TENANT_ID) -> dict:
"""修订运行落库:候选稿哈希绑定评判,append-only 记账。"""
from db import connect
if not work_ref or not isinstance(detail, dict):
raise ReviseContractError("修订落库缺少 work_ref/detail")
if not isinstance(audit, dict):
raise ReviseContractError("修订落库缺少审计报告")
if conclusion not in {"passed", "blocked", "no_gain"}:
raise ReviseContractError(f"修订结论非法: {conclusion}")
audit_sha = hashlib.sha256(
json.dumps(audit, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
).hexdigest()
detail = {**detail, "audit_sha256": audit_sha}
candidate_sha = hashlib.sha256(candidate_text.encode("utf-8")).hexdigest() if candidate_text else None
run_id = _run_id(work_ref, text_hash, json.dumps(detail.get("patches", []), sort_keys=True))
run_sql = (
"INSERT INTO example_run (run_id, work_id, trigger_source, trigger_detail, "
"terminal_state, finished_at, creator, tenant_id) "
"VALUES (%s, NULL, 'user', %s::jsonb, 'completed', CURRENT_TIMESTAMP, %s, %s) "
"ON CONFLICT (run_id) DO UPDATE SET terminal_state='completed', "
"finished_at=CURRENT_TIMESTAMP, trigger_detail=EXCLUDED.trigger_detail, "
"updater=EXCLUDED.creator, update_time=CURRENT_TIMESTAMP"
)
quality_sql = (
"INSERT INTO example_quality_result "
"(run_id, candidate_sha256, judge_kind, dimension, scale_version, conclusion, detail, creator, tenant_id) "
"VALUES (%s, %s, 'review', 'ai_flavor_revision', %s, %s, %s::jsonb, %s, %s)"
)
with connect() as conn:
with conn.transaction():
conn.execute(run_sql, (run_id, json.dumps(detail, ensure_ascii=False), creator, tenant_id))
exists = conn.execute(
"SELECT 1 FROM example_quality_result WHERE tenant_id=%s AND run_id=%s "
"AND judge_kind='review' AND dimension='ai_flavor_revision' "
"AND COALESCE(candidate_sha256,'')=%s",
(tenant_id, run_id, candidate_sha or ""),
).fetchone()
if exists is None:
conn.execute(quality_sql, (
run_id, candidate_sha, detail.get("rule_library_version"),
conclusion, json.dumps(detail, ensure_ascii=False), creator, tenant_id,
))
return {"run_id": run_id, "candidate_sha256": candidate_sha}
def _persist_downgrade(*, work_ref: str, text_hash: str, reason: str) -> dict:
"""降级也是运行事实:落 example_run,避免「静默没发生」。"""
from db import connect
run_id = _run_id(work_ref, text_hash, "downgrade")
detail = {"status": "downgraded_to_audit", "reason": reason, "work_ref": work_ref}
with connect() as conn:
with conn.transaction():
conn.execute(
"INSERT INTO example_run (run_id, work_id, trigger_source, trigger_detail, "
"terminal_state, finished_at, creator, tenant_id) "
"VALUES (%s, NULL, 'user', %s::jsonb, 'completed', CURRENT_TIMESTAMP, %s, %s) "
"ON CONFLICT (run_id) DO NOTHING",
(run_id, json.dumps(detail, ensure_ascii=False), CREATOR, TENANT_ID),
)
return {"run_id": run_id}
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="技能 4 修订:最小 patch + 硬门 + 审计,作者确认前永远是候选")
parser.add_argument("--text-file", type=Path, required=True)
parser.add_argument("--artifact", type=Path, required=True, help="诊断产物 JSON(缺它拒绝启动)")
parser.add_argument("--patches", type=Path, required=True, help="patch 清单 JSON 数组(经仲裁)")
parser.add_argument("--task-contract", type=Path, required=True,
help="任务合同 JSON:mode=Patch 须带事实快照或显式授权")
parser.add_argument("--snapshot", type=Path, help="事实快照 JSON")
parser.add_argument("--voice-ledger", type=Path, help="声音账 JSON(技能 1 产物)")
parser.add_argument("--pairwise", type=Path, help="跨模型成对选择记录 JSON")
parser.add_argument("--rewrite-model", default="claude")
parser.add_argument("--work-ref", required=True)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--offline", action="store_true", help="只产文件,不写 muse-example")
args = parser.parse_args(argv)
try:
text = args.text_file.read_text(encoding="utf-8")
artifact = _load_json(args.artifact, "诊断产物")
patches_raw = json.loads(args.patches.read_text(encoding="utf-8"))
if not isinstance(patches_raw, list):
raise ReviseContractError("patch 清单必须是 JSON 数组")
task_contract = _load_json(args.task_contract, "任务合同")
snapshot = _load_json(args.snapshot, "事实快照") if args.snapshot else None
ledger = _load_json(args.voice_ledger, "声音账") if args.voice_ledger else None
if ledger is None and not args.offline:
ledger = load_current_baseline(args.work_ref)
pairwise = _load_json(args.pairwise, "成对选择记录") if args.pairwise else None
record, audit = run_revision(
text, artifact=artifact, patches=patches_raw, task_contract=task_contract,
fact_snapshot=snapshot, voice_ledger=ledger, pairwise=pairwise,
rewrite_model=args.rewrite_model, from_db=not args.offline,
)
except DowngradedToAudit as exc:
# 受控降级不是错误:Patch 降为 Audit,不产候选稿,降级事实照常落库
persistence = {"status": "offline"} if args.offline else _persist_downgrade(
work_ref=args.work_ref, text_hash=artifact.get("text_hash", ""), reason=exc.reason)
print(json.dumps({"status": "downgraded_to_audit", "reason": exc.reason,
"persistence": persistence}, ensure_ascii=False))
return 0
except (ReviseContractError, RevisionNotAuthorized, PatchError, PairwiseNotExecuted,
report_mod.ForbiddenScoreError, load.LoadError, ValueError, OSError) as exc:
print(f"REVISE_CONTRACT_FAILED: {exc}")
return 2
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(audit, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
if (not record["hard_gate"]["pass"] or not record["regression_gate"]["pass"]
or record["voice_gate"].get("pass") is not True):
# 声音账缺失/样本不足是 unverified,不得借 pairwise 结果伪装成 passed。
conclusion = "blocked"
elif record["pairwise_choice"]["choice"] in {"original", "tie", "both_bad"}:
conclusion = "no_gain"
else:
conclusion = "passed"
detail = {
"work_ref": args.work_ref,
"patches": [{"finding_id": p["finding_id"], "action": p["action"]} for p in patches_raw],
"hard_gate_pass": record["hard_gate"]["pass"],
"regression_pass": record["regression_gate"]["pass"],
"pairwise": bool(record["pairwise_choice"]),
"audit_ref": f"revise://ai-flavor/{args.output.name}",
}
if args.offline:
persistence = {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
else:
persistence = persist_revision(
work_ref=args.work_ref, text_hash=artifact.get("text_hash", ""),
candidate_text=record["candidate_text"], audit=audit,
conclusion=conclusion, detail=detail,
)
print(json.dumps({
"status": conclusion,
"hard_gate_pass": record["hard_gate"]["pass"],
"regression_pass": record["regression_gate"]["pass"],
"candidate_chars": len(record["candidate_text"]),
"output": str(args.output),
"persistence": persistence,
}, ensure_ascii=False))
return 0
if __name__ == "__main__":
raise SystemExit(main())