muse-agent-example/humanization/tests/test_seed_rules_db.py
zizi d80df3ed7c 治理: 剩余问题收口——humanization 规则/样例数据库权威 + PG 集成全通过 + 行为评测脚手架 + raw 冲突备忘录
范围(不含 design-story-foundation、docs/design、docs/write-chapter、
craft/、humanization/README.md 等进行中改动):

1. humanization 规则/样例运行时数据库权威
   - db/ddl/111:example_ai_flavor_rule / example_ai_flavor_sample /
     example_ai_flavor_rule_event(append-only 生命周期留痕),已应用到 muse-example
   - deai/load_db.py:数据库装载器,激活门/重复检测/指纹与文件装载器同源;
     数据库失败关闭,不静默回退 Git 文件资产
   - humanization/tools/seed_rules_db.py:YAML 种子单事务同步,幂等、
     变化留痕、--strict 对 db-only 行失败关闭;真实库已种入 26 规则/107 样例
   - prevent/diagnose/revise 生产路径切到数据库读取(--offline 显式读文件),
     落库前新鲜度检查与合同声明来源一致;四个 SKILL.md 数据库合同同步
   - 真实库验证:规则库指纹与文件种子一致(v-609bc40e21d0b5db),
     三个生产脚本端到端从库装载通过

2. PostgreSQL 集成:显式授权后 9/9 通过
   - 此前被依赖门阻断的 6 个 _db/smoke 测试全部通过
   - extract rollback 冒烟改为回滚事务内自给夹具(pending 窗/草稿缺失时自建),
     不再依赖瞬时生产状态;夹具残留核验为 0

3. Skill 行为评测脚手架(真实执行数量仍为 0)
   - harness/evals/skill_eval.py:场景合同、六类评测范畴、适配器和结构化裁决报告
   - diagnose-ai-flavor 参考场景 4 条 + 管道自测 7 项通过
   - 真实模型适配器未授权时以稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭;
     清单登记 skill_behavior_eval 条目,默认被依赖门阻断

4. evaluate-frozen-replay raw 存储边界冲突
   - docs/2026-08-19 备忘录:平台 DB-first 合同(创始人批准)与回放链
     仓外 vault 强制的冲突事实、两个选项和裁决前约束;运行时合同未单方面改写

5. harness 自身修复
   - runner 对账语义:行为评测入口不参与测试资产双向等值,但登记文件必须存在;
     manifest 保留 skill_behavior_eval 布尔字段并校验类型
   - 新增 2 条对账回归用例

验证证据: harness 三组自测 15+15+7 通过;静态审计 32 Skill / 0 问题;
76 个非数据库条目通过;9 个 PostgreSQL 集成条目显式授权后通过;
行为评测条目默认阻断;py_compile 与 git diff --check 通过。
未调用真实模型、embedding 或额度;真实行为评测执行数量仍为 0。
2026-08-19 02:41:25 +08:00

135 lines
5.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""humanization 种子同步工具离线测试:幂等、事件留痕、db-only 失败关闭、dry-run 不连库。"""
import contextlib
import io
import json
import pathlib
import sys
import unittest
PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[2]
sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "src"))
sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "tools"))
from deai import load, load_db # noqa: E402
import seed_rules_db as seed_tool # noqa: E402
class _Txn:
def __enter__(self):
return self
def __exit__(self, *exc):
return False
class _SeedConn:
"""记录写语句的假连接;现有行以 id->sha 注入。"""
def __init__(self, existing_rules=None, existing_samples=None):
self.existing_rules = dict(existing_rules or {})
self.existing_samples = dict(existing_samples or {})
self.writes = []
self.txn_count = 0
def transaction(self):
self.txn_count += 1
return _Txn()
def execute(self, sql, params=None):
normalized = " ".join(sql.split())
if normalized.startswith("SELECT rule_id, content_sha256 FROM example_ai_flavor_rule"):
return _Rows([(rid, sha) for rid, sha in sorted(self.existing_rules.items())])
if normalized.startswith("SELECT sample_id, content_sha256 FROM example_ai_flavor_sample"):
return _Rows([(sid, sha) for sid, sha in sorted(self.existing_samples.items())])
kind = (
"rule_insert" if normalized.startswith("INSERT INTO example_ai_flavor_rule (")
else "rule_update" if normalized.startswith("UPDATE example_ai_flavor_rule")
else "sample_insert" if normalized.startswith("INSERT INTO example_ai_flavor_sample")
else "sample_update" if normalized.startswith("UPDATE example_ai_flavor_sample")
else "rule_event" if normalized.startswith("INSERT INTO example_ai_flavor_rule_event")
else None
)
if kind is None:
raise AssertionError(f"未预期的 SQL: {normalized}")
self.writes.append((kind, params))
return _Rows([])
class _Rows:
def __init__(self, rows):
self.rows = list(rows)
def fetchall(self):
return list(self.rows)
def _corpus():
samples = load.load_samples()
rules = load.load_rules(samples=samples)
return samples, rules
class SeedRulesDbTest(unittest.TestCase):
def test_first_seed_inserts_all_and_records_events(self):
samples, rules = _corpus()
conn = _SeedConn()
summary = seed_tool.seed(conn, rules=rules, samples=samples)
self.assertEqual(summary["rules"]["inserted"], len(rules))
self.assertEqual(summary["samples"]["inserted"], len(samples))
self.assertEqual(summary["rules"]["updated"], 0)
self.assertEqual(summary["events"], len(rules))
kinds = [kind for kind, _ in conn.writes]
self.assertEqual(kinds.count("rule_insert"), len(rules))
self.assertEqual(kinds.count("sample_insert"), len(samples))
self.assertEqual(kinds.count("rule_event"), len(rules))
def test_second_seed_with_same_content_is_idempotent(self):
samples, rules = _corpus()
existing_rules = {rid: load_db.canonical_sha(rule) for rid, rule in rules.items()}
existing_samples = {sid: load_db.canonical_sha(s) for sid, s in samples.items()}
conn = _SeedConn(existing_rules=existing_rules, existing_samples=existing_samples)
summary = seed_tool.seed(conn, rules=rules, samples=samples)
self.assertEqual(summary["rules"]["unchanged"], len(rules))
self.assertEqual(summary["samples"]["unchanged"], len(samples))
self.assertEqual(conn.writes, [])
def test_changed_rule_is_updated_with_event(self):
samples, rules = _corpus()
changed = dict(rules["l001"], fix_hint="更新后的修复提示")
rules = dict(rules, l001=changed)
existing_rules = {rid: "0" * 64 for rid in rules}
existing_samples = {sid: load_db.canonical_sha(s) for sid, s in samples.items()}
conn = _SeedConn(existing_rules=existing_rules, existing_samples=existing_samples)
summary = seed_tool.seed(conn, rules=rules, samples=samples)
self.assertEqual(summary["rules"]["updated"], len(rules))
self.assertEqual(summary["samples"]["unchanged"], len(samples))
rule_writes = [kind for kind, _ in conn.writes if kind in ("rule_update", "rule_event")]
self.assertEqual(rule_writes.count("rule_update"), len(rules))
self.assertEqual(rule_writes.count("rule_event"), len(rules))
def test_db_only_rows_are_reported_and_strict_fails_closed(self):
samples, rules = _corpus()
conn = _SeedConn(existing_rules={"z999": "0" * 64})
summary = seed_tool.seed(conn, rules=rules, samples=samples)
self.assertEqual(summary["db_only_rules"], ["z999"])
conn_strict = _SeedConn(existing_rules={"z999": "0" * 64})
with self.assertRaisesRegex(seed_tool.SeedError, "失败关闭"):
seed_tool.seed(conn_strict, rules=rules, samples=samples, strict=True)
def test_dry_run_does_not_touch_database(self):
buffer = io.StringIO()
with contextlib.redirect_stdout(buffer):
code = seed_tool.main(["--dry-run"])
self.assertEqual(code, 0)
plan = json.loads(buffer.getvalue())
self.assertEqual(plan["status"], "dry_run")
samples, rules = _corpus()
self.assertEqual(plan["rules"], len(rules))
self.assertEqual(plan["samples"], len(samples))
self.assertEqual(plan["library_version"], load.rule_library_version(rules))
if __name__ == "__main__":
unittest.main()