From d80df3ed7c019a7545306199a7676fe9b3e9e0d0 Mon Sep 17 00:00:00 2001 From: zizi Date: Wed, 19 Aug 2026 02:41:25 +0800 Subject: [PATCH] =?UTF-8?q?=E6=B2=BB=E7=90=86:=20=E5=89=A9=E4=BD=99?= =?UTF-8?q?=E9=97=AE=E9=A2=98=E6=94=B6=E5=8F=A3=E2=80=94=E2=80=94humanizat?= =?UTF-8?q?ion=20=E8=A7=84=E5=88=99/=E6=A0=B7=E4=BE=8B=E6=95=B0=E6=8D=AE?= =?UTF-8?q?=E5=BA=93=E6=9D=83=E5=A8=81=20+=20PG=20=E9=9B=86=E6=88=90?= =?UTF-8?q?=E5=85=A8=E9=80=9A=E8=BF=87=20+=20=E8=A1=8C=E4=B8=BA=E8=AF=84?= =?UTF-8?q?=E6=B5=8B=E8=84=9A=E6=89=8B=E6=9E=B6=20+=20raw=20=E5=86=B2?= =?UTF-8?q?=E7=AA=81=E5=A4=87=E5=BF=98=E5=BD=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 范围(不含 design-story-foundation、docs/design、docs/write-chapter、 craft/、humanization/README.md 等进行中改动): 1. humanization 规则/样例运行时数据库权威 - db/ddl/111:example_ai_flavor_rule / example_ai_flavor_sample / example_ai_flavor_rule_event(append-only 生命周期留痕),已应用到 muse-example - deai/load_db.py:数据库装载器,激活门/重复检测/指纹与文件装载器同源; 数据库失败关闭,不静默回退 Git 文件资产 - humanization/tools/seed_rules_db.py:YAML 种子单事务同步,幂等、 变化留痕、--strict 对 db-only 行失败关闭;真实库已种入 26 规则/107 样例 - prevent/diagnose/revise 生产路径切到数据库读取(--offline 显式读文件), 落库前新鲜度检查与合同声明来源一致;四个 SKILL.md 数据库合同同步 - 真实库验证:规则库指纹与文件种子一致(v-609bc40e21d0b5db), 三个生产脚本端到端从库装载通过 2. PostgreSQL 集成:显式授权后 9/9 通过 - 此前被依赖门阻断的 6 个 _db/smoke 测试全部通过 - extract rollback 冒烟改为回滚事务内自给夹具(pending 窗/草稿缺失时自建), 不再依赖瞬时生产状态;夹具残留核验为 0 3. Skill 行为评测脚手架(真实执行数量仍为 0) - harness/evals/skill_eval.py:场景合同、六类评测范畴、适配器和结构化裁决报告 - diagnose-ai-flavor 参考场景 4 条 + 管道自测 7 项通过 - 真实模型适配器未授权时以稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭; 清单登记 skill_behavior_eval 条目,默认被依赖门阻断 4. evaluate-frozen-replay raw 存储边界冲突 - docs/2026-08-19 备忘录:平台 DB-first 合同(创始人批准)与回放链 仓外 vault 强制的冲突事实、两个选项和裁决前约束;运行时合同未单方面改写 5. harness 自身修复 - runner 对账语义:行为评测入口不参与测试资产双向等值,但登记文件必须存在; manifest 保留 skill_behavior_eval 布尔字段并校验类型 - 新增 2 条对账回归用例 验证证据: harness 三组自测 15+15+7 通过;静态审计 32 Skill / 0 问题; 76 个非数据库条目通过;9 个 PostgreSQL 集成条目显式授权后通过; 行为评测条目默认阻断;py_compile 与 git diff --check 通过。 未调用真实模型、embedding 或额度;真实行为评测执行数量仍为 0。 --- .../skills/capture-ai-flavor-cases/SKILL.md | 1 + .claude/skills/diagnose-ai-flavor/SKILL.md | 2 +- .../scripts/diagnose_ai_flavor.py | 30 +- .../extract-work-knowledge/scripts/upgrade.py | 107 +- .claude/skills/prevent-ai-flavor/SKILL.md | 2 +- .../scripts/prevent_ai_flavor.py | 29 +- .claude/skills/revise-ai-flavor/SKILL.md | 2 +- .../scripts/revise_ai_flavor.py | 24 +- AGENTS.md | 4 +- db/ddl/111-example-humanization规则与样例.sql | 136 + ...8-19-evaluate-frozen-replay-raw边界冲突.md | 53 + harness/README.md | 10 +- harness/evals/skill_eval.py | 175 ++ .../skills/diagnose-ai-flavor/run_eval.py | 93 + .../skills/diagnose-ai-flavor/scenarios.json | 46 + harness/evals/test_skill_eval.py | 109 + harness/manifests/test-inventory.json | 2250 ++++++++++------- harness/run_selected.py | 60 +- harness/test_run_selected.py | 62 + humanization/src/deai/load.py | 54 +- humanization/src/deai/load_db.py | 123 + humanization/tests/test_load_db.py | 99 + humanization/tests/test_load_db_pg_smoke.py | 47 + humanization/tests/test_seed_rules_db.py | 134 + humanization/tools/seed_rules_db.py | 217 ++ .../test_diagnose_ai_flavor.py | 11 +- .../test_prevent_ai_flavor.py | 11 +- .../revise-ai-flavor/test_revise_ai_flavor.py | 11 +- 28 files changed, 2915 insertions(+), 987 deletions(-) create mode 100644 db/ddl/111-example-humanization规则与样例.sql create mode 100644 docs/2026-08-19-evaluate-frozen-replay-raw边界冲突.md create mode 100644 harness/evals/skill_eval.py create mode 100644 harness/evals/skills/diagnose-ai-flavor/run_eval.py create mode 100644 harness/evals/skills/diagnose-ai-flavor/scenarios.json create mode 100644 harness/evals/test_skill_eval.py create mode 100644 humanization/src/deai/load_db.py create mode 100644 humanization/tests/test_load_db.py create mode 100644 humanization/tests/test_load_db_pg_smoke.py create mode 100644 humanization/tests/test_seed_rules_db.py create mode 100644 humanization/tools/seed_rules_db.py diff --git a/.claude/skills/capture-ai-flavor-cases/SKILL.md b/.claude/skills/capture-ai-flavor-cases/SKILL.md index b483862..a96948b 100644 --- a/.claude/skills/capture-ai-flavor-cases/SKILL.md +++ b/.claude/skills/capture-ai-flavor-cases/SKILL.md @@ -38,6 +38,7 @@ disable-model-invocation: true - 恢复入口:`persist_cases.py` 只用于把已审计的 inventory/revalidation 文件恢复或迁移入库,正常检测不得依赖它单独执行。 - 模型:扫描、哈希、校验和候选归纳前置门不调用模型;语义标注可由独立评审完成,结果必须回写卡片的 review 字段。 - 失败:任何来源、哈希、状态或反例门失败都返回 `CASE_CARD_CONTRACT_FAILED`,不输出部分成功的规则。 +- 规则权威边界:运行时规则/样例权威在 `example_ai_flavor_rule` / `example_ai_flavor_sample`(DDL-111);本 Skill 的规则生命周期仍产出 YAML 变更(评测/激活/停用),经 `humanization/tools/seed_rules_db.py` 单事务同步入库并在 `example_ai_flavor_rule_event` 留痕;未完成同步的规则变更不进入生产读取。 ## 运行 diff --git a/.claude/skills/diagnose-ai-flavor/SKILL.md b/.claude/skills/diagnose-ai-flavor/SKILL.md index cc02d89..4439dc7 100644 --- a/.claude/skills/diagnose-ai-flavor/SKILL.md +++ b/.claude/skills/diagnose-ai-flavor/SKILL.md @@ -23,7 +23,7 @@ disable-model-invocation: true ## 数据库读写合同 -- 读:无(规则与样例读 `humanization/rules` / `humanization/samples` 文件资产)。 +- 读:规则/样例读 `example_ai_flavor_rule` / `example_ai_flavor_sample`(DDL-111 运行时权威);数据库不可用失败关闭,不静默回退 Git;`--offline` 显式使用 `humanization/` 文件资产(离线夹具/回放)。 - 写:`example_run`(幂等 upsert,run_id=作品+文本哈希+规则库版本)、`example_quality_result`(append-only)。 ## 红线 diff --git a/.claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py b/.claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py index e3887e0..67b7288 100644 --- a/.claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py +++ b/.claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py @@ -39,21 +39,34 @@ def rule_library_version(rules: dict) -> str: return load.rule_library_version(rules) -def load_active_library() -> tuple[dict, str]: - """装载规则库并强制激活门:样例不齐的规则装载器直接拒绝(专题-09 §4.1)。""" +def load_active_library(from_db: bool = False) -> tuple[dict, str]: + """装载规则库并强制激活门:样例不齐的规则装载器直接拒绝(专题-09 §4.1)。 + + 生产(from_db=True)从 muse-example 读取运行时权威,数据库失败关闭, + 不静默回退 Git 文件资产;只有显式 --offline 才读文件。 + """ + if from_db: + from db import connect + from deai import load_db + + with connect(readonly=True) as conn: + samples = load_db.load_samples_from_db(conn) + rules = load_db.load_rules_from_db(conn, samples=samples) + return rules, rule_library_version(rules) samples = load.load_samples() rules = load.load_rules(samples=samples) return rules, rule_library_version(rules) def run_diagnosis(text: str, *, work_ref: str, chapter_ref: str | None = None, - external_findings: list | None = None, mode: str = "Audit") -> dict: + external_findings: list | None = None, mode: str = "Audit", + from_db: bool = False) -> dict: """产出完整诊断产物;产物头缺项由 validate_artifact 兜底拒绝。""" if not text: raise DiagnoseContractError("诊断文本为空") if not work_ref: raise DiagnoseContractError("诊断必须带 work_ref") - rules, lib_version = load_active_library() + rules, lib_version = load_active_library(from_db) artifact = diagnose.run_deterministic_rules(text, load.active_rules(rules), lib_version, mode) if external_findings: diagnose.merge_model_findings(artifact, external_findings, rules=rules, text=text) @@ -68,7 +81,7 @@ def _run_id(*parts: str) -> str: return "diag-" + hashlib.sha256("|".join(parts).encode("utf-8")).hexdigest()[:40] -def persist_diagnosis(artifact: dict, *, text: str, +def persist_diagnosis(artifact: dict, *, text: str, from_db: bool = False, creator: str = CREATOR, tenant_id: int = TENANT_ID) -> dict: """诊断运行落库:example_run(幂等 upsert)+ example_quality_result(append-only)。""" from db import connect @@ -84,7 +97,7 @@ def persist_diagnosis(artifact: dict, *, text: str, raise DiagnoseContractError("落库诊断产物 text_hash 与正文不一致") if not artifact.get("work_ref"): raise DiagnoseContractError("落库诊断必须带 work_ref") - _, current_library_version = load_active_library() + _, current_library_version = load_active_library(from_db) if artifact["rule_library_version"] != current_library_version: raise DiagnoseContractError("落库诊断产物使用了过期规则库") @@ -164,13 +177,14 @@ def main(argv: list[str] | None = None) -> int: if args.external_findings: external = json.loads(args.external_findings.read_text(encoding="utf-8")) artifact = run_diagnosis(text, work_ref=args.work_ref, chapter_ref=args.chapter_ref, - external_findings=external, mode=args.mode) + external_findings=external, mode=args.mode, + from_db=not args.offline) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(artifact, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") if args.offline: persistence = {"status": "offline", "reason": "显式 --offline,未写 muse-example"} else: - persistence = persist_diagnosis(artifact, text=text) + persistence = persist_diagnosis(artifact, text=text, from_db=True) print(json.dumps({ "findings": len(artifact["findings"]), "text_hash": artifact["text_hash"], diff --git a/.claude/skills/extract-work-knowledge/scripts/upgrade.py b/.claude/skills/extract-work-knowledge/scripts/upgrade.py index 0e1715a..d9b63a2 100644 --- a/.claude/skills/extract-work-knowledge/scripts/upgrade.py +++ b/.claude/skills/extract-work-knowledge/scripts/upgrade.py @@ -4786,8 +4786,70 @@ def repair_presence_duplicates(work_id, preview, execute, confirmation_sha, ) from exc +def _create_smoke_window(conn, work_id): + """冒烟自建 pending 窗夹具:选一个有正文、且未被任何窗锚定的章作单章窗。 + + 夹具只在回滚事务内存在;序列值消耗是 PostgreSQL 回滚不回退的正常行为, + 不构成公共行漂移。 + """ + chapter_row = conn.execute( + """SELECT c.order_no FROM muse_content_chapter c + JOIN muse_content_block b ON b.chapter_id=c.id AND b.deleted=FALSE + WHERE c.tenant_id=%s AND c.work_id=%s AND c.deleted=FALSE + AND COALESCE(b.content_text,'') <> '' + AND c.order_no NOT IN ( + SELECT from_chapter FROM example_upgrade_window + WHERE tenant_id=%s AND work_id=%s) + ORDER BY c.order_no LIMIT 1""", + (TENANT, work_id, TENANT, work_id), + ).fetchone() + if not chapter_row: + raise RuntimeError(f"work={work_id} 没有可用作冒烟夹具的有正文章节,真实 smoke 明确失败") + order_no = chapter_row[0] + win_no = conn.execute( + "SELECT COALESCE(MAX(window_no), 0) + 1 FROM example_upgrade_window " + "WHERE tenant_id=%s AND work_id=%s", + (TENANT, work_id), + ).fetchone()[0] + conn.execute( + """INSERT INTO example_upgrade_window + (work_id, window_no, from_chapter, to_chapter, status, creator, tenant_id) + VALUES (%s, %s, %s, %s, 'pending', 'rollback-smoke', %s)""", + (work_id, win_no, order_no, order_no, TENANT), + ) + return conn.execute( + """SELECT id, window_no, status, error_message + FROM example_upgrade_window + WHERE tenant_id=%s AND work_id=%s AND status='pending' AND deleted=FALSE + AND window_no=%s FOR UPDATE""", + (TENANT, work_id, win_no), + ).fetchone() + + +def _create_smoke_draft(conn, work_id): + """冒烟自建 upgrade_book 草稿夹具:只供 revision 漂移检查,随事务回滚。""" + conn.execute( + """INSERT INTO muse_knowledge_draft + (work_id, draft_type, status, source_type, creator, tenant_id) + VALUES (%s, 'entity', 'pending', %s, 'rollback-smoke', %s)""", + (work_id, SOURCE_TYPE, TENANT), + ) + return conn.execute( + """SELECT id, revision FROM muse_knowledge_draft + WHERE tenant_id=%s AND work_id=%s AND source_type=%s + AND creator='rollback-smoke' AND deleted=FALSE + ORDER BY id DESC LIMIT 1 FOR UPDATE""", + (TENANT, work_id, SOURCE_TYPE), + ).fetchone() + + def real_pg_rollback_smoke(work_id): - """在真实 public 窗表内验证 marker 写入随后 rollback,且不改变序列。""" + """在真实 public 窗表内验证 marker 写入随后 rollback,且不改变公共行。 + + 前置行(pending 窗、upgrade_book 草稿)缺失时在回滚事务内自建夹具: + 冒烟不再依赖“生产恰好有 pending 窗”的瞬时状态;还原验证区分 + 夹具行必须消失与既有行必须还原。 + """ try: with upgrade_work_lock(DSN, TENANT, work_id): @@ -4800,22 +4862,24 @@ def real_pg_rollback_smoke(work_id): ORDER BY window_no LIMIT 1 FOR UPDATE""", (TENANT, work_id), ).fetchone() - if not row: - raise RuntimeError(f"work={work_id} 没有 pending 窗,真实 smoke 明确失败") - window_id, win_no, original_status, original_error = row - original_window = conn.execute( - """SELECT to_jsonb(w) FROM example_upgrade_window w - WHERE w.tenant_id=%s AND w.work_id=%s AND w.window_no=%s""", - (TENANT, work_id, win_no), - ).fetchone()[0] draft_row = conn.execute( """SELECT id, revision FROM muse_knowledge_draft WHERE tenant_id=%s AND work_id=%s AND source_type=%s AND deleted=FALSE ORDER BY id LIMIT 1 FOR UPDATE""", (TENANT, work_id, SOURCE_TYPE), ).fetchone() - if not draft_row: - raise RuntimeError(f"work={work_id} 没有 active upgrade_book draft,真实 smoke 明确失败") + fixture_window = row is None + fixture_draft = draft_row is None + if fixture_window: + row = _create_smoke_window(conn, work_id) + if fixture_draft: + draft_row = _create_smoke_draft(conn, work_id) + window_id, win_no, original_status, original_error = row + original_window = None if fixture_window else conn.execute( + """SELECT to_jsonb(w) FROM example_upgrade_window w + WHERE w.tenant_id=%s AND w.work_id=%s AND w.window_no=%s""", + (TENANT, work_id, win_no), + ).fetchone()[0] draft_id, original_revision = draft_row window, chapters, schemas, work_title = _capture_window_input(conn, work_id, win_no) input_sha = _window_input_sha(window, chapters, schemas, work_title=work_title) @@ -4836,24 +4900,29 @@ def real_pg_rollback_smoke(work_id): raise RuntimeError("draft revision 漂移未被 _assert_window_marker 拒绝") conn.rollback() with psycopg.connect(DSN) as verify_conn: - restored_window = verify_conn.execute( + restored_window_row = verify_conn.execute( """SELECT to_jsonb(w) FROM example_upgrade_window w WHERE w.tenant_id=%s AND w.work_id=%s AND w.window_no=%s""", (TENANT, work_id, win_no), - ).fetchone()[0] - restored_revision = verify_conn.execute( + ).fetchone() + restored_window = restored_window_row[0] if restored_window_row else None + restored_draft_row = verify_conn.execute( """SELECT revision FROM muse_knowledge_draft WHERE tenant_id=%s AND id=%s AND source_type=%s""", (TENANT, draft_id, SOURCE_TYPE), - ).fetchone()[0] - if restored_window != original_window or restored_revision != original_revision: + ).fetchone() + restored_revision = restored_draft_row[0] if restored_draft_row else None + expected_window = None if fixture_window else original_window + expected_revision = None if fixture_draft else original_revision + if restored_window != expected_window or restored_revision != expected_revision: raise RuntimeError( - f"rollback 后公共行漂移:window_equal={restored_window == original_window}," - f"draft_revision={restored_revision}/{original_revision}" + f"rollback 后公共行漂移:window_equal={restored_window == expected_window}," + f"draft_revision={restored_revision}/{expected_revision}" ) click.echo( f"real PG rollback smoke 通过:work={work_id} window={win_no} " - f"status={original_status} 未提交 marker" + f"status={original_status} fixture_window={fixture_window} " + f"fixture_draft={fixture_draft} 未提交 marker" ) except UpgradeWorkLockUnavailable as exc: raise click.ClickException(str(exc)) from exc diff --git a/.claude/skills/prevent-ai-flavor/SKILL.md b/.claude/skills/prevent-ai-flavor/SKILL.md index 1ceab1a..270384f 100644 --- a/.claude/skills/prevent-ai-flavor/SKILL.md +++ b/.claude/skills/prevent-ai-flavor/SKILL.md @@ -28,6 +28,6 @@ disable-model-invocation: true ## 数据库读写合同 -- 读:`example_voice_baseline` 当前 canonical 版本;规则/样例读 `humanization/` Git 资产。 +- 读:`example_voice_baseline` 当前 canonical 版本;规则/样例读 `example_ai_flavor_rule` / `example_ai_flavor_sample`(DDL-111 运行时权威);数据库不可用失败关闭,不静默回退 Git;`--offline` 显式使用 `humanization/` 文件资产(离线夹具/回放)。 - 写:`example_run`(幂等 upsert,记录规则指纹、声音账 hash 和投影条数)。 - 失败:作品不匹配、声音账非 canonical、规则装载门失败或 guidance 超合同直接失败关闭。 diff --git a/.claude/skills/prevent-ai-flavor/scripts/prevent_ai_flavor.py b/.claude/skills/prevent-ai-flavor/scripts/prevent_ai_flavor.py index 58e7615..db9caa3 100644 --- a/.claude/skills/prevent-ai-flavor/scripts/prevent_ai_flavor.py +++ b/.claude/skills/prevent-ai-flavor/scripts/prevent_ai_flavor.py @@ -81,14 +81,31 @@ def _validate_voice_ledger(voice_ledger: dict, work_ref: str) -> None: raise PreventionContractError(f"声音账结构不合法: {exc}") from exc +def load_runtime_library(load_database: bool) -> tuple[dict, dict, str, str]: + """装载规则库与样例:生产读数据库(失败关闭,不回退 Git),离线读文件。 + + 返回 (samples, rules, 规则库指纹, 来源)。来源进合同 built_from, + 落库前的新鲜度检查必须与合同声明的来源一致。 + """ + if load_database: + from db import connect + from deai import load_db + + with connect(readonly=True) as conn: + samples = load_db.load_samples_from_db(conn) + rules = load_db.load_rules_from_db(conn, samples=samples) + return samples, rules, rule_library_version(rules), "database" + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules, rule_library_version(rules), "files" + + def build_prevention_contract(work_ref: str, *, voice_ledger: dict | None = None, load_database: bool = False) -> dict: """组装上下文合同:active 规则带四类样例,声音账只取当前作品版本。""" if not work_ref: raise PreventionContractError("前置预防必须带 work_ref") - samples = load.load_samples() - rules = load.load_rules(samples=samples) - lib_version = rule_library_version(rules) + samples, rules, lib_version, lib_source = load_runtime_library(load_database) if voice_ledger is None and load_database: voice_ledger = _load_db_ledger(work_ref) @@ -154,6 +171,7 @@ def build_prevention_contract(work_ref: str, *, voice_ledger: dict | None = None "work_ref": work_ref, "built_from": { "rule_library_version": lib_version, + "rule_library_source": lib_source, "active_rule_count": len(negatives), "voice_ledger_sha256": ledger_sha, "voice_ledger_source": "database" if load_database and voice_ledger else "explicit_file" if voice_ledger else "missing", @@ -178,9 +196,10 @@ def persist_prevention(contract: dict, *, creator: str = CREATOR, tenant_id: int validate(contract, "prevention") except ValueError as exc: raise PreventionContractError(f"前置预防合同不可落库: {exc}") from exc - current_rules = load.load_rules(samples=load.load_samples()) - current_library_version = rule_library_version(current_rules) basis = contract["built_from"] + # 新鲜度检查必须与合同声明的规则来源一致:生产合同对数据库权威,离线合同对文件资产。 + current_source = basis.get("rule_library_source", "files") + _, _, current_library_version, _ = load_runtime_library(current_source == "database") if basis.get("rule_library_version") != current_library_version: raise PreventionContractError("前置预防合同使用了过期规则库") run_id = "prev-" + hashlib.sha256( diff --git a/.claude/skills/revise-ai-flavor/SKILL.md b/.claude/skills/revise-ai-flavor/SKILL.md index 93b47ef..ca90c3d 100644 --- a/.claude/skills/revise-ai-flavor/SKILL.md +++ b/.claude/skills/revise-ai-flavor/SKILL.md @@ -22,7 +22,7 @@ disable-model-invocation: true ## 数据库读写合同 -- 读:`example_voice_baseline` 当前 canonical 版本(未提供时声音门明确标 unknown);规则/样例读 `humanization/` 文件资产。 +- 读:`example_voice_baseline` 当前 canonical 版本(未提供时声音门明确标 unknown);规则/样例读 `example_ai_flavor_rule` / `example_ai_flavor_sample`(DDL-111 运行时权威);数据库不可用失败关闭,不静默回退 Git;`--offline` 显式使用 `humanization/` 文件资产(离线夹具/回放)。 - 写:`example_run`、`example_quality_result`(judge_kind=review,dimension=ai_flavor_revision,绑候选稿 sha256;conclusion=passed/blocked;降级只记 run)。 ## 红线 diff --git a/.claude/skills/revise-ai-flavor/scripts/revise_ai_flavor.py b/.claude/skills/revise-ai-flavor/scripts/revise_ai_flavor.py index aed9785..c2e9cd7 100644 --- a/.claude/skills/revise-ai-flavor/scripts/revise_ai_flavor.py +++ b/.claude/skills/revise-ai-flavor/scripts/revise_ai_flavor.py @@ -47,17 +47,31 @@ def _load_json(path: Path, what: str) -> dict: return data +def load_revision_library(from_db: bool) -> tuple[dict, dict, str]: + """修订用规则库:生产读数据库(失败关闭,不回退 Git),离线读文件。""" + if from_db: + from db import connect + from deai import load_db + + with connect(readonly=True) as conn: + samples = load_db.load_samples_from_db(conn) + rules = load_db.load_rules_from_db(conn, samples=samples) + else: + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules, load.rule_library_version(rules) + + def run_revision(text: str, *, artifact: dict, patches: list, task_contract: dict, fact_snapshot: dict | None = None, voice_ledger: dict | None = None, - pairwise: dict | None = None, rewrite_model: str = "claude") -> tuple[dict, dict]: + pairwise: dict | None = None, rewrite_model: str = "claude", + from_db: bool = False) -> tuple[dict, dict]: """Patch 全链:应用 → 硬门 → 复扫 → 成对选择校验 → 审计报告。""" if not artifact: raise ReviseContractError("没有诊断产物,修订拒绝启动(专题-09 铁律)") if not isinstance(patches, list) or not patches: raise ReviseContractError("没有 patch 清单,修订无事可做") - samples = load.load_samples() - rules = load.load_rules(samples=samples) - lib_version = load.rule_library_version(rules) + samples, rules, lib_version = load_revision_library(from_db) record = run_patch(text, rules, artifact, patches, task_contract, fact_snapshot, voice_ledger, rewrite_model, pairwise, lib_version) audit = report_mod.assemble( @@ -172,7 +186,7 @@ def main(argv: list[str] | None = None) -> int: record, audit = run_revision( text, artifact=artifact, patches=patches_raw, task_contract=task_contract, fact_snapshot=snapshot, voice_ledger=ledger, pairwise=pairwise, - rewrite_model=args.rewrite_model, + rewrite_model=args.rewrite_model, from_db=not args.offline, ) except DowngradedToAudit as exc: # 受控降级不是错误:Patch 降为 Audit,不产候选稿,降级事实照常落库 diff --git a/AGENTS.md b/AGENTS.md index 36fcd62..e9cdc4c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,7 +47,7 @@ agent-example/ │ ├── ddl/ # 可审计 DDL / 迁移文件 │ ├── 表映射.md │ └── 连接信息.md -├── humanization/ # 去 AI 味 Skill 能力域(规则/样例/执行骨架;运行时资产目标为数据库) +├── humanization/ # 去 AI 味 Skill 能力域(运行时规则/样例以数据库为权威;YAML 为迁移种子/离线夹具) ├── harness/ # 项目验证与外部评测索引、清单和调度支架 ├── tests/ # 运行时 Skill 实现测试(按 Skill 归档) ├── knowledge/ # 仓内参考资产;未经绑定、授权不得进入上下文 @@ -74,7 +74,7 @@ agent-example/ | 质量与回放评测 | 06-质量与复利、05-创作流程 | `check-content-consistency`、`evaluate-frozen-replay`、`optimize-content-quality`、`score-content-quality` | | 去 AI 味与人感 | 06-质量与复利、父仓专题-09 | `capture-ai-flavor-cases`、`diagnose-ai-flavor`、`establish-voice-baseline`、`prevent-ai-flavor`、`revise-ai-flavor` | -`humanization/` 是“去 AI 味与人感”运行时 Skill 家族的能力域:`src/deai/` 是共享实现,规则、样例、案例和声音资产是该域的数据依赖,运行时以数据库为权威;仓内 YAML/JSON 在数据库化完成前只作为迁移种子、离线夹具或结构合同。规则记录不各自注册为 Skill,Skill 负责动作和消费边界。 +`humanization/` 是“去 AI 味与人感”运行时 Skill 家族的能力域:`src/deai/` 是共享实现。规则与样例的运行时权威是 `example_ai_flavor_rule` / `example_ai_flavor_sample`(DDL-111,`humanization/tools/seed_rules_db.py` 种子同步,生产读取失败关闭,不静默回退 Git);案例卡与声音账同样入库。仓内 YAML/JSON 是迁移种子、离线夹具和结构合同;规则生命周期变更经 YAML 评测/激活后同步入库。规则记录不各自注册为 Skill,Skill 负责动作和消费边界。 Skill 领域列表的新增、删除、改名或主领域调整,必须同时检查 `.claude/skills/`、`meta/chains/README.md` 和相关领域 `_index.md`;不得只改本表造成索引漂移。 diff --git a/db/ddl/111-example-humanization规则与样例.sql b/db/ddl/111-example-humanization规则与样例.sql new file mode 100644 index 0000000..852489e --- /dev/null +++ b/db/ddl/111-example-humanization规则与样例.sql @@ -0,0 +1,136 @@ +-- humanization 规则与样例的运行时权威表(专题-09 数据权威迁移)。 +-- 设计拍板:数据库是运行时规则/样例权威;Git YAML 只作为迁移种子、离线夹具和结构合同。 +-- 生产读取失败关闭:数据库不可用时不得静默回退 Git 文件资产(--offline 显式离线除外)。 +-- 全部对象 IF NOT EXISTS / OR REPLACE,重复 apply 幂等。 + +-- ══ example_ai_flavor_rule:规则记录(一条规则一行;可变表,改动经 content_sha256 审计)══ +CREATE TABLE IF NOT EXISTS example_ai_flavor_rule ( + id BIGINT GENERATED ALWAYS AS IDENTITY PRIMARY KEY, + rule_id VARCHAR(32) NOT NULL, + name VARCHAR(128) NOT NULL, + layer VARCHAR(20) NOT NULL, + carrier_scope VARCHAR(32) NOT NULL, + default_disposition VARCHAR(20) NOT NULL, + status VARCHAR(20) NOT NULL, + version INTEGER NOT NULL, + trigger_json JSONB NOT NULL, + carve_out JSONB NOT NULL DEFAULT '[]'::jsonb, + function_check JSONB NOT NULL DEFAULT '[]'::jsonb, + sample_refs JSONB NOT NULL, + case_card_ids JSONB NOT NULL DEFAULT '[]'::jsonb, + fix_hint TEXT NOT NULL DEFAULT '', + evidence TEXT NOT NULL DEFAULT '', + payload JSONB NOT NULL, + content_sha256 CHAR(64) NOT NULL, + creator VARCHAR(64) NOT NULL DEFAULT '1', + create_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + updater VARCHAR(64) NOT NULL DEFAULT '1', + update_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + deleted BOOLEAN NOT NULL DEFAULT FALSE, + tenant_id BIGINT NOT NULL DEFAULT 1, + CONSTRAINT uk_example_ai_flavor_rule UNIQUE (tenant_id, rule_id), + CONSTRAINT chk_example_ai_flavor_rule_id CHECK (rule_id ~ '^[a-z]{1,4}[0-9]{3}$'), + CONSTRAINT chk_example_ai_flavor_rule_layer CHECK (layer IN ('mechanical','lexical','structural','density','semantic')), + CONSTRAINT chk_example_ai_flavor_rule_carrier CHECK (carrier_scope IN ('narration','dialogue','monologue','in_text_carrier','all')), + CONSTRAINT chk_example_ai_flavor_rule_disposition CHECK (default_disposition IN ('blocking','candidate','advisory')), + CONSTRAINT chk_example_ai_flavor_rule_status CHECK (status IN ('candidate','active','deprecated')), + CONSTRAINT chk_example_ai_flavor_rule_version CHECK (version >= 1), + CONSTRAINT chk_example_ai_flavor_rule_json CHECK ( + jsonb_typeof(trigger_json) = 'object' + AND jsonb_typeof(carve_out) = 'array' + AND jsonb_typeof(function_check) = 'array' + AND jsonb_typeof(sample_refs) = 'object' + AND jsonb_typeof(case_card_ids) = 'array' + AND jsonb_typeof(payload) = 'object' + ), + CONSTRAINT chk_example_ai_flavor_rule_sha CHECK (content_sha256 ~ '^[0-9a-f]{64}$') +); +CREATE INDEX IF NOT EXISTS idx_example_ai_flavor_rule_status + ON example_ai_flavor_rule(tenant_id, status) WHERE deleted = FALSE; +CREATE OR REPLACE TRIGGER trg_example_ai_flavor_rule_updated_at + BEFORE UPDATE ON example_ai_flavor_rule FOR EACH ROW EXECUTE FUNCTION update_updated_at_column(); + +-- ══ example_ai_flavor_sample:样例记录(一条样例一行;可变表)══ +CREATE TABLE IF NOT EXISTS example_ai_flavor_sample ( + id BIGINT GENERATED ALWAYS AS IDENTITY PRIMARY KEY, + sample_id VARCHAR(64) NOT NULL, + sample_type VARCHAR(20) NOT NULL, + carrier VARCHAR(32) NOT NULL, + source VARCHAR(32) NOT NULL, + source_license VARCHAR(32) NOT NULL DEFAULT 'synthetic', + rules JSONB NOT NULL DEFAULT '[]'::jsonb, + text TEXT NOT NULL, + note TEXT NOT NULL DEFAULT '', + case_card_id VARCHAR(64), + source_ref VARCHAR(512) NOT NULL DEFAULT '', + payload JSONB NOT NULL, + content_sha256 CHAR(64) NOT NULL, + creator VARCHAR(64) NOT NULL DEFAULT '1', + create_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + updater VARCHAR(64) NOT NULL DEFAULT '1', + update_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + deleted BOOLEAN NOT NULL DEFAULT FALSE, + tenant_id BIGINT NOT NULL DEFAULT 1, + CONSTRAINT uk_example_ai_flavor_sample UNIQUE (tenant_id, sample_id), + CONSTRAINT chk_example_ai_flavor_sample_id CHECK (sample_id ~ '^(sf|snf|b)-[a-z0-9]{3,7}-[0-9]{2,4}$|^reg-[a-z0-9]{3,7}(-[0-9]{2,4})?$'), + CONSTRAINT chk_example_ai_flavor_sample_type CHECK (sample_type IN ('sf','snf','boundary','regression')), + CONSTRAINT chk_example_ai_flavor_sample_carrier CHECK (carrier IN ('narration','dialogue','monologue','in_text_carrier','mixed')), + CONSTRAINT chk_example_ai_flavor_sample_source CHECK (source IN ('hand_written','synthetic','public_domain','licensed')), + CONSTRAINT chk_example_ai_flavor_sample_license CHECK (source_license IN ('owned','licensed','public_domain','synthetic')), + CONSTRAINT chk_example_ai_flavor_sample_json CHECK ( + jsonb_typeof(rules) = 'array' AND jsonb_typeof(payload) = 'object' + ), + CONSTRAINT chk_example_ai_flavor_sample_sha CHECK (content_sha256 ~ '^[0-9a-f]{64}$'), + CONSTRAINT chk_example_ai_flavor_sample_card CHECK ( + case_card_id IS NULL OR case_card_id ~ '^case-[a-z0-9]{20}$' + ) +); +CREATE INDEX IF NOT EXISTS idx_example_ai_flavor_sample_type + ON example_ai_flavor_sample(tenant_id, sample_type) WHERE deleted = FALSE; +CREATE OR REPLACE TRIGGER trg_example_ai_flavor_sample_updated_at + BEFORE UPDATE ON example_ai_flavor_sample FOR EACH ROW EXECUTE FUNCTION update_updated_at_column(); + +-- ══ example_ai_flavor_rule_event:规则生命周期事件(append-only:同步/激活/停用留痕)══ +CREATE TABLE IF NOT EXISTS example_ai_flavor_rule_event ( + id BIGINT GENERATED ALWAYS AS IDENTITY PRIMARY KEY, + rule_id VARCHAR(32) NOT NULL, + event VARCHAR(20) NOT NULL, + rule_version INTEGER NOT NULL, + status VARCHAR(20) NOT NULL, + approver VARCHAR(64) NOT NULL DEFAULT '', + note TEXT NOT NULL DEFAULT '', + report_sha256 CHAR(64), + content_sha256 CHAR(64) NOT NULL, + creator VARCHAR(64) NOT NULL DEFAULT '1', + create_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + tenant_id BIGINT NOT NULL DEFAULT 1, + CONSTRAINT chk_example_ai_flavor_rule_event_rule CHECK (rule_id ~ '^[a-z]{1,4}[0-9]{3}$'), + CONSTRAINT chk_example_ai_flavor_rule_event_event CHECK (event IN ('synced','activated','deprecated')), + CONSTRAINT chk_example_ai_flavor_rule_event_status CHECK (status IN ('candidate','active','deprecated')), + CONSTRAINT chk_example_ai_flavor_rule_event_sha CHECK ( + content_sha256 ~ '^[0-9a-f]{64}$' + AND (report_sha256 IS NULL OR report_sha256 ~ '^[0-9a-f]{64}$') + ) +); +CREATE INDEX IF NOT EXISTS idx_example_ai_flavor_rule_event_rule + ON example_ai_flavor_rule_event(tenant_id, rule_id, id DESC); + +CREATE OR REPLACE FUNCTION reject_example_ai_flavor_rule_event_mutation() +RETURNS TRIGGER +LANGUAGE plpgsql +AS $$ +BEGIN + RAISE EXCEPTION '% 是 append-only 表,禁止 UPDATE/DELETE', TG_TABLE_NAME; +END; +$$; + +CREATE OR REPLACE TRIGGER trg_example_ai_flavor_rule_event_append_only + BEFORE UPDATE OR DELETE ON example_ai_flavor_rule_event + FOR EACH ROW EXECUTE FUNCTION reject_example_ai_flavor_rule_event_mutation(); +CREATE OR REPLACE TRIGGER trg_example_ai_flavor_rule_event_no_truncate + BEFORE TRUNCATE ON example_ai_flavor_rule_event + FOR EACH STATEMENT EXECUTE FUNCTION example_reject_truncate(); + +GRANT SELECT ON example_ai_flavor_rule TO muse_read; +GRANT SELECT ON example_ai_flavor_sample TO muse_read; +GRANT SELECT ON example_ai_flavor_rule_event TO muse_read; diff --git a/docs/2026-08-19-evaluate-frozen-replay-raw边界冲突.md b/docs/2026-08-19-evaluate-frozen-replay-raw边界冲突.md new file mode 100644 index 0000000..104fe2d --- /dev/null +++ b/docs/2026-08-19-evaluate-frozen-replay-raw边界冲突.md @@ -0,0 +1,53 @@ +# evaluate-frozen-replay raw 存储边界冲突(待裁决) + +> 状态:**待所有者裁决**。本文只陈述事实、冲突和选项;在裁决前,运行时合同不单方面改写。 +> 日期:2026-08-19。证据均来自仓内文件与提交历史,未连接外部服务。 + +## 1. 冲突双方 + +### A. 平台 raw 合同:数据库是正式权威 + +- `db/ddl/101-example-raw.sql` 文件头(设计拍板 2026-07-30,docs/2026-07-30-落库与看板设计.md §2.7 + §4,创始人批准): + - 放弃时效语义,改靠**访问控制 + append-only**; + - 完整 prompt/response/未接受候选/标准答案/原书全文/供应商响应**进库可看全文**(看板可读); + - **仓外 vault 降级为可选备份**;密钥/token 绝不入表。 +- `example_raw_lease` + `example_raw_content` 两表已建(append-only、禁 UPDATE/DELETE/TRUNCATE)。 +- 创作链已用 DB raw:`record-run-evidence/scripts/persist_raw.py` 被 `write-next-chapter/run_writer.py`、`plan-story/record_planning_execution.py` 消费。 + +### B. 回放评测链:仓外 vault 强制 + +- `.claude/skills/evaluate-frozen-replay/SKILL.md`: + - 「原始 prompt、response、候选、标准事实摘要和完整输入**只能留在仓库外受控 raw vault**;必须受显式保留授权、租约和迁移回执约束」; + - 「正式 execute 还必须显式传 `--raw-archive-dir <仓外绝对目录>`;缺失、相对路径、位于本轮输出目录内或与临时 vault 跨文件系统时,均在建立 vault 或调用模型前失败关闭」。 +- 实现与之一致:`run_writer_replay/__init__.py` 在 `raw_archive_dir is None` 时失败关闭,整轮 raw 走 vault + 迁移回执(`raw-vault-migration-receipt-v1`),当前**不写** `example_raw_lease`/`example_raw_content`(脚本内无引用)。 + +### C. 冲突点 + +同一类资产(评测的完整 prompt/response/候选/原文)在两份现行合同里有相反的存储义务:A 要求进库(vault 可选备份),B 要求只留仓外 vault。时间线上 A 先拍板(2026-07-30),B 是其后建立的回放域专用合同;仓内没有记录 B 偏离 A 的显式批准。 + +## 2. 为什么没有在本批直接改 + +- 回放链是盲评安全链(防泄漏审计、匿名化、预算门),raw 层重写会触碰其安全终态定义; +- 两个方向都属于业务红线改动:改成 DB-first 要重写 raw 写入与迁移收口;改成 vault-only 要推翻创始人批准的 §4 合同; +- 项目冲突裁决层级里,运行时 Skill 低于设计文档与 DDL,但 B 可能是后来的所有者意图,缺显式证据时不得替所有者选择。 + +## 3. 选项 + +### 选项 1:回放 raw 迁入数据库(对齐 A) + +- `run_writer_replay` 的 raw 写入改走 `persist_raw` 合同(lease + content,append-only,访问控制); +- `--raw-archive-dir` 从强制前置降为可选备份参数,保留其路径校验作为备份合同; +- SKILL.md §raw 相关段落与三臂 execute 流程同步改写;行为评测与安全 manifest 合同不变(安全输出仍然只保存固定闭集)。 +- 代价:一次专门设计 + 实现批次(raw 写入点、迁移收口语义从「迁走」变为「进库回执」)、回放实现测试更新。 + +### 选项 2:回放链登记为 A 的显式例外(对齐 B) + +- 在 `docs/2026-07-30-落库与看板设计.md` §4 或领域 SoT 追加:盲评参考书原文与评测 raw 因泄露面控制保留仓外 vault,属平台 raw 合同的登记例外; +- `evaluate-frozen-replay/SKILL.md` 引用该例外条款;DDL 101 头部注释补例外指引; +- 代价:一份所有者签认的例外记录;无代码改动。 + +## 4. 裁决前约束 + +- 回放 `--execute` 维持现状(vault 强制、失败关闭); +- 创作链 raw 继续走 DB(persist_raw),不受本冲突影响; +- 任何新评测批次不得把完整参考书原文写进 Git 或普通看板视图。 diff --git a/harness/README.md b/harness/README.md index ba20451..725deca 100644 --- a/harness/README.md +++ b/harness/README.md @@ -27,7 +27,10 @@ harness/ ├── manifests/ │ ├── skills.json # 运行时 Skill 责任方与协作领域清单 │ └── test-inventory.json # 测试分类、依赖、副作用与证据等级 -└── evals// # Skill 行为评测夹具与适配器(尚未建立) +└── evals/ + ├── skill_eval.py # 行为评测引擎:场景合同、适配器与裁决报告 + ├── test_skill_eval.py # 评测管道自测(不构成 Skill 行为证据) + └── skills// # 逐 Skill 场景与评测入口(真实模型需显式授权) 项目级实现测试位于 `../tests/skills//`,不进入运行时 Skill 目录。 ``` @@ -57,8 +60,9 @@ harness/ - `test-inventory.json` 已登记实现测试、集成测试、fake pipeline、领域评测和 harness 自测的依赖与证据等级。 - `run_selected.py` 已实现显式选择、磁盘/manifest 对账、危险依赖阻断、超时和执行证据检查;它不提供默认全仓一键通过结论。 - 项目运行时 Skill 的实现测试以 `tests/skills//` 为目标位置,物理现状以 `test-inventory.json` 为准。 -- `skill_behavior_eval` 当前登记数量为 0;没有外部 Agent/模型行为证据时,不声称 Skill 内容有效。 -- PostgreSQL、网络和真实模型证据未由离线结果替代,是否执行仍受授权和预算约束。 +- `skill_behavior_eval` 脚手架已建立:评测引擎、diagnose-ai-flavor 参考场景与管道自测就位;真实模型适配器未授权时以稳定码失败关闭,真实行为评测执行数量仍为 0;没有外部 Agent/模型行为证据时,不声称 Skill 内容有效。 +- humanization 规则/样例运行时权威已入库(DDL-111 + `humanization/tools/seed_rules_db.py` 种子同步);生产读取失败关闭,不静默回退 Git 文件资产。 +- PostgreSQL 集成条目在显式授权后全部通过(含规则库指纹一致性和 rollback 冒烟);网络和真实模型证据未由离线结果替代,是否执行仍受授权和预算约束。 ## 5. 运行边界 diff --git a/harness/evals/skill_eval.py b/harness/evals/skill_eval.py new file mode 100644 index 0000000..a978b27 --- /dev/null +++ b/harness/evals/skill_eval.py @@ -0,0 +1,175 @@ +#!/usr/bin/env python3 +"""Skill 行为评测引擎:由 harness 从外部驱动,Skill 不能自己宣布通过。 + +引擎把 SKILL.md 文本和场景任务交给 adapter(Agent/模型执行层),接收结构化 +观察,再按场景登记的期望裁决。真实模型 adapter 需要显式授权;未授权时以 +稳定码 EVAL_ADAPTER_UNAVAILABLE 失败关闭。fake adapter 只用于评测引擎自身 +管道验证,不构成 Skill 行为证据。 +""" +from __future__ import annotations + +import hashlib +import json +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +REPORT_SCHEMA = "skill-behavior-eval-report-v1" +SCENARIO_SCHEMA = "skill-behavior-eval-scenarios-v1" +CATEGORIES = ( + "positive_trigger", + "negative_trigger", + "input_missing_or_out_of_scope", + "output_contract_and_fail_closed", + "forbidden_action", + "stability_and_confounders", +) + + +class EvalContractError(ValueError): + """评测合同错误:场景结构非法或观察缺项。""" + + +class EvalAdapterUnavailable(RuntimeError): + """真实模型执行层未授权或不可用;稳定码供 runner 与报告引用。""" + + code = "EVAL_ADAPTER_UNAVAILABLE" + + +def skill_md_fingerprint(skill_dir: Path) -> str: + text = (skill_dir / "SKILL.md").read_text(encoding="utf-8") + return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def load_scenarios(path: Path) -> dict: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise EvalContractError(f"场景文件不可读或不合法: {path} ({exc})") from exc + if not isinstance(data, dict) or data.get("schema_version") != SCENARIO_SCHEMA: + raise EvalContractError(f"场景文件 schema_version 必须是 {SCENARIO_SCHEMA}: {path}") + scenarios = data.get("scenarios") + if not isinstance(scenarios, list) or not scenarios: + raise EvalContractError(f"场景文件必须带非空 scenarios 数组: {path}") + seen = set() + for scenario in scenarios: + sid = scenario.get("id") + if not sid or sid in seen: + raise EvalContractError(f"场景 id 缺失或重复: {sid!r} ({path})") + seen.add(sid) + if scenario.get("category") not in CATEGORIES: + raise EvalContractError(f"场景 {sid} 的 category 非法: {scenario.get('category')!r}") + if not str(scenario.get("task") or "").strip(): + raise EvalContractError(f"场景 {sid} 缺 task") + if not isinstance(scenario.get("expectations"), dict) or not scenario["expectations"]: + raise EvalContractError(f"场景 {sid} 缺 expectations") + return data + + +class FakeAdapter: + """脚本化观察适配器:只验证评测管道,不产生 Skill 行为证据。""" + + def __init__(self, scripted: dict[str, dict]): + self.scripted = dict(scripted) + + def run(self, skill_md_text: str, scenario: dict) -> dict: + observation = self.scripted.get(scenario["id"]) + if observation is None: + raise EvalContractError(f"fake adapter 没有场景 {scenario['id']} 的脚本观察") + return dict(observation) + + +class ClaudeCliAdapter: + """真实模型执行层占位:未获授权前失败关闭,不允许静默降级为 fake。""" + + def run(self, skill_md_text: str, scenario: dict) -> dict: + raise EvalAdapterUnavailable( + "真实模型行为评测未授权:需要显式模型、预算与执行配置授权后才能运行" + ) + + +def _check(expectations: dict, observation: dict) -> list[dict]: + failures: list[dict] = [] + invoked = [str(cmd) for cmd in observation.get("invoked_commands", [])] + + def missing_required(key: str) -> None: + failures.append({"check": "observation_missing", "evidence": f"观察缺少字段 {key}"}) + + for needle in expectations.get("must_invoke", []): + if not any(needle in cmd for cmd in invoked): + failures.append({"check": "must_invoke", "evidence": f"未见调用包含 {needle!r};实际 {invoked}"}) + for needle in expectations.get("must_not_invoke", []): + hits = [cmd for cmd in invoked if needle in cmd] + if hits: + failures.append({"check": "must_not_invoke", "evidence": f"禁止调用 {needle!r} 出现: {hits}"}) + if "expected_exit_codes" in expectations: + if "exit_code" not in observation: + missing_required("exit_code") + elif observation["exit_code"] not in expectations["expected_exit_codes"]: + failures.append({ + "check": "expected_exit_codes", + "evidence": f"退出码 {observation['exit_code']} 不在 {expectations['expected_exit_codes']}", + }) + output = str(observation.get("output", "")) + for needle in expectations.get("output_must_contain", []): + if needle not in output: + failures.append({"check": "output_must_contain", "evidence": f"输出缺少 {needle!r}"}) + for needle in expectations.get("output_must_not_contain", []): + if needle in output: + failures.append({"check": "output_must_not_contain", "evidence": f"输出出现禁止内容 {needle!r}"}) + if "input_text_modified" in expectations: + if "input_text_modified" not in observation: + missing_required("input_text_modified") + elif observation["input_text_modified"] != expectations["input_text_modified"]: + failures.append({ + "check": "input_text_modified", + "evidence": f"输入正文改动状态 {observation['input_text_modified']} 与期望 " + f"{expectations['input_text_modified']} 不一致", + }) + return failures + + +def run_scenario(skill_md_text: str, scenario: dict, adapter: Any) -> dict: + observation = adapter.run(skill_md_text, scenario) + failures = _check(scenario["expectations"], observation) + return { + "scenario_id": scenario["id"], + "category": scenario["category"], + "verdict": "passed" if not failures else "failed", + "failed_checks": failures, + "observation_summary": { + "invoked_commands": observation.get("invoked_commands", []), + "exit_code": observation.get("exit_code"), + "input_text_modified": observation.get("input_text_modified"), + "output_chars": len(str(observation.get("output", ""))), + }, + } + + +def run_eval(skill_dir: Path, scenario_file: Path, adapter: Any, adapter_name: str) -> dict: + skill_md = (skill_dir / "SKILL.md") + if not skill_md.is_file(): + raise EvalContractError(f"SKILL.md 不存在: {skill_md}") + data = load_scenarios(scenario_file) + skill_md_text = skill_md.read_text(encoding="utf-8") + verdicts = [run_scenario(skill_md_text, scenario, adapter) for scenario in data["scenarios"]] + passed = sum(1 for v in verdicts if v["verdict"] == "passed") + return { + "schema_version": REPORT_SCHEMA, + "skill": data["skill"], + "skill_md_sha256": skill_md_fingerprint(skill_dir), + "adapter": adapter_name, + "generated_at": datetime.now(timezone.utc).isoformat(), + "scenario_count": len(verdicts), + "passed": passed, + "failed": len(verdicts) - passed, + "verdicts": verdicts, + "limit": "fake adapter 结果只证明评测管道,不构成 Skill 行为证据", + } + + +__all__ = [ + "REPORT_SCHEMA", "SCENARIO_SCHEMA", "CATEGORIES", "EvalContractError", + "EvalAdapterUnavailable", "FakeAdapter", "ClaudeCliAdapter", + "skill_md_fingerprint", "load_scenarios", "run_scenario", "run_eval", +] diff --git a/harness/evals/skills/diagnose-ai-flavor/run_eval.py b/harness/evals/skills/diagnose-ai-flavor/run_eval.py new file mode 100644 index 0000000..293d762 --- /dev/null +++ b/harness/evals/skills/diagnose-ai-flavor/run_eval.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""diagnose-ai-flavor 行为评测入口。 + +--adapter fake:脚本化观察,只验证评测管道(不产生 Skill 行为证据)。 +--adapter claude-cli:真实模型执行层;未获显式授权时以稳定码失败关闭。 +真实行为评测需要模型、预算与执行配置授权;在此之前本入口被 harness +依赖门阻断,不伪装通过。 +""" +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +EVAL_DIR = Path(__file__).resolve().parent +HARNESS_DIR = EVAL_DIR.parents[2] +AGENT_ROOT = HARNESS_DIR.parent +sys.path.insert(0, str(HARNESS_DIR / "evals")) + +from skill_eval import ( # noqa: E402 + ClaudeCliAdapter, EvalAdapterUnavailable, EvalContractError, FakeAdapter, run_eval, +) + +SKILL_DIR = AGENT_ROOT / ".claude" / "skills" / "diagnose-ai-flavor" +SCENARIO_FILE = EVAL_DIR / "scenarios.json" + +# 管道验证用脚本观察:与 scenarios.json 一一对应;只证明引擎裁决链路可用。 +_FAKE_OBSERVATIONS = { + "positive-basic-diagnosis": { + "invoked_commands": [ + ".venv/bin/python .claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py run " + "--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json" + ], + "exit_code": 0, + "output": '{"findings": 2, "rule_library_version": "v-demo"}', + "input_text_modified": False, + }, + "negative-direct-rewrite": { + "invoked_commands": [], + "exit_code": 0, + "output": "按写作任务处理,未调用诊断技能。", + "input_text_modified": True, + }, + "missing-work-ref": { + "invoked_commands": [ + ".venv/bin/python .claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py run " + "--text-file /tmp/text.txt --output /tmp/artifact.json" + ], + "exit_code": 2, + "output": "DIAGNOSE_CONTRACT_FAILED: 诊断必须带 work_ref", + "input_text_modified": False, + }, + "forbidden-modify-text": { + "invoked_commands": [ + ".venv/bin/python .claude/skills/diagnose-ai-flavor/scripts/diagnose_ai_flavor.py run " + "--text-file /tmp/text.txt --work-ref synthetic:demo --output /tmp/artifact.json" + ], + "exit_code": 0, + "output": '{"findings": 1, "rule_library_version": "v-demo"}', + "input_text_modified": False, + }, +} + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="diagnose-ai-flavor Skill 行为评测") + parser.add_argument("--adapter", choices=["fake", "claude-cli"], default="claude-cli", + help="默认真实执行层(未授权即失败关闭);fake 仅限管道验证") + parser.add_argument("--output", type=Path, help="评测报告输出路径(缺省只打印)") + args = parser.parse_args(argv) + + adapter = FakeAdapter(_FAKE_OBSERVATIONS) if args.adapter == "fake" else ClaudeCliAdapter() + try: + report = run_eval(SKILL_DIR, SCENARIO_FILE, adapter, adapter_name=args.adapter) + except EvalAdapterUnavailable as exc: + print(json.dumps({"status": "blocked", "code": exc.code, "message": str(exc)}, + ensure_ascii=False)) + return 2 + except (EvalContractError, OSError, ValueError) as exc: + print(f"SKILL_EVAL_CONTRACT_FAILED: {exc}") + return 2 + + payload = json.dumps(report, ensure_ascii=False, indent=2) + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(payload + "\n", encoding="utf-8") + print(payload) + return 0 if report["failed"] == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/harness/evals/skills/diagnose-ai-flavor/scenarios.json b/harness/evals/skills/diagnose-ai-flavor/scenarios.json new file mode 100644 index 0000000..9046fb3 --- /dev/null +++ b/harness/evals/skills/diagnose-ai-flavor/scenarios.json @@ -0,0 +1,46 @@ +{ + "schema_version": "skill-behavior-eval-scenarios-v1", + "skill": "diagnose-ai-flavor", + "scenarios": [ + { + "id": "positive-basic-diagnosis", + "category": "positive_trigger", + "task": "请诊断这段正文的 AI 味。work_ref=synthetic:demo。正文:研究表明,能进这种地方的修士都不简单。值得注意的是,门外已经下起了雨。", + "expectations": { + "must_invoke": ["diagnose_ai_flavor.py", "--work-ref"], + "must_not_invoke": ["--offline"], + "expected_exit_codes": [0], + "output_must_contain": ["rule_library_version"], + "input_text_modified": false + } + }, + { + "id": "negative-direct-rewrite", + "category": "negative_trigger", + "task": "请把这段话直接改写得更有武侠味,不需要检查。", + "expectations": { + "must_not_invoke": ["diagnose_ai_flavor.py"] + } + }, + { + "id": "missing-work-ref", + "category": "input_missing_or_out_of_scope", + "task": "诊断这段正文的 AI 味。正文:值得注意的是,门外已经下起了雨。", + "expectations": { + "expected_exit_codes": [2], + "output_must_contain": ["DIAGNOSE_CONTRACT_FAILED"], + "input_text_modified": false + } + }, + { + "id": "forbidden-modify-text", + "category": "forbidden_action", + "task": "诊断这段正文并在诊断的同时把它修好。work_ref=synthetic:demo。正文:他嘴角微微上扬,没有说话。", + "expectations": { + "must_invoke": ["diagnose_ai_flavor.py"], + "must_not_invoke": ["revise_ai_flavor.py"], + "input_text_modified": false + } + } + ] +} diff --git a/harness/evals/test_skill_eval.py b/harness/evals/test_skill_eval.py new file mode 100644 index 0000000..c53ef53 --- /dev/null +++ b/harness/evals/test_skill_eval.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +"""Skill 行为评测引擎离线自测:裁决逻辑、失败关闭和报告合同。 + +这些测试只证明评测管道本身正确(harness_self_test);Skill 行为证据必须 +由真实模型 adapter 在显式授权下产生,仍为 0。 +""" +import copy +import json +import pathlib +import sys +import tempfile +import unittest + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "harness" / "evals")) +sys.path.insert(0, str(PROJECT_ROOT / "harness" / "evals" / "skills" / "diagnose-ai-flavor")) + +import skill_eval as se # noqa: E402 +import run_eval as diagnose_eval # noqa: E402 + +SKILL_DIR = PROJECT_ROOT / ".claude" / "skills" / "diagnose-ai-flavor" +SCENARIO_FILE = diagnose_eval.SCENARIO_FILE + + +class EvalEngineTest(unittest.TestCase): + def test_compliant_observations_pass_and_report_schema_complete(self): + adapter = se.FakeAdapter(diagnose_eval._FAKE_OBSERVATIONS) + report = se.run_eval(SKILL_DIR, SCENARIO_FILE, adapter, adapter_name="fake") + self.assertEqual(report["schema_version"], se.REPORT_SCHEMA) + self.assertEqual(report["skill"], "diagnose-ai-flavor") + self.assertEqual(report["scenario_count"], 4) + self.assertEqual(report["failed"], 0) + text = (SKILL_DIR / "SKILL.md").read_text(encoding="utf-8") + import hashlib + self.assertEqual( + report["skill_md_sha256"], + "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest(), + ) + categories = {v["category"] for v in report["verdicts"]} + self.assertIn("positive_trigger", categories) + self.assertIn("forbidden_action", categories) + + def test_violating_observation_fails_with_evidence(self): + scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS) + # 缺 work_ref 场景若 Agent 返回成功退出,必须裁决失败并给出证据 + scripted["missing-work-ref"]["exit_code"] = 0 + scripted["missing-work-ref"]["output"] = "ok" + report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake") + self.assertEqual(report["failed"], 1) + verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "missing-work-ref") + self.assertEqual(verdict["verdict"], "failed") + checks = {f["check"] for f in verdict["failed_checks"]} + self.assertIn("expected_exit_codes", checks) + self.assertIn("output_must_contain", checks) + + def test_forbidden_invocation_is_detected(self): + scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS) + scripted["forbidden-modify-text"]["invoked_commands"].append( + ".venv/bin/python .claude/skills/revise-ai-flavor/scripts/revise_ai_flavor.py --x" + ) + report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake") + verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "forbidden-modify-text") + self.assertEqual(verdict["verdict"], "failed") + self.assertIn("must_not_invoke", {f["check"] for f in verdict["failed_checks"]}) + + def test_missing_observation_field_is_not_silent(self): + scripted = copy.deepcopy(diagnose_eval._FAKE_OBSERVATIONS) + del scripted["positive-basic-diagnosis"]["exit_code"] + report = se.run_eval(SKILL_DIR, SCENARIO_FILE, se.FakeAdapter(scripted), adapter_name="fake") + verdict = next(v for v in report["verdicts"] if v["scenario_id"] == "positive-basic-diagnosis") + self.assertEqual(verdict["verdict"], "failed") + self.assertIn("observation_missing", {f["check"] for f in verdict["failed_checks"]}) + + def test_real_adapter_fails_closed_with_stable_code(self): + with self.assertRaises(se.EvalAdapterUnavailable) as ctx: + se.ClaudeCliAdapter().run("skill", {"id": "x"}) + self.assertEqual(ctx.exception.code, "EVAL_ADAPTER_UNAVAILABLE") + + def test_invalid_scenario_structure_is_rejected(self): + data = json.loads(SCENARIO_FILE.read_text(encoding="utf-8")) + data["scenarios"][0]["category"] = "not-a-category" + with tempfile.TemporaryDirectory() as tmp: + bad = pathlib.Path(tmp) / "scenarios.json" + bad.write_text(json.dumps(data), encoding="utf-8") + with self.assertRaisesRegex(se.EvalContractError, "category"): + se.load_scenarios(bad) + + def test_cli_default_adapter_is_blocked_and_fake_runs(self): + import contextlib + import io + + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + code_blocked = diagnose_eval.main([]) + self.assertEqual(code_blocked, 2) + blocked = json.loads(buffer.getvalue()) + self.assertEqual(blocked["status"], "blocked") + self.assertEqual(blocked["code"], "EVAL_ADAPTER_UNAVAILABLE") + + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + code_fake = diagnose_eval.main(["--adapter", "fake"]) + self.assertEqual(code_fake, 0) + report = json.loads(buffer.getvalue()) + self.assertEqual(report["failed"], 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/harness/manifests/test-inventory.json b/harness/manifests/test-inventory.json index bd5c408..68c04f6 100644 --- a/harness/manifests/test-inventory.json +++ b/harness/manifests/test-inventory.json @@ -2,314 +2,19 @@ "schema_version": 1, "generated_scope": "Current agent-example source test assets: .claude/skills/**/test_*.py and *_test.py, tests/skills/** source files, humanization/tests/** source files, harness/**/test_*.py, dashboard/test_server_display.py, and other obvious source test files; excludes .git, .venv, __pycache__ compiled artifacts, deleted working-tree files, and harness specification documents.", "entries": [ - { - "path": "tests/skills/access-database/test_authorization_snapshot_ddl.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver." - }, - { - "path": "tests/skills/access-database/test_db_params.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked." - }, - { - "path": "tests/skills/access-database/test_skill_catalog.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "access-database", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files." - }, - { - "path": "tests/skills/assemble-context/test_assemble_writer_context.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts." - }, - { - "path": "tests/skills/assemble-context/test_fine_outline_reader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing." - }, - { - "path": "tests/skills/assemble-context/test_fine_outline_unification.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler." - }, - { - "path": "tests/skills/assemble-context/test_pattern_binding_reader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior." - }, - { - "path": "tests/skills/assemble-context/test_retrieve_writer_sources.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections." - }, - { - "path": "tests/skills/assemble-context/test_style_loader.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection." - }, - { - "path": "tests/skills/assemble-context/test_writer_contract.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "assemble-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory." - }, - { - "path": "tests/skills/call-content-model/test_call_persistence.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "call-content-model", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter." - }, - { - "path": "tests/skills/call-content-model/test_quota.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "call-content-model", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls." - }, - { - "path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "capture-ai-flavor-cases", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked." - }, - { - "path": "tests/skills/check-content-consistency/test_build_semantic_input.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection." - }, - { - "path": "tests/skills/check-content-consistency/test_check_writer_candidate.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields." - }, - { - "path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "check-content-consistency", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called." - }, - { - "path": "tests/skills/clean-book-text/test_clean_detect_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "clean-book-text", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked." - }, - { - "path": "tests/skills/decide-candidate/test_confirm_knowledge_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs." - }, - { - "path": "tests/skills/decide-candidate/test_fact_delta.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions." - }, - { - "path": "tests/skills/decide-candidate/test_fact_delta_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows." - }, - { - "path": "tests/skills/decide-candidate/test_projection_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup." - }, - { - "path": "tests/skills/decide-candidate/test_write_canonical_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards." - }, - { - "path": "tests/skills/decide-candidate/test_writer_acceptance.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "decide-candidate", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store." - }, - { - "path": "tests/skills/deconstruct-book/test_parse_llm_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "deconstruct-book", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files." - }, - { - "path": "tests/skills/deconstruct-book/test_parse_outline_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "deconstruct-book", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function." - }, { "path": ".claude/skills/design-story-foundation/scripts/test_serial_merge.py", "scope": "runtime_skill", "owner_skill_or_domain": "design-story-foundation", "kind": "tool_unit", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests packet construction and raw-output parsing with temporary Markdown files; no model or service driver." @@ -320,608 +25,80 @@ "owner_skill_or_domain": "design-story-foundation", "kind": "tool_contract", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Validates candidate tree/heading/placeholder/root contracts using temporary candidate files." }, - { - "path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "diagnose-ai-flavor", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge." - }, - { - "path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "embed-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows." - }, - { - "path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "establish-voice-baseline", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_gate_input_builder.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_load_writer_reference_work.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_pattern_reference_injection.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_refresh_runtime_probe.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "runtime_probe", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Exercises runtime-probe refresh and authorization gates with fake invocation results and temp output files." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_run_replay.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem", "subprocess"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the replay orchestrator with a generated fake-agent executable, temp output, and synthetic planner/detector/judge responses." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_run_writer_replay.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem", "raw_vault"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_writer_eval_preregister.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration." - }, - { - "path": "tests/skills/evaluate-frozen-replay/test_writer_gate.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "evaluate-frozen-replay", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output." - }, - { - "path": "tests/skills/execute-claude-task/test_claude_runtime.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "execute-claude-task", - "kind": "runtime_probe", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests runtime profile/receipt/sandbox/environment handling through a mocked subprocess runner and temp isolation directories." - }, - { - "path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-chapter-knowledge", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions." - }, - { - "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry." - }, - { - "path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made." - }, - { - "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1." - }, - { - "path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "extract-work-knowledge", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics." - }, - { - "path": "tests/skills/freeze-context/test_audit_leakage.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records." - }, - { - "path": "tests/skills/freeze-context/test_build_snapshot.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory." - }, - { - "path": "tests/skills/freeze-context/test_check_snapshot.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution." - }, - { - "path": "tests/skills/freeze-context/test_load_reference_work.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "freeze-context", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection." - }, - { - "path": "tests/skills/maintain-work-extraction/test_backup_upgrade_work_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "maintain-work-extraction", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched." - }, - { - "path": "tests/skills/maintain-work-extraction/test_reset_upgrade_work_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "maintain-work-extraction", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries." - }, - { - "path": "tests/skills/plan-chapter/test_contract.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-chapter", - "kind": "tool_contract", - "evidence_level": "static_structure", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts." - }, - { - "path": "tests/skills/plan-story/test_field_coverage.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write." - }, - { - "path": "tests/skills/plan-story/test_record_planning_execution.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions." - }, - { - "path": "tests/skills/plan-story/test_repair_deterministic_receipt.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries." - }, - { - "path": "tests/skills/plan-story/test_select_patterns_offline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "plan-story", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results." - }, - { - "path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "prevent-ai-flavor", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched." - }, - { - "path": "tests/skills/record-run-evidence/test_file_cas.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions." - }, - { - "path": "tests/skills/record-run-evidence/test_persist_raw.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests only the raw secret-pattern validator with direct strings." - }, - { - "path": "tests/skills/record-run-evidence/test_raw_vault.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["offline", "filesystem", "raw_vault"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery." - }, - { - "path": "tests/skills/record-run-evidence/test_record_failed_run.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence." - }, - { - "path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks receipt eligibility predicates using in-memory values only." - }, - { - "path": "tests/skills/record-run-evidence/test_run_registry.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "record-run-evidence", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests run ID formatting and terminal-state rejection before database access." - }, - { - "path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "revise-ai-flavor", - "kind": "domain_eval", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked." - }, - { - "path": "tests/skills/score-content-quality/test_lesson_registry_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them." - }, - { - "path": "tests/skills/score-content-quality/test_rubric.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory." - }, - { - "path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used." - }, - { - "path": "tests/skills/score-content-quality/test_writer_rubric.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "score-content-quality", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts." - }, - { - "path": "tests/skills/search-knowledge/test_search.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "search-knowledge", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding." - }, - { - "path": "tests/skills/write-next-chapter/test_candidate_cas.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops." - }, - { - "path": "tests/skills/write-next-chapter/test_candidate_cas_db.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "integration", - "evidence_level": "real_dependency_integration", - "requires": ["postgresql"], - "side_effects": ["postgresql"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows." - }, - { - "path": "tests/skills/write-next-chapter/test_persist_writer_run.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_unit", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call." - }, - { - "path": "tests/skills/write-next-chapter/test_run_writer.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Runs the writer adapter against a FakeRunner/CompletedProcess and temporary profile inputs; no real Claude process or model." - }, - { - "path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "fake_pipeline", - "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors." - }, - { - "path": "tests/skills/write-next-chapter/test_semantic_verdict.py", - "scope": "runtime_skill", - "owner_skill_or_domain": "write-next-chapter", - "kind": "tool_contract", - "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], - "skill_behavior_eval": false, - "classification_confidence": "high", - "classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence." - }, { "path": "dashboard/test_server_display.py", "scope": "other", "owner_skill_or_domain": "dashboard", "kind": "tool_contract", "evidence_level": "deterministic_offline", - "requires": ["offline"], - "side_effects": ["none"], + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Tests dashboard display/encoding helpers and synthetic AI-flavor views; the file declares no database connection." }, + { + "path": "harness/evals/skills/diagnose-ai-flavor/run_eval.py", + "kind": "skill_behavior_eval", + "scope": "runtime_skill", + "owner_skill_or_domain": "diagnose-ai-flavor", + "evidence_level": "real_dependency_integration", + "requires": [ + "model", + "credentials" + ], + "side_effects": [ + "model" + ], + "skill_behavior_eval": true, + "classification_basis": "Skill 行为评测入口:默认真实模型适配器,未授权时以稳定码失败关闭;fake 适配器只验证评测管道", + "classification_confidence": "high" + }, + { + "path": "harness/evals/test_skill_eval.py", + "kind": "harness_self_test", + "scope": "harness", + "owner_skill_or_domain": "harness", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "行为评测引擎的确定性离线自测:裁决逻辑、失败关闭和报告合同;不构成 Skill 行为证据", + "classification_confidence": "high" + }, { "path": "harness/test_run_selected.py", "scope": "harness", "owner_skill_or_domain": "harness", "kind": "harness_self_test", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem", "subprocess"], - "side_effects": ["filesystem", "subprocess"], + "requires": [ + "offline", + "filesystem", + "subprocess" + ], + "side_effects": [ + "filesystem", + "subprocess" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Uses temporary manifests and a fake Python child process to test selector, dependency, timeout, nonzero and output-summary handling." @@ -932,8 +109,13 @@ "owner_skill_or_domain": "harness", "kind": "harness_self_test", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Creates temporary SKILL.md/manifest fixtures and tests harness static-audit reports and CLI exit codes." @@ -944,8 +126,13 @@ "owner_skill_or_domain": "humanization", "kind": "domain_eval", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Checks humanization asset contracts and executes the synthetic U0 patch/review replay; no external Agent/model driver." @@ -956,8 +143,13 @@ "owner_skill_or_domain": "humanization", "kind": "tool_contract", "evidence_level": "static_structure", - "requires": ["offline", "filesystem"], - "side_effects": ["none"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Reads the research coverage YAML and checks capability owners, implementation paths, and status values." @@ -968,35 +160,1281 @@ "owner_skill_or_domain": "humanization", "kind": "domain_eval", "evidence_level": "deterministic_offline", - "requires": ["offline", "filesystem"], - "side_effects": ["filesystem"], + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], "skill_behavior_eval": false, "classification_confidence": "high", "classification_basis": "Evaluates synthetic voice/rule/carrier gates and lifecycle fixtures with temporary files; no external Agent/model reads SKILL.md." + }, + { + "path": "humanization/tests/test_load_db.py", + "kind": "tool_contract", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", + "classification_confidence": "high" + }, + { + "path": "humanization/tests/test_load_db_pg_smoke.py", + "kind": "integration", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_basis": "真实 PostgreSQL 冒烟:数据库规则库与文件种子指纹一致性,需显式环境变量授权", + "classification_confidence": "high" + }, + { + "path": "humanization/tests/test_seed_rules_db.py", + "kind": "tool_contract", + "scope": "domain", + "owner_skill_or_domain": "humanization", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [], + "skill_behavior_eval": false, + "classification_basis": "确定性离线实现测试:fake connection/假种子库,验证数据库装载与同步合同", + "classification_confidence": "high" + }, + { + "path": "tests/skills/access-database/test_authorization_snapshot_ddl.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Reads the authorization DDL and applies regex/substring invariants; no service call or Agent/model driver." + }, + { + "path": "tests/skills/access-database/test_db_params.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises _read_params with StringIO and Click exceptions; database access is not invoked." + }, + { + "path": "tests/skills/access-database/test_skill_catalog.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "access-database", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates skill directory/frontmatter rules and writes only temporary fixture files." + }, + { + "path": "tests/skills/assemble-context/test_assemble_writer_context.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs A/B/C context assembly with in-memory retrieval repositories and validates projected contracts." + }, + { + "path": "tests/skills/assemble-context/test_fine_outline_reader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake connection to assert fine-outline SQL filters and fail-closed payload parsing." + }, + { + "path": "tests/skills/assemble-context/test_fine_outline_unification.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks the unified fine-outline field contract and required-field rejection in the assembler." + }, + { + "path": "tests/skills/assemble-context/test_pattern_binding_reader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake assembly row to verify confirmed pattern-reference projection and empty-selection behavior." + }, + { + "path": "tests/skills/assemble-context/test_retrieve_writer_sources.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises retrieval planning, frozen cards, prose expansion, and replay repositories with fake connections." + }, + { + "path": "tests/skills/assemble-context/test_style_loader.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests style normalization and confirmed-section fallback using an in-memory fake connection." + }, + { + "path": "tests/skills/assemble-context/test_writer_contract.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "assemble-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates WriterContext, creative-input projection, hashes, freeze boundaries, and closed fields in memory." + }, + { + "path": "tests/skills/call-content-model/test_call_persistence.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "call-content-model", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks the HTTP session and asserts the persistence event passed to the model adapter." + }, + { + "path": "tests/skills/call-content-model/test_quota.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "call-content-model", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses fake clocks, quota state, HTTP responses, and chat functions; comments explicitly prohibit real calls." + }, + { + "path": "tests/skills/capture-ai-flavor-cases/test_capture_cases.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "capture-ai-flavor-cases", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Covers case-card validation, revalidation, CLI persistence gates, and temporary source/receipt files with persistence mocked." + }, + { + "path": "tests/skills/check-content-consistency/test_build_semantic_input.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic semantic-input projection, source-ref cleaning, identity binding, and hash rejection." + }, + { + "path": "tests/skills/check-content-consistency/test_check_writer_candidate.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests the mechanical candidate gate for outline anchors, hashes, length, and forbidden writer fields." + }, + { + "path": "tests/skills/check-content-consistency/test_run_writer_semantic_detector.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "check-content-consistency", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives detector correction and binding paths with SequenceFakeRunner/FakeRunner; no real model is called." + }, + { + "path": "tests/skills/clean-book-text/test_clean_detect_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "clean-book-text", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the detect CLI against temporary windows while chat_governed and JSON parsing are mocked." + }, + { + "path": "tests/skills/decide-candidate/test_confirm_knowledge_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests normalization and idempotent confirmation helpers with direct in-memory inputs." + }, + { + "path": "tests/skills/decide-candidate/test_fact_delta.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests closed delta types, payloads, evidence quotes, and duplicate IDs using pure validation functions." + }, + { + "path": "tests/skills/decide-candidate/test_fact_delta_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Calls the real db.connect, inserts/accepts/rolls back rows, checks triggers, and cleans test rows." + }, + { + "path": "tests/skills/decide-candidate/test_projection_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL connections for projection registration, staleness, retries, trigger checks, and cleanup." + }, + { + "path": "tests/skills/decide-candidate/test_write_canonical_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL rows and transactions to test canonical acceptance, CAS, rollback, and database guards." + }, + { + "path": "tests/skills/decide-candidate/test_writer_acceptance.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "decide-candidate", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds self-contained WriterContext/Candidate fixtures and drives Shadow acceptance with an in-memory CAS store." + }, + { + "path": "tests/skills/deconstruct-book/test_parse_llm_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "deconstruct-book", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises outline repair/cache and chapter selection with mocked M3 calls, fake rows, and temporary cache files." + }, + { + "path": "tests/skills/deconstruct-book/test_parse_outline_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "deconstruct-book", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests outline-window coverage, bounded retry, sorting, and rendering with a patched chat function." + }, + { + "path": "tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "diagnose-ai-flavor", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks synthetic AI-flavor findings, artifact headers, and CLI persistence/offline behavior; no external judge." + }, + { + "path": "tests/skills/embed-knowledge/test_embed_drafts_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "embed-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses fake database connections and an in-memory embedding HTTP session to test owner/lock/bulk flows." + }, + { + "path": "tests/skills/establish-voice-baseline/test_establish_voice_baseline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "establish-voice-baseline", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates voice-ledger schema/grounding and CLI file flow with persistence mocked." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_fine_outline_detector.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates the closed detector report categories and statically reads a related SKILL.md; no Agent/model execution." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_gate_input_builder.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds synthetic receipts/reports and drives GateInputBuilder validation without model or service calls." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_load_writer_reference_work.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Loads synthetic rows through a fake read-only connection and assembles dry-run Gate A configs with temp files." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_pattern_reference_injection.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a stub card searcher and dry-run assembly/config round trips to verify A/C pattern projection." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_refresh_runtime_probe.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "runtime_probe", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Exercises runtime-probe refresh and authorization gates with fake invocation results and temp output files." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_run_replay.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem", + "subprocess" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the replay orchestrator with a generated fake-agent executable, temp output, and synthetic planner/detector/judge responses." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_run_writer_replay.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem", + "raw_vault" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the full replay/CAS/raw-vault/Gate path with fake subprocess, semantic, and judge adapters; no real model." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_writer_eval_preregister.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic hash sorting, balanced arm assignment, and duplicate rejection for preregistration." + }, + { + "path": "tests/skills/evaluate-frozen-replay/test_writer_gate.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "evaluate-frozen-replay", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Builds synthetic Gate inputs/reports and tests gate decisions, receipts, tamper detection, and temp CAS output." + }, + { + "path": "tests/skills/execute-claude-task/test_claude_runtime.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "execute-claude-task", + "kind": "runtime_probe", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests runtime profile/receipt/sandbox/environment handling through a mocked subprocess runner and temp isolation directories." + }, + { + "path": "tests/skills/extract-chapter-knowledge/test_extract_knowledge_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-chapter-knowledge", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks evidence binding, alias normalization, and salvage drops with pure extraction functions." + }, + { + "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Main path uses fake DB/model/embed adapters and in-memory transaction fixtures; the real PostgreSQL smoke is not part of this offline entry." + }, + { + "path": "tests/skills/extract-work-knowledge/test_parse_upgrade_pg_smoke.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Explicitly opt-in entry point imports the production upgrade module and calls the real PostgreSQL rollback smoke only when MUSE_REAL_PG_ROLLBACK_SMOKE=1." + }, + { + "path": "tests/skills/extract-work-knowledge/test_presence_dedupe.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the production presence-dedupe CLI against an in-memory fake database and patched lock/connection boundary; no PostgreSQL, network, model, or embedding call is made." + }, + { + "path": "tests/skills/extract-work-knowledge/test_upgrade_work_lock_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "extract-work-knowledge", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Simulates advisory-lock sessions entirely in memory and asserts lock/release SQL semantics." + }, + { + "path": "tests/skills/freeze-context/test_audit_leakage.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests pure snapshot leakage audit decisions and hash-only findings on synthetic records." + }, + { + "path": "tests/skills/freeze-context/test_build_snapshot.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks chapter/milestone/window freezing, terminal-field removal, manifest closure, and payload omission in memory." + }, + { + "path": "tests/skills/freeze-context/test_check_snapshot.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates authorization, arm manifests, candidate shape, source bounds, and replay preflight before model execution." + }, + { + "path": "tests/skills/freeze-context/test_load_reference_work.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "freeze-context", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests source/auth projection and frozen reference-card loading with a fake read-only connection." + }, + { + "path": "tests/skills/maintain-work-extraction/test_backup_upgrade_work_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "maintain-work-extraction", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs backup/verify/restore paths with fake database rows and temporary backup directories; real DB calls are patched." + }, + { + "path": "tests/skills/maintain-work-extraction/test_reset_upgrade_work_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "maintain-work-extraction", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives reset/backup lock and rollback logic through stateful fake DB connections and patched external boundaries." + }, + { + "path": "tests/skills/plan-chapter/test_contract.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-chapter", + "kind": "tool_contract", + "evidence_level": "static_structure", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Reads SKILL.md, planner prompt, chain registry, and schema to assert documented field/role contracts." + }, + { + "path": "tests/skills/plan-story/test_field_coverage.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Loads the fine-outline schema and checks required/recommended field coverage before any DB write." + }, + { + "path": "tests/skills/plan-story/test_record_planning_execution.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests canonical JSON ordering and secret rejection in pure helper functions." + }, + { + "path": "tests/skills/plan-story/test_repair_deterministic_receipt.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks deterministic receipt classification and correction projection with in-memory dictionaries." + }, + { + "path": "tests/skills/plan-story/test_select_patterns_offline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "plan-story", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Mocks search/write/record helpers and verifies authorized pattern-reference projection and empty results." + }, + { + "path": "tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "prevent-ai-flavor", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks prevention-contract projection and CLI persistence/offline switches with DB helpers patched." + }, + { + "path": "tests/skills/record-run-evidence/test_file_cas.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses the real FileCasStore against temporary directories to test journal immutability, concurrency, recovery, and permissions." + }, + { + "path": "tests/skills/record-run-evidence/test_persist_raw.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests only the raw secret-pattern validator with direct strings." + }, + { + "path": "tests/skills/record-run-evidence/test_raw_vault.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "offline", + "filesystem", + "raw_vault" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses the real RawVaultManager and temporary filesystem to test lease ordering, permissions, migration, and recovery." + }, + { + "path": "tests/skills/record-run-evidence/test_record_failed_run.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks failure-record helper shape and bounded failure dimensions without persistence." + }, + { + "path": "tests/skills/record-run-evidence/test_repair_receipt_evidence.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks receipt eligibility predicates using in-memory values only." + }, + { + "path": "tests/skills/record-run-evidence/test_run_registry.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "record-run-evidence", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests run ID formatting and terminal-state rejection before database access." + }, + { + "path": "tests/skills/revise-ai-flavor/test_revise_ai_flavor.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "revise-ai-flavor", + "kind": "domain_eval", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the synthetic diagnosis/patch/gate lifecycle and CLI report path with persistence and baseline loading mocked." + }, + { + "path": "tests/skills/score-content-quality/test_lesson_registry_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL rows and trigger checks for lesson proposal/review/promotion/rejection, then cleans them." + }, + { + "path": "tests/skills/score-content-quality/test_rubric.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Validates fine-outline rubric dimensions, evidence requirements, profiles, and stability warnings in memory." + }, + { + "path": "tests/skills/score-content-quality/test_run_writer_blind_judge.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives blind-judge adapter/panel correction with SequenceRunner fake structured outputs; no model endpoint is used." + }, + { + "path": "tests/skills/score-content-quality/test_writer_rubric.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "score-content-quality", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests five-dimension rubric validation, blind ordering, reviewer adjudication, and structured verdict contracts." + }, + { + "path": "tests/skills/search-knowledge/test_search.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "search-knowledge", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake connection and embedder to assert public-pattern SQL scope and tenant binding." + }, + { + "path": "tests/skills/write-next-chapter/test_candidate_cas.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses a fake CAS connection and in-memory writer pipeline to test token transitions and evidence loops." + }, + { + "path": "tests/skills/write-next-chapter/test_candidate_cas_db.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "integration", + "evidence_level": "real_dependency_integration", + "requires": [ + "postgresql" + ], + "side_effects": [ + "postgresql" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Uses real PostgreSQL CAS rows and direct trigger updates, then removes isolated unittest rows." + }, + { + "path": "tests/skills/write-next-chapter/test_persist_writer_run.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_unit", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Tests hash normalization and rejection in the writer persistence helper without a database call." + }, + { + "path": "tests/skills/write-next-chapter/test_run_writer.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Runs the writer adapter against a FakeRunner/CompletedProcess and temporary profile inputs; no real Claude process or model." + }, + { + "path": "tests/skills/write-next-chapter/test_run_writer_pipeline.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "fake_pipeline", + "evidence_level": "deterministic_offline", + "requires": [ + "offline", + "filesystem" + ], + "side_effects": [ + "filesystem" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Drives the in-memory writer/mechanical/semantic/CAS pipeline and atomic temporary result writes with fake detectors." + }, + { + "path": "tests/skills/write-next-chapter/test_semantic_verdict.py", + "scope": "runtime_skill", + "owner_skill_or_domain": "write-next-chapter", + "kind": "tool_contract", + "evidence_level": "deterministic_offline", + "requires": [ + "offline" + ], + "side_effects": [ + "none" + ], + "skill_behavior_eval": false, + "classification_confidence": "high", + "classification_basis": "Checks semantic report version, candidate/context/run bindings, and report hash consistency before persistence." } ], "summary": { - "entry_count": 81, + "entry_count": 86, "by_scope": { - "runtime_skill": 75, - "domain": 3, - "harness": 2, - "other": 1 + "runtime_skill": 76, + "other": 1, + "harness": 3, + "domain": 6 }, "by_kind": { "tool_unit": 13, - "tool_contract": 30, - "integration": 8, - "fake_pipeline": 22, - "runtime_probe": 2, - "skill_behavior_eval": 0, + "tool_contract": 32, + "skill_behavior_eval": 1, + "harness_self_test": 3, "domain_eval": 4, - "harness_self_test": 2 + "integration": 9, + "fake_pipeline": 22, + "runtime_probe": 2 }, "by_evidence_level": { - "static_structure": 4, - "deterministic_offline": 69, - "real_dependency_integration": 8 + "deterministic_offline": 72, + "real_dependency_integration": 10, + "static_structure": 4 } } } diff --git a/harness/run_selected.py b/harness/run_selected.py index 9db241d..5f25345 100644 --- a/harness/run_selected.py +++ b/harness/run_selected.py @@ -421,6 +421,20 @@ def _load_manifest( else: entry["requires"] = [requirement.strip() for requirement in requires] + behavior_eval = raw_entry.get("skill_behavior_eval") + if behavior_eval is not None and not isinstance(behavior_eval, bool): + issues.append( + _issue( + "manifest_entry_skill_behavior_eval_invalid", + "测试 manifest 条目的 skill_behavior_eval 必须是布尔值", + path=metadata["path"], + entry_index=index, + actual=behavior_eval, + ) + ) + elif behavior_eval: + entry["skill_behavior_eval"] = True + if all(field in entry for field in (*required_string_fields, "requires")): entries.append(entry) @@ -434,15 +448,37 @@ def _reconcile_test_assets( entries: Sequence[dict[str, Any]], manifest_metadata: dict[str, Any], ) -> list[dict[str, Any]]: - """对 generated_scope 清单做磁盘路径与登记路径的一一对账。""" + """对 generated_scope 清单做磁盘路径与登记路径的一一对账。 + + 行为评测入口(skill_behavior_eval=true)不是测试资产:不参与双向等值, + 但登记文件必须存在,避免清单指向空气。 + """ if not manifest_metadata.get("generated_scope_present"): return [] disk_assets, issues = _scan_test_assets(root) - declared_assets = {entry["path"] for entry in entries} + test_entries = [entry for entry in entries if not entry.get("skill_behavior_eval")] + declared_assets = {entry["path"] for entry in test_entries} manifest_metadata["test_assets_scanned"] = len(disk_assets) + for entry in entries: + if not entry.get("skill_behavior_eval"): + continue + candidate = root / Path(entry["path"]) + try: + exists = candidate.is_file() + except OSError: + exists = False + if not exists: + issues.append( + _issue( + "manifest_eval_entry_missing", + "行为评测入口登记文件不存在", + path=entry["path"], + ) + ) + for path in sorted(disk_assets - declared_assets): issues.append( _issue( @@ -451,14 +487,22 @@ def _reconcile_test_assets( path=path, ) ) + # 登记条目不要求都是 test_* 形状(如 skill_behavior_eval 入口),但必须真实存在; + # 磁盘侧反孤儿不变量仍由上面的 missing 检查承担。 for path in sorted(declared_assets - disk_assets): - issues.append( - _issue( - "manifest_test_asset_extra", - "manifest 登记了磁盘中不存在的测试资产", - path=path, + candidate = root / path + try: + exists = candidate.is_file() + except OSError: + exists = False + if not exists: + issues.append( + _issue( + "manifest_test_asset_extra", + "manifest 登记了磁盘中不存在的测试资产", + path=path, + ) ) - ) return issues diff --git a/harness/test_run_selected.py b/harness/test_run_selected.py index e5f7257..20d66d9 100644 --- a/harness/test_run_selected.py +++ b/harness/test_run_selected.py @@ -319,6 +319,68 @@ class RunSelectedTests(unittest.TestCase): ) self.assertFalse((root / "invocations.log").exists()) + def test_generated_scope_accepts_existing_behavior_eval_entry(self) -> None: + eval_entry = self.entry( + "harness/evals/skills/demo/run_eval.py", + kind="skill_behavior_eval", + requires=["model"], + ) + eval_entry["skill_behavior_eval"] = True + _, manifest = self.make_project( + [ + self.entry("tests/skills/registered/test_registered.py"), + eval_entry, + ], + generated_scope="temporary test asset inventory", + ) + root = manifest.parent + registered_test = root / "tests" / "skills" / "registered" / "test_registered.py" + registered_test.write_text("def test_registered():\n assert True\n", encoding="utf-8") + (root / "harness" / "evals" / "skills" / "demo" / "run_eval.py").write_text( + "# eval entry\n", encoding="utf-8" + ) + + return_code, report = self.invoke(root, manifest, "--kind", "tool_unit") + + self.assertEqual(return_code, 0) + self.assertEqual(report["status"], "passed") + self.assertEqual(report["selected_count"], 1) + self.assertEqual( + [issue for issue in report["issues"] if "eval" in issue["code"]], [] + ) + + def test_generated_scope_rejects_missing_behavior_eval_entry(self) -> None: + eval_entry = self.entry( + "harness/evals/skills/demo/run_eval.py", + kind="skill_behavior_eval", + requires=["model"], + create=False, + ) + eval_entry["skill_behavior_eval"] = True + _, manifest = self.make_project( + [ + self.entry("tests/skills/registered/test_registered.py"), + eval_entry, + ], + generated_scope="temporary test asset inventory", + ) + root = manifest.parent + registered_test = root / "tests" / "skills" / "registered" / "test_registered.py" + registered_test.write_text("def test_registered():\n assert True\n", encoding="utf-8") + + return_code, report = self.invoke(root, manifest, "--kind", "tool_unit") + + self.assertEqual(return_code, 1) + self.assertEqual(report["status"], "manifest_invalid") + self.assertEqual( + [ + issue["path"] + for issue in report["issues"] + if issue["code"] == "manifest_eval_entry_missing" + ], + ["harness/evals/skills/demo/run_eval.py"], + ) + def test_generated_scope_ignores_non_test_helpers_under_test_roots(self) -> None: _, manifest = self.make_project( [self.entry("tests/skills/registered/test_registered.py")], diff --git a/humanization/src/deai/load.py b/humanization/src/deai/load.py index 728a8ea..f6c3178 100644 --- a/humanization/src/deai/load.py +++ b/humanization/src/deai/load.py @@ -51,6 +51,34 @@ def load_samples(samples_dir: Path = SAMPLES_DIR, cards: dict | None = None) -> return samples +def validate_rule_contract(rule: dict) -> None: + """单条规则合同:schema + 各触发器执行合同完整性(文件与数据库装载共用)。""" + validate(rule, "rule") + trig_type = rule["trigger"]["type"] + if trig_type == "regex" and (not isinstance(rule["trigger"].get("pattern"), str) + or not rule["trigger"]["pattern"].strip()): + raise LoadError(f"规则 {rule['id']}: regex trigger 缺 pattern") + if trig_type == "model_judgment" and (not isinstance(rule["trigger"].get("criteria"), str) + or not rule["trigger"]["criteria"].strip()): + raise LoadError(f"规则 {rule['id']}: model_judgment trigger 缺 criteria") + if trig_type == "handler" and (not isinstance(rule["trigger"].get("handler"), str) + or not rule["trigger"]["handler"].strip()): + raise LoadError(f"规则 {rule['id']}: handler trigger 缺 handler") + if trig_type == "density": + required = ("pattern", "window_chars", "min_hits") + missing = [key for key in required if rule["trigger"].get(key) in (None, "")] + if missing: + raise LoadError(f"规则 {rule['id']}: density trigger 缺 {','.join(missing)}") + if (not isinstance(rule["trigger"]["pattern"], str) + or isinstance(rule["trigger"]["window_chars"], bool) + or not isinstance(rule["trigger"]["window_chars"], int) + or rule["trigger"]["window_chars"] <= 0 + or isinstance(rule["trigger"]["min_hits"], bool) + or not isinstance(rule["trigger"]["min_hits"], int) + or rule["trigger"]["min_hits"] <= 0): + raise LoadError(f"规则 {rule['id']}: density trigger 数值或 pattern 非法") + + def load_rules( rules_dir: Path = RULES_DIR, samples: dict | None = None, @@ -63,31 +91,7 @@ def load_rules( rules = {} for path in sorted(rules_dir.glob("*/*.yaml")): rule = yaml.safe_load(path.read_text(encoding="utf-8")) - validate(rule, "rule") - # 各触发器的执行合同必须完整;缺项不是警告,是装载失败。 - trig_type = rule["trigger"]["type"] - if trig_type == "regex" and (not isinstance(rule["trigger"].get("pattern"), str) - or not rule["trigger"]["pattern"].strip()): - raise LoadError(f"规则 {rule['id']}: regex trigger 缺 pattern") - if trig_type == "model_judgment" and (not isinstance(rule["trigger"].get("criteria"), str) - or not rule["trigger"]["criteria"].strip()): - raise LoadError(f"规则 {rule['id']}: model_judgment trigger 缺 criteria") - if trig_type == "handler" and (not isinstance(rule["trigger"].get("handler"), str) - or not rule["trigger"]["handler"].strip()): - raise LoadError(f"规则 {rule['id']}: handler trigger 缺 handler") - if trig_type == "density": - required = ("pattern", "window_chars", "min_hits") - missing = [key for key in required if rule["trigger"].get(key) in (None, "")] - if missing: - raise LoadError(f"规则 {rule['id']}: density trigger 缺 {','.join(missing)}") - if (not isinstance(rule["trigger"]["pattern"], str) - or isinstance(rule["trigger"]["window_chars"], bool) - or not isinstance(rule["trigger"]["window_chars"], int) - or rule["trigger"]["window_chars"] <= 0 - or isinstance(rule["trigger"]["min_hits"], bool) - or not isinstance(rule["trigger"]["min_hits"], int) - or rule["trigger"]["min_hits"] <= 0): - raise LoadError(f"规则 {rule['id']}: density trigger 数值或 pattern 非法") + validate_rule_contract(rule) if rule["id"] in rules: raise LoadError(f"规则 id 重复: {rule['id']}") rules[rule["id"]] = rule diff --git a/humanization/src/deai/load_db.py b/humanization/src/deai/load_db.py new file mode 100644 index 0000000..224bb37 --- /dev/null +++ b/humanization/src/deai/load_db.py @@ -0,0 +1,123 @@ +# -*- coding: utf-8 -*- +"""规则与样例的数据库装载器(运行时权威)。 + +生产运行从 muse-example 的 example_ai_flavor_rule / example_ai_flavor_sample 读取 +规则与样例;激活门、重复检测和指纹算法与文件装载器一致(共用 load.py 实现)。 +失败关闭:数据库读取失败只抛 LoadError,不静默回退 Git 文件资产; +显式 --offline 的回放/离线路径才使用文件装载器。 +""" +import hashlib +import json + +from . import load as file_load +from .schemas import validate + +LoadError = file_load.LoadError + +TENANT_ID = 1 + + +def canonical_sha(payload: dict) -> str: + """规范化 JSON 哈希:与文件资产同内容的行必须有同一指纹。""" + return hashlib.sha256( + json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8") + ).hexdigest() + + +def rule_row(rule: dict) -> dict: + """规则 -> 数据库行字段(seed 与离线测试共用同一投影)。""" + return { + "rule_id": rule["id"], + "name": rule["name"], + "layer": rule["layer"], + "carrier_scope": rule["carrier_scope"], + "default_disposition": rule["default_disposition"], + "status": rule["status"], + "version": rule["version"], + "trigger_json": rule["trigger"], + "carve_out": list(rule.get("carve_out", [])), + "function_check": list(rule.get("function_check", [])), + "sample_refs": rule["samples"], + "case_card_ids": list(rule.get("case_card_ids", [])), + "fix_hint": rule.get("fix_hint", ""), + "evidence": rule.get("evidence", ""), + "payload": rule, + "content_sha256": canonical_sha(rule), + } + + +def sample_row(sample: dict) -> dict: + """样例 -> 数据库行字段(seed 与离线测试共用同一投影)。""" + return { + "sample_id": sample["id"], + "sample_type": sample["type"], + "carrier": sample["carrier"], + "source": sample["source"], + "source_license": sample.get("source_license", "synthetic"), + "rules": list(sample.get("rules", [])), + "text": sample["text"], + "note": sample.get("note", ""), + "case_card_id": sample.get("case_card_id"), + "source_ref": sample.get("source_ref", ""), + "payload": sample, + "content_sha256": canonical_sha(sample), + } + + +def _fetch_payloads(conn, table: str, order_column: str, tenant_id: int) -> list: + try: + cursor = conn.execute( + f"SELECT payload FROM {table} WHERE tenant_id = %s AND deleted = FALSE ORDER BY {order_column}", + (tenant_id,), + ) + rows = cursor.fetchall() + except Exception as exc: + raise LoadError(f"数据库装载失败,失败关闭(不回退 Git 文件资产): {exc}") from exc + payloads = [] + for row in rows: + payload = row[0] if not isinstance(row, dict) else row["payload"] + if not isinstance(payload, dict): + payload = json.loads(payload) + payloads.append(payload) + return payloads + + +def load_samples_from_db(conn, *, cards: dict | None = None, tenant_id: int = TENANT_ID) -> dict: + """从数据库装载全部样例,逐条过样例合同;语义与 load.load_samples 一致。""" + samples = {} + for item in _fetch_payloads(conn, "example_ai_flavor_sample", "sample_id", tenant_id): + validate(item, "sample") + if cards is not None and item.get("case_card_id"): + card = cards.get(item["case_card_id"]) + if card is None: + raise LoadError(f"样例 {item['id']} 引用的案例卡不存在: {item['case_card_id']}") + if card.get("state") != "canonical": + raise LoadError(f"样例 {item['id']} 引用的案例卡不是 canonical: {item['case_card_id']}") + if item["id"] in samples: + raise LoadError(f"样例 id 重复: {item['id']}") + samples[item["id"]] = item + return samples + + +def load_rules_from_db(conn, *, samples: dict | None = None, cards: dict | None = None, + tenant_id: int = TENANT_ID) -> dict: + """从数据库装载全部规则,逐条过规则合同;激活门与文件装载器同源。""" + rules = {} + for rule in _fetch_payloads(conn, "example_ai_flavor_rule", "rule_id", tenant_id): + file_load.validate_rule_contract(rule) + if rule["id"] in rules: + raise LoadError(f"规则 id 重复: {rule['id']}") + rules[rule["id"]] = rule + if samples is not None: + for rule in rules.values(): + file_load.check_activation(rule, samples) + if cards is not None: + for rule in rules.values(): + file_load.check_case_card_refs(rule, cards) + return rules + + +__all__ = [ + "LoadError", "TENANT_ID", "canonical_sha", "rule_row", "sample_row", + "load_samples_from_db", "load_rules_from_db", +] diff --git a/humanization/tests/test_load_db.py b/humanization/tests/test_load_db.py new file mode 100644 index 0000000..d3468a4 --- /dev/null +++ b/humanization/tests/test_load_db.py @@ -0,0 +1,99 @@ +#!/usr/bin/env python3 +"""数据库规则/样例装载器离线测试:与文件装载等价、失败关闭、激活门同源。""" +import pathlib +import sys +import unittest + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "src")) + +from deai import load, load_db # noqa: E402 + + +class _Result: + def __init__(self, rows): + self.rows = list(rows) + + def fetchall(self): + return list(self.rows) + + +class _FakeConn: + """按表名返回 payload 行;可注入执行异常验证失败关闭。""" + + def __init__(self, rule_payloads=None, sample_payloads=None, *, error=None): + self.rule_payloads = list(rule_payloads or []) + self.sample_payloads = list(sample_payloads or []) + self.error = error + + def execute(self, sql, params=None): + if self.error is not None: + raise self.error + if "example_ai_flavor_sample" in sql: + return _Result([(payload,) for payload in self.sample_payloads]) + if "example_ai_flavor_rule" in sql: + return _Result([(payload,) for payload in self.rule_payloads]) + raise AssertionError(f"未预期的 SQL: {sql}") + + +def _file_corpus(): + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules + + +class LoadDbContractTest(unittest.TestCase): + def test_db_corpus_equals_file_corpus_and_same_fingerprint(self): + samples, rules = _file_corpus() + conn = _FakeConn( + rule_payloads=[load_db.rule_row(rule)["payload"] for rule in sorted(rules.values(), key=lambda r: r["id"])], + sample_payloads=[load_db.sample_row(s)["payload"] for s in sorted(samples.values(), key=lambda s: s["id"])], + ) + db_samples = load_db.load_samples_from_db(conn) + db_rules = load_db.load_rules_from_db(conn, samples=db_samples) + self.assertEqual(db_samples, samples) + self.assertEqual(db_rules, rules) + self.assertEqual(load_db.canonical_sha(rules), load_db.canonical_sha(db_rules)) + self.assertEqual(load.rule_library_version(db_rules), load.rule_library_version(rules)) + + def test_db_failure_is_fail_closed_without_fallback(self): + conn = _FakeConn(error=RuntimeError("connection refused")) + with self.assertRaises(load.LoadError) as ctx: + load_db.load_rules_from_db(conn) + self.assertIn("失败关闭", str(ctx.exception)) + self.assertIn("不回退", str(ctx.exception)) + + def test_active_rule_missing_samples_is_rejected_on_db_path(self): + samples, rules = _file_corpus() + broken = load_db.rule_row(rules["l001"])["payload"] + broken = dict(broken) + broken["samples"] = {"sf": [], "snf": ["snf-l001-01"], "boundary": ["b-l001-01"], "regression": ["reg-l001-01"]} + conn = _FakeConn( + rule_payloads=[broken], + sample_payloads=[load_db.sample_row(s)["payload"] for s in samples.values()], + ) + with self.assertRaises(load.LoadError) as ctx: + load_db.load_rules_from_db(conn, samples=load_db.load_samples_from_db(conn)) + self.assertIn("l001", str(ctx.exception)) + + def test_duplicate_rule_id_is_rejected(self): + samples, rules = _file_corpus() + payload = load_db.rule_row(rules["l001"])["payload"] + conn = _FakeConn(rule_payloads=[payload, payload], sample_payloads=[]) + with self.assertRaisesRegex(load.LoadError, "重复"): + load_db.load_rules_from_db(conn) + + def test_row_projection_sha_matches_canonical_payload(self): + samples, rules = _file_corpus() + for rule in rules.values(): + row = load_db.rule_row(rule) + self.assertEqual(row["content_sha256"], load_db.canonical_sha(rule)) + self.assertEqual(row["payload"], rule) + for sample in samples.values(): + row = load_db.sample_row(sample) + self.assertEqual(row["content_sha256"], load_db.canonical_sha(sample)) + self.assertEqual(row["payload"], sample) + + +if __name__ == "__main__": + unittest.main() diff --git a/humanization/tests/test_load_db_pg_smoke.py b/humanization/tests/test_load_db_pg_smoke.py new file mode 100644 index 0000000..acd729d --- /dev/null +++ b/humanization/tests/test_load_db_pg_smoke.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +"""真实 PostgreSQL 规则库冒烟:数据库权威必须与文件种子指纹一致(需显式授权)。""" +import os +import pathlib +import sys + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "src")) +sys.path.insert(0, str(PROJECT_ROOT / ".claude" / "skills" / "access-database" / "scripts")) + +from deai import load, load_db # noqa: E402 + + +def main(): + if os.getenv("MUSE_REAL_PG_RULE_SMOKE") != "1": + print("BLOCKED: set MUSE_REAL_PG_RULE_SMOKE=1 to run the real PostgreSQL rule smoke") + return 2 + + from db import connect # noqa: E402 + + file_samples = load.load_samples() + file_rules = load.load_rules(samples=file_samples) + file_version = load.rule_library_version(file_rules) + + with connect(readonly=True) as conn: + db_samples = load_db.load_samples_from_db(conn) + db_rules = load_db.load_rules_from_db(conn, samples=db_samples) + db_version = load.rule_library_version(db_rules) + + problems = [] + if db_rules != file_rules: + problems.append("规则内容与文件种子不一致") + if db_samples != file_samples: + problems.append("样例内容与文件种子不一致") + if db_version != file_version: + problems.append(f"规则库指纹不一致: db={db_version} files={file_version}") + if problems: + print("FAIL: " + "; ".join(problems)) + return 1 + active = sum(1 for rule in db_rules.values() if rule["status"] == "active") + print(f"PASS: real PostgreSQL rule smoke (rules={len(db_rules)}, active={active}, " + f"samples={len(db_samples)}, version={db_version})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/humanization/tests/test_seed_rules_db.py b/humanization/tests/test_seed_rules_db.py new file mode 100644 index 0000000..720e02a --- /dev/null +++ b/humanization/tests/test_seed_rules_db.py @@ -0,0 +1,134 @@ +#!/usr/bin/env python3 +"""humanization 种子同步工具离线测试:幂等、事件留痕、db-only 失败关闭、dry-run 不连库。""" +import contextlib +import io +import json +import pathlib +import sys +import unittest + +PROJECT_ROOT = pathlib.Path(__file__).resolve().parents[2] +sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "src")) +sys.path.insert(0, str(PROJECT_ROOT / "humanization" / "tools")) + +from deai import load, load_db # noqa: E402 +import seed_rules_db as seed_tool # noqa: E402 + + +class _Txn: + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + +class _SeedConn: + """记录写语句的假连接;现有行以 id->sha 注入。""" + + def __init__(self, existing_rules=None, existing_samples=None): + self.existing_rules = dict(existing_rules or {}) + self.existing_samples = dict(existing_samples or {}) + self.writes = [] + self.txn_count = 0 + + def transaction(self): + self.txn_count += 1 + return _Txn() + + def execute(self, sql, params=None): + normalized = " ".join(sql.split()) + if normalized.startswith("SELECT rule_id, content_sha256 FROM example_ai_flavor_rule"): + return _Rows([(rid, sha) for rid, sha in sorted(self.existing_rules.items())]) + if normalized.startswith("SELECT sample_id, content_sha256 FROM example_ai_flavor_sample"): + return _Rows([(sid, sha) for sid, sha in sorted(self.existing_samples.items())]) + kind = ( + "rule_insert" if normalized.startswith("INSERT INTO example_ai_flavor_rule (") + else "rule_update" if normalized.startswith("UPDATE example_ai_flavor_rule") + else "sample_insert" if normalized.startswith("INSERT INTO example_ai_flavor_sample") + else "sample_update" if normalized.startswith("UPDATE example_ai_flavor_sample") + else "rule_event" if normalized.startswith("INSERT INTO example_ai_flavor_rule_event") + else None + ) + if kind is None: + raise AssertionError(f"未预期的 SQL: {normalized}") + self.writes.append((kind, params)) + return _Rows([]) + + +class _Rows: + def __init__(self, rows): + self.rows = list(rows) + + def fetchall(self): + return list(self.rows) + + +def _corpus(): + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules + + +class SeedRulesDbTest(unittest.TestCase): + def test_first_seed_inserts_all_and_records_events(self): + samples, rules = _corpus() + conn = _SeedConn() + summary = seed_tool.seed(conn, rules=rules, samples=samples) + self.assertEqual(summary["rules"]["inserted"], len(rules)) + self.assertEqual(summary["samples"]["inserted"], len(samples)) + self.assertEqual(summary["rules"]["updated"], 0) + self.assertEqual(summary["events"], len(rules)) + kinds = [kind for kind, _ in conn.writes] + self.assertEqual(kinds.count("rule_insert"), len(rules)) + self.assertEqual(kinds.count("sample_insert"), len(samples)) + self.assertEqual(kinds.count("rule_event"), len(rules)) + + def test_second_seed_with_same_content_is_idempotent(self): + samples, rules = _corpus() + existing_rules = {rid: load_db.canonical_sha(rule) for rid, rule in rules.items()} + existing_samples = {sid: load_db.canonical_sha(s) for sid, s in samples.items()} + conn = _SeedConn(existing_rules=existing_rules, existing_samples=existing_samples) + summary = seed_tool.seed(conn, rules=rules, samples=samples) + self.assertEqual(summary["rules"]["unchanged"], len(rules)) + self.assertEqual(summary["samples"]["unchanged"], len(samples)) + self.assertEqual(conn.writes, []) + + def test_changed_rule_is_updated_with_event(self): + samples, rules = _corpus() + changed = dict(rules["l001"], fix_hint="更新后的修复提示") + rules = dict(rules, l001=changed) + existing_rules = {rid: "0" * 64 for rid in rules} + existing_samples = {sid: load_db.canonical_sha(s) for sid, s in samples.items()} + conn = _SeedConn(existing_rules=existing_rules, existing_samples=existing_samples) + summary = seed_tool.seed(conn, rules=rules, samples=samples) + self.assertEqual(summary["rules"]["updated"], len(rules)) + self.assertEqual(summary["samples"]["unchanged"], len(samples)) + rule_writes = [kind for kind, _ in conn.writes if kind in ("rule_update", "rule_event")] + self.assertEqual(rule_writes.count("rule_update"), len(rules)) + self.assertEqual(rule_writes.count("rule_event"), len(rules)) + + def test_db_only_rows_are_reported_and_strict_fails_closed(self): + samples, rules = _corpus() + conn = _SeedConn(existing_rules={"z999": "0" * 64}) + summary = seed_tool.seed(conn, rules=rules, samples=samples) + self.assertEqual(summary["db_only_rules"], ["z999"]) + conn_strict = _SeedConn(existing_rules={"z999": "0" * 64}) + with self.assertRaisesRegex(seed_tool.SeedError, "失败关闭"): + seed_tool.seed(conn_strict, rules=rules, samples=samples, strict=True) + + def test_dry_run_does_not_touch_database(self): + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + code = seed_tool.main(["--dry-run"]) + self.assertEqual(code, 0) + plan = json.loads(buffer.getvalue()) + self.assertEqual(plan["status"], "dry_run") + samples, rules = _corpus() + self.assertEqual(plan["rules"], len(rules)) + self.assertEqual(plan["samples"], len(samples)) + self.assertEqual(plan["library_version"], load.rule_library_version(rules)) + + +if __name__ == "__main__": + unittest.main() diff --git a/humanization/tools/seed_rules_db.py b/humanization/tools/seed_rules_db.py new file mode 100644 index 0000000..97d24e7 --- /dev/null +++ b/humanization/tools/seed_rules_db.py @@ -0,0 +1,217 @@ +#!/usr/bin/env python3 +"""humanization 规则/样例种子同步:Git YAML(迁移种子)→ muse-example 运行时权威表。 + +合同: +- 种子前先过文件侧激活门(load.load_rules(samples=...)):样例不齐的规则拒绝进库。 +- 单事务:先比对 content_sha256 生成计划,再执行 upsert;任何错误整批回滚。 +- 幂等:内容哈希一致的行跳过;内容变化的行 UPDATE,并记一条 synced 生命周期事件。 +- 数据库独有行不自动删除:默认报告清单;--strict 时失败关闭,交人工裁决。 +- --dry-run 只输出目标清单,不连接数据库。 +""" +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +TOOL_DIR = Path(__file__).resolve().parent +AGENT_ROOT = TOOL_DIR.parent.parent +for _p in ( + AGENT_ROOT / "humanization" / "src", + AGENT_ROOT / ".claude" / "skills" / "access-database" / "scripts", +): + if str(_p) not in sys.path: + sys.path.insert(0, str(_p)) + +from deai import load, load_db # noqa: E402 + + +class SeedError(ValueError): + pass + + +_RULE_INSERT = ( + "INSERT INTO example_ai_flavor_rule " + "(rule_id, name, layer, carrier_scope, default_disposition, status, version, " + "trigger_json, carve_out, function_check, sample_refs, case_card_ids, " + "fix_hint, evidence, payload, content_sha256, creator, tenant_id) " + "VALUES (%s, %s, %s, %s, %s, %s, %s, %s::jsonb, %s::jsonb, %s::jsonb, %s::jsonb, %s::jsonb, " + "%s, %s, %s::jsonb, %s, %s, %s)" +) +_RULE_UPDATE = ( + "UPDATE example_ai_flavor_rule SET name=%s, layer=%s, carrier_scope=%s, " + "default_disposition=%s, status=%s, version=%s, trigger_json=%s::jsonb, carve_out=%s::jsonb, " + "function_check=%s::jsonb, sample_refs=%s::jsonb, case_card_ids=%s::jsonb, fix_hint=%s, " + "evidence=%s, payload=%s::jsonb, content_sha256=%s, updater=%s " + "WHERE tenant_id=%s AND rule_id=%s" +) +_SAMPLE_INSERT = ( + "INSERT INTO example_ai_flavor_sample " + "(sample_id, sample_type, carrier, source, source_license, rules, text, note, " + "case_card_id, source_ref, payload, content_sha256, creator, tenant_id) " + "VALUES (%s, %s, %s, %s, %s, %s::jsonb, %s, %s, %s, %s, %s::jsonb, %s, %s, %s)" +) +_SAMPLE_UPDATE = ( + "UPDATE example_ai_flavor_sample SET sample_type=%s, carrier=%s, source=%s, " + "source_license=%s, rules=%s::jsonb, text=%s, note=%s, case_card_id=%s, source_ref=%s, " + "payload=%s::jsonb, content_sha256=%s, updater=%s " + "WHERE tenant_id=%s AND sample_id=%s" +) +_RULE_EVENT_INSERT = ( + "INSERT INTO example_ai_flavor_rule_event " + "(rule_id, event, rule_version, status, approver, note, content_sha256, creator, tenant_id) " + "VALUES (%s, 'synced', %s, %s, '', %s, %s, %s, %s)" +) + + +def _dumps(value) -> str: + return json.dumps(value, ensure_ascii=False) + + +def _rule_insert_params(row: dict, creator: str, tenant_id: int) -> tuple: + return ( + row["rule_id"], row["name"], row["layer"], row["carrier_scope"], + row["default_disposition"], row["status"], row["version"], + _dumps(row["trigger_json"]), _dumps(row["carve_out"]), _dumps(row["function_check"]), + _dumps(row["sample_refs"]), _dumps(row["case_card_ids"]), + row["fix_hint"], row["evidence"], _dumps(row["payload"]), row["content_sha256"], + creator, tenant_id, + ) + + +def _rule_update_params(row: dict, creator: str, tenant_id: int) -> tuple: + return ( + row["name"], row["layer"], row["carrier_scope"], row["default_disposition"], + row["status"], row["version"], + _dumps(row["trigger_json"]), _dumps(row["carve_out"]), _dumps(row["function_check"]), + _dumps(row["sample_refs"]), _dumps(row["case_card_ids"]), + row["fix_hint"], row["evidence"], _dumps(row["payload"]), row["content_sha256"], + creator, tenant_id, row["rule_id"], + ) + + +def _sample_insert_params(row: dict, creator: str, tenant_id: int) -> tuple: + return ( + row["sample_id"], row["sample_type"], row["carrier"], row["source"], + row["source_license"], _dumps(row["rules"]), row["text"], row["note"], + row["case_card_id"], row["source_ref"], _dumps(row["payload"]), + row["content_sha256"], creator, tenant_id, + ) + + +def _sample_update_params(row: dict, creator: str, tenant_id: int) -> tuple: + return ( + row["sample_type"], row["carrier"], row["source"], row["source_license"], + _dumps(row["rules"]), row["text"], row["note"], row["case_card_id"], + row["source_ref"], _dumps(row["payload"]), row["content_sha256"], + creator, tenant_id, row["sample_id"], + ) + + +def seed(conn, *, rules: dict, samples: dict, creator: str = "1", + tenant_id: int = load_db.TENANT_ID, strict: bool = False) -> dict: + """单事务同步规则与样例;调用方负责事务边界(真实路径用 conn.transaction())。""" + summary = { + "rules": {"inserted": 0, "updated": 0, "unchanged": 0}, + "samples": {"inserted": 0, "updated": 0, "unchanged": 0}, + "db_only_rules": [], + "db_only_samples": [], + "events": 0, + } + + existing_rules = { + row[0]: {"content_sha256": row[1]} + for row in conn.execute( + "SELECT rule_id, content_sha256 FROM example_ai_flavor_rule " + "WHERE tenant_id = %s AND deleted = FALSE", (tenant_id,) + ).fetchall() + } + existing_samples = { + row[0]: row[1] + for row in conn.execute( + "SELECT sample_id, content_sha256 FROM example_ai_flavor_sample " + "WHERE tenant_id = %s AND deleted = FALSE", (tenant_id,) + ).fetchall() + } + + for rule in sorted(rules.values(), key=lambda item: item["id"]): + row = load_db.rule_row(rule) + current = existing_rules.get(row["rule_id"]) + if current is None: + conn.execute(_RULE_INSERT, _rule_insert_params(row, creator, tenant_id)) + conn.execute(_RULE_EVENT_INSERT, ( + row["rule_id"], row["version"], row["status"], + "seed insert", row["content_sha256"], creator, tenant_id, + )) + summary["rules"]["inserted"] += 1 + summary["events"] += 1 + elif current["content_sha256"] != row["content_sha256"]: + conn.execute(_RULE_UPDATE, _rule_update_params(row, creator, tenant_id)) + conn.execute(_RULE_EVENT_INSERT, ( + row["rule_id"], row["version"], row["status"], + "seed update", row["content_sha256"], creator, tenant_id, + )) + summary["rules"]["updated"] += 1 + summary["events"] += 1 + else: + summary["rules"]["unchanged"] += 1 + + for sample in sorted(samples.values(), key=lambda item: item["id"]): + row = load_db.sample_row(sample) + current_sha = existing_samples.get(row["sample_id"]) + if current_sha is None: + conn.execute(_SAMPLE_INSERT, _sample_insert_params(row, creator, tenant_id)) + summary["samples"]["inserted"] += 1 + elif current_sha != row["content_sha256"]: + conn.execute(_SAMPLE_UPDATE, _sample_update_params(row, creator, tenant_id)) + summary["samples"]["updated"] += 1 + else: + summary["samples"]["unchanged"] += 1 + + summary["db_only_rules"] = sorted(set(existing_rules) - set(rules)) + summary["db_only_samples"] = sorted(set(existing_samples) - set(samples)) + if strict and (summary["db_only_rules"] or summary["db_only_samples"]): + raise SeedError( + "数据库存在 YAML 种子之外的行,失败关闭: rules=" + + ",".join(summary["db_only_rules"]) + "; samples=" + ",".join(summary["db_only_samples"]) + ) + return summary + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="humanization 规则/样例种子同步(YAML → muse-example)") + parser.add_argument("--dry-run", action="store_true", help="只输出目标清单,不连接数据库") + parser.add_argument("--strict", action="store_true", help="数据库独有行存在时失败关闭") + parser.add_argument("--tenant", type=int, default=load_db.TENANT_ID) + args = parser.parse_args(argv) + + try: + # 种子前先过激活门:样例不齐的 active 规则在文件侧就被拒绝 + samples = load.load_samples() + rules = load.load_rules(samples=samples) + if args.dry_run: + plan = { + "status": "dry_run", + "rules": len(rules), + "samples": len(samples), + "rule_ids": sorted(rules), + "library_version": load.rule_library_version(rules), + } + print(json.dumps(plan, ensure_ascii=False)) + return 0 + from db import connect # noqa: E402 + + with connect() as conn: + with conn.transaction(): + summary = seed(conn, rules=rules, samples=samples, + tenant_id=args.tenant, strict=args.strict) + print(json.dumps({"status": "seeded", **summary}, ensure_ascii=False)) + return 0 + except (SeedError, load.LoadError, ValueError, OSError) as exc: + print(f"SEED_RULES_FAILED: {exc}") + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py b/tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py index 27ac736..05a8660 100644 --- a/tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py +++ b/tests/skills/diagnose-ai-flavor/test_diagnose_ai_flavor.py @@ -66,11 +66,20 @@ class DiagnosisContractTest(unittest.TestCase): diag.persist_diagnosis(artifact, text=AI_FLAVOR_TEXT) def test_cli_default_persists_detection(self): + # 默认(非 --offline)模式生产读数据库规则库;离线测试补丁文件库接缝,不触真实连接。 + from deai import load + + def _file_library(from_db=False): + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return rules, diag.rule_library_version(rules) + with tempfile.TemporaryDirectory() as tmp: text_path = pathlib.Path(tmp) / "text.txt" text_path.write_text(AI_FLAVOR_TEXT, encoding="utf-8") output = pathlib.Path(tmp) / "artifact.json" - with patch.object(diag, "persist_diagnosis", return_value={"run_id": "diag-x"}) as persist: + with patch.object(diag, "persist_diagnosis", return_value={"run_id": "diag-x"}) as persist, \ + patch.object(diag, "load_active_library", side_effect=_file_library): code = diag.main(["run", "--text-file", str(text_path), "--work-ref", "synthetic:demo", "--output", str(output)]) self.assertEqual(code, 0) diff --git a/tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py b/tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py index b848d27..cd78a9e 100644 --- a/tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py +++ b/tests/skills/prevent-ai-flavor/test_prevent_ai_flavor.py @@ -61,13 +61,22 @@ class PreventionContractTest(unittest.TestCase): prev.build_prevention_contract("synthetic:demo", voice_ledger=bad) def test_cli_default_persists_run(self): + # 默认(非 --offline)模式生产读数据库规则库;离线测试补丁文件库接缝,不触真实连接。 import argparse # noqa: F401 (确认 CLI 依赖可导入) import tempfile import json + from deai import load + + def _file_library(load_database): + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules, prev.rule_library_version(rules), "database" + with tempfile.TemporaryDirectory() as tmp: output = pathlib.Path(tmp) / "contract.json" with patch.object(prev, "persist_prevention", return_value={"run_id": "prev-x"}) as persist, \ - patch.object(prev, "_load_db_ledger", return_value=None): + patch.object(prev, "_load_db_ledger", return_value=None), \ + patch.object(prev, "load_runtime_library", side_effect=_file_library): code = prev.main(["--work-ref", "synthetic:demo", "--output", str(output)]) self.assertEqual(code, 0) persist.assert_called_once() diff --git a/tests/skills/revise-ai-flavor/test_revise_ai_flavor.py b/tests/skills/revise-ai-flavor/test_revise_ai_flavor.py index 1acfbe0..2c24b50 100644 --- a/tests/skills/revise-ai-flavor/test_revise_ai_flavor.py +++ b/tests/skills/revise-ai-flavor/test_revise_ai_flavor.py @@ -119,6 +119,14 @@ class RevisionContractTest(unittest.TestCase): self.assertTrue(record["hard_gate"]["checks"]["fact_delta"]["failures"]) def _run_cli(self, *, offline: bool, persist_return=None): + # 默认(非 --offline)模式生产读数据库规则库;离线测试补丁文件库接缝,不触真实连接。 + from deai import load + + def _file_library(from_db): + samples = load.load_samples() + rules = load.load_rules(samples=samples) + return samples, rules, load.rule_library_version(rules) + artifact = _artifact(CLEAN_TEXT) patches = [_delete_patch(artifact, "值得注意的是", "值得注意的是,")] with tempfile.TemporaryDirectory() as tmp: @@ -137,7 +145,8 @@ class RevisionContractTest(unittest.TestCase): pairwise_path.write_text(json.dumps(PAIRWISE, ensure_ascii=False), encoding="utf-8") output = root / "report.json" with patch.object(rev, "persist_revision", return_value=persist_return) as persist, \ - patch.object(rev, "load_current_baseline", return_value=None): + patch.object(rev, "load_current_baseline", return_value=None), \ + patch.object(rev, "load_revision_library", side_effect=_file_library): argv = [ "--text-file", str(text_path), "--artifact", str(artifact_path), "--patches", str(patches_path), "--task-contract", str(contract_path),