1080 lines
52 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""AI 味案例卡的确定性发现、自动落库与候选规则构造。
扫描/回填/创作反馈命令默认在检测完成后自动写入 agent-example 的
``muse-example``;``--offline`` 只用于明确的离线合同测试或恢复准备。
脚本不调用模型;语义判断由人工或独立评审补上。
"""
from __future__ import annotations
import argparse
import copy
import hashlib
import json
import re
import sys
from collections import Counter
from datetime import date
from pathlib import Path
from typing import Iterable
import yaml
_AGENT_ROOT = Path(__file__).resolve().parents[4]
_HUMANIZATION_SRC = _AGENT_ROOT / "humanization" / "src"
if str(_HUMANIZATION_SRC) not in sys.path:
sys.path.insert(0, str(_HUMANIZATION_SRC))
# 作为 CLI 执行时也注册稳定模块名,自动落库模块复用同一份合同异常类型,
# 避免失败路径被重复 import 变成未捕获 traceback。
if __name__ == "__main__":
sys.modules.setdefault("capture_cases", sys.modules[__name__])
SCHEMA_VERSION = "ai-flavor-case-v1"
CARD_TYPE = "ai_flavor_case"
REVALIDATION_SCHEMA_VERSION = "ai-flavor-revalidation-v1"
STATES = {"shadow", "canonical", "rejected", "archived"}
LABELS = {"unclassified", "sf", "snf", "boundary", "regression"}
LAYERS = {"mechanical", "lexical", "structural", "density", "semantic", "unknown"}
CARRIERS = {"narration", "dialogue", "monologue", "in_text_carrier", "mixed", "unknown"}
CAPTURE_MODES = {"backfill", "live_feedback", "synthetic", "review_import"}
REVALIDATION_STATUSES = {"verified", "stale", "unavailable", "card_mismatch"}
LICENSES = {"owned", "licensed", "public_domain", "research_only", "unauthorized", "synthetic"}
NON_REPO_LICENSES = {"research_only", "unauthorized"}
STOREABLE_LICENSES = LICENSES - NON_REPO_LICENSES
class CaseCardError(ValueError):
"""案例卡合同或升级门禁失败。"""
PATTERNS = (
{
"key": "lexical.meta_disclaimer",
"layer": "lexical",
"pattern": r"(?:值得注意的是|值得一提的是|不言而喻|众所周知|换句话说)",
"diagnosis": "叙述者元话语可能没有新增信息;必须检查是否承担声线或节奏功能。",
},
{
"key": "lexical.stock_micro_expression",
"layer": "lexical",
"pattern": r"(?:嘴角(?:微微|轻轻|悄然)?(?:上扬|勾起)|眼中闪过(?:一丝|一抹)?(?:光|精光|异彩)|眸光(?:微闪|深邃))",
"diagnosis": "库存微表情候选;必须检查是否是角色签名动作或场景独有反应。",
},
{
"key": "lexical.abstract_atmosphere",
"layer": "lexical",
"pattern": r"(?:一股[^。!?\n]{0,24}气息[^。!?\n]{0,24}(?:弥漫|袭来|扑面而来)|空气仿佛凝固)",
"diagnosis": "抽象气氛候选;必须检查感官细节、因果和场面功能,不能看到词就删除。",
},
)
def _sha256_bytes(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def text_sha256(text: str) -> str:
return _sha256_bytes(text.encode("utf-8"))
def _resolve_source_path(source_ref: str, source_root: Path | None = None) -> Path | None:
"""把脱敏来源引用解析为受控本地文件;URI 和无法落到文件的引用返回 None。"""
if not source_ref or "://" in source_ref:
return None
ref = Path(source_ref).expanduser()
if ref.is_absolute():
resolved = ref.resolve()
if source_root is not None:
try:
resolved.relative_to(source_root.expanduser().resolve())
except ValueError:
return None
return resolved
if source_root is None:
return None
root = source_root.expanduser().resolve()
candidates = [root / ref]
# inventory 默认把 source_root_ref(例如“小说清单”)写进 source_ref;
# 调用方也可以直接把 source_root 指到该目录,因此兼容两种入口。
if ref.parts and ref.parts[0] in {root.name, root.parent.name}:
candidates.append(root / Path(*ref.parts[1:]))
for candidate in candidates:
resolved = candidate.resolve()
try:
resolved.relative_to(root)
except ValueError:
continue
if resolved.is_file():
return resolved
return None
def _source_snapshot(*, source_ref: str, source_root: Path | None = None,
source_path: Path | None = None, source_text: str | None = None) -> dict:
"""读取一次来源,供单卡和批量重验证复用。"""
if source_text is not None and source_path is not None:
raise CaseCardError("source_text 与 source_path 不能同时提供")
if source_text is not None:
return {"status": "loaded", "actual_sha256": text_sha256(source_text), "text": source_text, "reason": ""}
path = source_path.expanduser().resolve() if source_path is not None else _resolve_source_path(source_ref, source_root)
if path is None:
return {"status": "unavailable", "actual_sha256": None, "text": None, "reason": "source_ref_not_file"}
try:
raw = path.read_bytes()
except OSError:
return {"status": "unavailable", "actual_sha256": None, "text": None, "reason": "source_not_found"}
actual_sha256 = _sha256_bytes(raw)
try:
text = raw.decode("utf-8")
except UnicodeError:
return {"status": "unavailable", "actual_sha256": actual_sha256, "text": None,
"reason": "source_not_utf8"}
return {"status": "loaded", "actual_sha256": actual_sha256, "text": text, "reason": ""}
def _card_anchor_reason(card: dict, text: str) -> str | None:
"""在全文哈希相同后,再检查卡片的位置和命中是否仍自洽。"""
source = card["source"]
location = source.get("location") or {}
start, end = location.get("char_start"), location.get("char_end")
if not isinstance(start, int) or not isinstance(end, int) or start < 0 or end < start or end > len(text):
return "card_anchor_out_of_range"
if text_sha256(text[start:end]) != source["excerpt_sha256"]:
return "card_excerpt_hash_mismatch"
surface_location = source.get("surface_location")
if surface_location:
surface_start, surface_end = surface_location.get("char_start"), surface_location.get("char_end")
expected_surface = card.get("observation", {}).get("surface", "")
if (not isinstance(surface_start, int) or not isinstance(surface_end, int)
or surface_start < 0 or surface_end < surface_start or surface_end > len(text)):
return "card_surface_anchor_out_of_range"
if text[surface_start:surface_end] != expected_surface:
return "card_surface_mismatch"
return None
def _revalidation_result(card: dict, snapshot: dict, *, checked_on: str) -> dict:
source = card["source"]
result = {
"card_id": card["id"],
"work_ref": source.get("work_ref", ""),
"source_ref": source.get("source_ref", ""),
"expected_source_sha256": source["source_sha256"],
"actual_source_sha256": snapshot.get("actual_sha256"),
"status": snapshot["status"],
"reason": snapshot.get("reason", ""),
"checked_on": checked_on,
}
if snapshot["status"] == "loaded":
if snapshot["actual_sha256"] != source["source_sha256"]:
result["status"] = "stale"
result["reason"] = "source_hash_mismatch"
else:
reason = _card_anchor_reason(card, snapshot["text"])
if reason:
result["status"] = "card_mismatch"
result["reason"] = reason
else:
result["status"] = "verified"
result["reason"] = "source_hash_and_anchor_match"
return result
def revalidate_card(card: dict, *, source_root: Path | None = None, source_path: Path | None = None,
source_text: str | None = None, checked_on: str | None = None) -> dict:
"""重新读取来源并返回不可变验证结果;不会自动改写案例卡状态。"""
validate_card(card)
snapshot = _source_snapshot(source_ref=card["source"].get("source_ref", ""),
source_root=source_root, source_path=source_path, source_text=source_text)
return _revalidation_result(card, snapshot, checked_on=checked_on or date.today().isoformat())
def revalidate_cards(cards: list[dict], *, source_root: Path | None = None,
checked_on: str | None = None) -> list[dict]:
"""批量重验证;同一来源只读取一次。"""
checked_on = checked_on or date.today().isoformat()
snapshots: dict[str, dict] = {}
results = []
for card in cards:
validate_card(card)
source_ref = card["source"].get("source_ref", "")
if source_ref not in snapshots:
snapshots[source_ref] = _source_snapshot(source_ref=source_ref, source_root=source_root)
results.append(_revalidation_result(card, snapshots[source_ref], checked_on=checked_on))
return results
def build_revalidation_report(cards: list[dict], *, source_root: Path | None = None,
checked_on: str | None = None,
source_texts: dict[str, str] | None = None) -> dict:
if source_texts is None:
results = revalidate_cards(cards, source_root=source_root, checked_on=checked_on)
else:
results = []
for card in cards:
if card["id"] not in source_texts:
raise CaseCardError(f"缺少案例卡 {card['id']} 的反馈正文")
results.append(revalidate_card(card, source_text=source_texts[card["id"]], checked_on=checked_on))
counts = Counter(result["status"] for result in results)
works: dict[tuple[str, str], dict] = {}
for result in results:
key = (result["work_ref"], result["source_ref"])
work = works.setdefault(key, {
"work_ref": result["work_ref"], "source_ref": result["source_ref"],
"expected_source_sha256": result["expected_source_sha256"],
"actual_source_sha256": result["actual_source_sha256"], "card_count": 0,
"status_counts": {},
})
work["card_count"] += 1
work["status_counts"][result["status"]] = work["status_counts"].get(result["status"], 0) + 1
if work["actual_source_sha256"] is None and result["actual_source_sha256"] is not None:
work["actual_source_sha256"] = result["actual_source_sha256"]
return {
"schema_version": REVALIDATION_SCHEMA_VERSION,
"checked_on": checked_on or date.today().isoformat(),
"source_root_ref": source_root.name if source_root else None,
"usable": bool(results) and all(result["status"] == "verified" for result in results),
"totals": {"cards": len(results), **{status: counts.get(status, 0) for status in sorted(REVALIDATION_STATUSES)}},
"works": list(sorted(works.values(), key=lambda item: (item["work_ref"], item["source_ref"]))),
"cards": results,
}
def _verification_results(verification) -> dict[str, dict]:
if not isinstance(verification, dict):
return {}
if verification.get("card_id"):
return {verification["card_id"]: verification}
items = verification.get("cards")
if not isinstance(items, list):
return {}
return {item.get("card_id"): item for item in items if isinstance(item, dict) and item.get("card_id")}
def require_verified(card: dict, verification) -> dict:
"""升级或消费前的 fail-closed 门:只接受本卡的 verified 回执。"""
validate_card(card)
result = _verification_results(verification).get(card["id"])
if (not result or result.get("status") != "verified"
or result.get("expected_source_sha256") != card["source"]["source_sha256"]
or result.get("actual_source_sha256") != card["source"]["source_sha256"]):
status = result.get("status") if result else "missing"
raise CaseCardError(f"案例卡 {card['id']} 来源未通过重验证: {status}")
return result
def require_verified_cards(cards: list[dict], verification) -> list[dict]:
return [require_verified(card, verification) for card in cards]
def _line_col(text: str, offset: int) -> tuple[int, int]:
line_start = text.count("\n", 0, offset) + 1
last_newline = text.rfind("\n", 0, offset)
return line_start, offset - (last_newline + 1)
def _location(text: str, start: int, end: int) -> dict:
line_start, col_start = _line_col(text, start)
line_end, col_end = _line_col(text, end)
return {
"line_start": line_start,
"line_end": line_end,
"char_start": start,
"char_end": end,
"column_start": col_start,
"column_end": col_end,
}
def _context(text: str, start: int, end: int) -> str:
"""取命中所在行及相邻一行,供有权保存的来源做人工复核。"""
line_start = text.rfind("\n", 0, start) + 1
line_end = text.find("\n", end)
if line_end < 0:
line_end = len(text)
previous_start = text.rfind("\n", 0, max(0, line_start - 1)) + 1
next_end = text.find("\n", line_end + 1)
if next_end < 0:
next_end = len(text)
return text[previous_start:next_end].strip("\n")
def _evidence_window(text: str, start: int, end: int, max_chars: int = 180) -> tuple[int, int]:
"""取命中词所在句的有限窗口,避免样例只有触发词或吞入整章。"""
left = max(text.rfind("。", 0, start), text.rfind("!", 0, start),
text.rfind("?", 0, start), text.rfind("\n", 0, start)) + 1
right_candidates = [p for p in (text.find("。", end), text.find("!", end),
text.find("?", end), text.find("\n", end)) if p >= 0]
right = min(right_candidates, default=len(text))
if right_candidates:
right += 1
if right - left > max_chars:
left = max(0, start - max_chars // 2)
right = min(len(text), max(end, start + max_chars // 2))
if right - left > max_chars:
right = left + max_chars
return left, right
def _require_string(value, field: str, *, allow_empty: bool = False) -> str:
if not isinstance(value, str) or (not allow_empty and not value.strip()):
raise CaseCardError(f"{field} 必须是{'可为空的' if allow_empty else ''}字符串")
return value
def validate_card(card: dict) -> dict:
"""验证单卡的字段和跨字段不变量。"""
if not isinstance(card, dict):
raise CaseCardError("卡片必须是对象")
for key in ("schema_version", "id", "card_type", "state", "label", "layer", "carrier", "capture_mode", "source", "observation"):
if key not in card:
raise CaseCardError(f"卡片缺少 {key}")
if card["schema_version"] != SCHEMA_VERSION:
raise CaseCardError(f"schema_version 必须为 {SCHEMA_VERSION}")
_require_string(card["id"], "id")
if not card["id"].startswith("case-"):
raise CaseCardError("id 必须以 case- 开头")
if card["card_type"] != CARD_TYPE:
raise CaseCardError("card_type 必须为 ai_flavor_case")
for field, allowed in (("state", STATES), ("label", LABELS), ("layer", LAYERS), ("carrier", CARRIERS), ("capture_mode", CAPTURE_MODES)):
if card[field] not in allowed:
raise CaseCardError(f"{field} 取值非法: {card[field]!r}")
source = card["source"]
if not isinstance(source, dict):
raise CaseCardError("source 必须是对象")
for key in ("kind", "license", "source_sha256", "excerpt_sha256", "location"):
if key not in source:
raise CaseCardError(f"source 缺少 {key}")
if source["kind"] not in {"existing_work", "creation_feedback", "synthetic", "public_domain"}:
raise CaseCardError(f"source.kind 非法: {source['kind']!r}")
if source["license"] not in LICENSES:
raise CaseCardError(f"source.license 非法: {source['license']!r}")
for key in ("source_sha256", "excerpt_sha256"):
if not re.fullmatch(r"[0-9a-f]{64}", source[key]):
raise CaseCardError(f"source.{key} 必须是 64 位小写 SHA-256")
if source["kind"] == "existing_work" and not source.get("work_ref"):
raise CaseCardError("existing_work 必须有 work_ref")
if card["capture_mode"] == "live_feedback" and source["kind"] != "creation_feedback":
raise CaseCardError("live_feedback 的 source.kind 必须为 creation_feedback")
location = source["location"]
if not isinstance(location, dict):
raise CaseCardError("source.location 必须是对象")
for key in ("line_start", "line_end", "char_start", "char_end"):
if not isinstance(location.get(key), int) or location[key] < 0:
raise CaseCardError(f"source.location.{key} 必须是非负整数")
if location["line_end"] < location["line_start"] or location["char_end"] < location["char_start"]:
raise CaseCardError("source.location 结束位置不能早于开始位置")
_require_string(card.get("excerpt", ""), "excerpt", allow_empty=True)
_require_string(card.get("context", ""), "context", allow_empty=True)
if source["license"] in NON_REPO_LICENSES:
if card.get("excerpt") or card.get("context"):
raise CaseCardError("未授权/研究限定来源必须 hash-only")
if card["state"] != "shadow":
raise CaseCardError("未授权/研究限定来源只能保持 shadow")
if card["state"] == "canonical":
if source["license"] not in STOREABLE_LICENSES:
raise CaseCardError("canonical 卡必须来自可保存的授权来源")
if not card.get("excerpt"):
raise CaseCardError("canonical 卡必须有可审阅片段")
if card["label"] == "unclassified":
raise CaseCardError("canonical 卡必须有评审标签")
review = card.get("review")
if not isinstance(review, dict) or not review.get("reviewer") or not review.get("note"):
raise CaseCardError("canonical 卡必须有 review.reviewer 和 review.note")
if card.get("excerpt") and source["excerpt_sha256"] != text_sha256(card["excerpt"]):
raise CaseCardError("excerpt_sha256 与 excerpt 不一致")
observation = card["observation"]
if not isinstance(observation, dict):
raise CaseCardError("observation 必须是对象")
for key in ("pattern_key", "surface", "diagnosis", "function_check", "risk_if_changed", "suggested_action"):
if key not in observation:
raise CaseCardError(f"observation 缺少 {key}")
for key in ("pattern_key", "surface", "diagnosis", "risk_if_changed", "suggested_action"):
_require_string(observation[key], f"observation.{key}")
if not isinstance(observation["function_check"], list) or not observation["function_check"]:
raise CaseCardError("observation.function_check 必须是非空数组")
return card
def _card_id(source_sha: str, start: int, end: int, pattern_key: str) -> str:
raw = f"{source_sha}:{start}:{end}:{pattern_key}".encode("utf-8")
return "case-" + _sha256_bytes(raw)[:20]
def build_case_card(*, text: str, source_sha256: str, source_kind: str, source_license: str,
source_ref: str, work_ref: str | None, start: int, end: int,
pattern: dict, capture_mode: str = "backfill", feedback: dict | None = None,
excerpt_start: int | None = None, excerpt_end: int | None = None) -> dict:
if source_sha256 != text_sha256(text):
raise CaseCardError("source_sha256 与原始正文不一致")
if start < 0 or end < start or end > len(text):
raise CaseCardError("surface 命中位置越界")
evidence_start = start if excerpt_start is None else excerpt_start
evidence_end = end if excerpt_end is None else excerpt_end
if evidence_start < 0 or evidence_end < evidence_start or evidence_end > len(text):
raise CaseCardError("evidence 位置越界")
excerpt = text[evidence_start:evidence_end]
stored = source_license in STOREABLE_LICENSES
source = {
"kind": source_kind,
"license": source_license,
"source_sha256": source_sha256,
"excerpt_sha256": text_sha256(excerpt),
"source_ref": source_ref,
"location": _location(text, evidence_start, evidence_end),
"surface_location": _location(text, start, end),
}
if work_ref:
source["work_ref"] = work_ref
card = {
"schema_version": SCHEMA_VERSION,
"id": _card_id(source_sha256, start, end, pattern["key"]),
"card_type": CARD_TYPE,
"state": "shadow",
"label": "unclassified",
"layer": pattern["layer"],
"carrier": "unknown",
"capture_mode": capture_mode,
"excerpt": excerpt if stored else "",
"context": _context(text, start, end) if stored else "",
"source": source,
"observation": {
"pattern_key": pattern["key"],
"surface": text[start:end],
"diagnosis": pattern["diagnosis"],
"function_check": ["是否承担具体叙事功能", "是否是角色/场内载体的有意声线"],
"risk_if_changed": "未经上下文复核直接删除可能损失人物声音、伏笔或节奏。",
"suggested_action": "保留 shadow,补充上下文后再标注 sf/snf/boundary。",
},
}
if feedback is not None:
card["feedback"] = copy.deepcopy(feedback)
return validate_card(card)
def capture_file(path: Path, *, source_license: str = "research_only", source_kind: str = "existing_work",
work_ref: str | None = None, patterns: Iterable[dict] = PATTERNS,
max_cards: int = 500) -> list[dict]:
if isinstance(max_cards, bool) or max_cards <= 0:
raise CaseCardError("max_cards 必须为正整数")
if source_license not in LICENSES:
raise CaseCardError(f"不支持的来源许可: {source_license}")
if source_kind not in {"existing_work", "public_domain", "synthetic"}:
raise CaseCardError("文件扫描 source_kind 必须是 existing_work/public_domain/synthetic")
raw = path.read_bytes()
text = raw.decode("utf-8")
source_sha = _sha256_bytes(raw)
cards = []
seen = set()
matches = []
for pattern in patterns:
matches.extend((match.start(), pattern["key"], pattern, match)
for match in re.finditer(pattern["pattern"], text, flags=re.MULTILINE))
for _start, _key, pattern, match in sorted(matches, key=lambda item: (item[0], item[1])):
evidence_start, evidence_end = _evidence_window(text, match.start(), match.end())
card = build_case_card(
text=text, source_sha256=source_sha, source_kind=source_kind,
source_license=source_license, source_ref=str(path), work_ref=work_ref,
start=match.start(), end=match.end(), pattern=pattern,
excerpt_start=evidence_start, excerpt_end=evidence_end,
)
if card["id"] not in seen:
cards.append(card)
seen.add(card["id"])
if len(cards) >= max_cards:
return cards
return cards
def capture_feedback(text: str, *, work_ref: str, run_ref: str, issue: str,
source_license: str = "owned", location: dict | None = None) -> dict:
if not text:
raise CaseCardError("反馈正文不能为空")
if not work_ref or not run_ref or not issue:
raise CaseCardError("反馈必须有 work_ref、run_ref 和 issue")
pattern = {
"key": "feedback.manual_observation",
"layer": "unknown",
"diagnosis": "创作反馈待人工归类,不把一次事故直接固化为规则。",
}
card = build_case_card(
text=text, source_sha256=text_sha256(text), source_kind="creation_feedback",
source_license=source_license, source_ref=run_ref, work_ref=work_ref,
start=0, end=len(text), pattern=pattern, capture_mode="live_feedback",
feedback={"run_ref": run_ref, "issue": issue, "candidate_sha256": text_sha256(text)},
)
card["source"]["location"] = location or {"line_start": 1, "line_end": text.count("\n") + 1,
"char_start": 0, "char_end": len(text),
"column_start": 0, "column_end": 0}
return validate_card(card)
def annotate_card(card: dict, *, label: str, carrier: str = "unknown", reviewer: str = "", note: str = "") -> dict:
validate_card(card)
if card["state"] != "shadow":
raise CaseCardError("只有 shadow 卡可以标注")
if label not in LABELS - {"unclassified"}:
raise CaseCardError(f"标注标签非法: {label}")
out = copy.deepcopy(card)
out["label"] = label
out["carrier"] = carrier
if reviewer or note:
out["review"] = {"reviewer": reviewer, "note": note}
return validate_card(out)
def confirm_card(card: dict, *, reviewer: str, note: str, verification=None) -> dict:
validate_card(card)
if card["source"]["license"] in NON_REPO_LICENSES:
raise CaseCardError("hash-only 卡不能在仓库内确认")
require_verified(card, verification)
out = copy.deepcopy(card)
out["state"] = "canonical"
out["review"] = {"reviewer": reviewer, "note": note}
return validate_card(out)
def project_sample(card: dict, *, verification=None) -> dict:
validate_card(card)
if card["state"] != "canonical":
raise CaseCardError("只有 canonical 卡可以投影样例")
require_verified(card, verification)
if card["carrier"] == "unknown":
raise CaseCardError("canonical 卡投影样例前必须确认 carrier,不能用 unknown")
source_map = {
"owned": "hand_written",
"licensed": "licensed",
"public_domain": "public_domain",
"synthetic": "synthetic",
}
return {
"id": "sample-" + card["id"],
"type": card["label"],
"rules": list(card.get("rule_candidate_ids", [])),
"carrier": card["carrier"],
"source": source_map[card["source"]["license"]],
"text": card["excerpt"],
"note": card["observation"]["diagnosis"],
"case_card_id": card["id"],
"source_ref": card["source"].get("source_ref", ""),
"source_license": card["source"]["license"],
}
def propose_rule(cards: list[dict], *, rule_id: str, name: str, fix_hint: str,
verification=None, layer: str = "lexical", carrier_scope: str = "all",
trigger: dict | None = None, carve_out: list[str] | None = None,
function_check: list[str] | None = None) -> dict:
if not cards:
raise CaseCardError("规则候选至少需要一张案例卡")
for card in cards:
validate_card(card)
# 候选仍可保持 candidate,但其证据必须先证明对应来源没有漂移。
require_verified_cards(cards, verification)
works = {card["source"].get("work_ref") for card in cards if card["source"].get("work_ref")}
if len(works) < 2:
raise CaseCardError("规则候选至少需要两个不同来源作品")
labels = {card["label"] for card in cards}
if "sf" not in labels or not labels.intersection({"snf", "boundary", "regression"}):
raise CaseCardError("规则候选必须同时有 sf 与 snf/boundary/regression 证据")
if layer not in LAYERS - {"unknown"}:
raise CaseCardError(f"规则 layer 非法: {layer}")
if carrier_scope not in {"narration", "dialogue", "monologue", "in_text_carrier", "all"}:
raise CaseCardError(f"规则 carrier_scope 非法: {carrier_scope}")
trigger = copy.deepcopy(trigger or {"type": "model_judgment", "criteria": "待独立功能判断"})
trigger_type = trigger.get("type")
if trigger_type == "regex" and not trigger.get("pattern"):
raise CaseCardError("regex 候选必须提供 pattern")
if trigger_type == "handler" and not trigger.get("handler"):
raise CaseCardError("handler 候选必须提供 handler")
if trigger_type == "density" and (
not trigger.get("pattern") or not trigger.get("window_chars") or not trigger.get("min_hits")):
raise CaseCardError("density 候选必须提供 pattern/window_chars/min_hits")
if trigger_type == "model_judgment" and not trigger.get("criteria"):
raise CaseCardError("model_judgment 候选必须提供 criteria")
function_check = list(function_check or ["是否承担具体叙事功能", "是否属于角色/场内载体的有意写法"])
labels = {card["label"] for card in cards}
samples = {stype: [] for stype in ("sf", "snf", "boundary", "regression")}
for card in cards:
if card["label"] in samples:
samples[card["label"]].append("sample-" + card["id"])
rule = {
"id": rule_id,
"name": name,
"layer": layer,
"carrier_scope": carrier_scope,
"trigger": trigger,
"carve_out": list(carve_out or []),
"default_disposition": "candidate",
"function_check": function_check,
"fix_hint": fix_hint,
"samples": samples,
"version": 1,
"status": "candidate",
"case_card_ids": [card["id"] for card in cards],
"source_work_refs": sorted(works),
"evidence": "由多来源案例卡归纳;四类样例引用为待投影的 sample-case-*,待回放和独立评审。",
}
# 候选也必须是可装载的完整规则合同;状态仍固定为 candidate。
try:
from deai.schemas import validate
validate(rule, "rule")
except ValueError as exc:
raise CaseCardError(f"规则候选合同不完整: {exc}") from exc
return rule
def load_bundle(paths: Iterable[Path]) -> list[dict]:
cards = []
seen = set()
for path in paths:
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
if isinstance(data, dict) and isinstance(data.get("cards"), list):
items = data["cards"]
elif isinstance(data, dict) and isinstance(data.get("card"), dict):
# annotate/confirm 回执包装可直接作为下一生命周期命令的输入。
items = [data["card"]]
else:
items = data
if not isinstance(items, list):
raise CaseCardError(f"{path}: cards 必须是数组或单卡回执")
for card in items:
validate_card(card)
if card["id"] in seen:
raise CaseCardError(f"案例卡 id 重复: {card['id']}")
seen.add(card["id"])
cards.append(card)
return cards
def load_verification(path: Path) -> dict:
"""读取 revalidate 产出的 JSON/YAML 回执,并做最小结构门禁。"""
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
if not isinstance(data, dict) or data.get("schema_version") != REVALIDATION_SCHEMA_VERSION:
raise CaseCardError(f"{path}: 不是 {REVALIDATION_SCHEMA_VERSION} 回执")
if not isinstance(data.get("cards"), list):
raise CaseCardError(f"{path}: 回执缺少 cards 数组")
for result in data["cards"]:
if not isinstance(result, dict) or result.get("status") not in REVALIDATION_STATUSES:
raise CaseCardError(f"{path}: 回执包含非法结果")
return data
def build_inventory(root: Path, *, source_license: str = "research_only",
source_kind: str = "existing_work", glob: str = "*.txt",
work_ref_prefix: str = "ref:", source_root_ref: str | None = None,
max_cards: int = 5000, generated_on: str | None = None) -> dict:
"""扫描目录并生成可持久化的汇总报告及完整 hash-only 卡清单。
报告保留每张卡的稳定 ID、模式、位置和来源哈希;当来源不是可保存许可时,
``capture_file`` 已将片段与上下文清空。``source_root_ref`` 只作为脱敏后的
相对引用写入报告,绝不把本机绝对路径带入共享资产。
"""
if max_cards <= 0:
raise CaseCardError("max_cards 必须为正整数")
if source_license not in LICENSES:
raise CaseCardError(f"不支持的来源许可: {source_license}")
if source_kind not in {"existing_work", "public_domain", "synthetic"}:
raise CaseCardError("目录扫描 source_kind 必须是 existing_work/public_domain/synthetic")
root = root.expanduser().resolve()
if not root.is_dir():
raise CaseCardError(f"扫描根目录不存在或不是目录: {root}")
paths = sorted(path for path in root.rglob(glob) if path.is_file())
if not paths:
raise CaseCardError(f"扫描根目录没有匹配 {glob!r} 的文件: {root}")
cards: list[dict] = []
works: list[dict] = []
total_patterns: Counter[str] = Counter()
seen_ids: set[str] = set()
prefix = source_root_ref.rstrip("/") if source_root_ref else ""
for path in paths:
relative = path.relative_to(root).as_posix()
work_name = str(Path(relative).with_suffix("")).replace("\\", "/")
work_ref = f"{work_ref_prefix}{work_name}"
raw = path.read_bytes()
source_ref = f"{prefix}/{relative}" if prefix else relative
file_cards = capture_file(
path,
source_license=source_license,
source_kind=source_kind,
work_ref=work_ref,
max_cards=max_cards,
)
pattern_counts = Counter(card["observation"]["pattern_key"] for card in file_cards)
total_patterns.update(pattern_counts)
for card in file_cards:
# 不改变卡片 ID;只将机器路径替换为报告中的相对来源引用。
card = copy.deepcopy(card)
card["source"]["source_ref"] = source_ref
validate_card(card)
if card["id"] in seen_ids:
raise CaseCardError(f"目录扫描产生重复案例卡 id: {card['id']}")
seen_ids.add(card["id"])
cards.append(card)
works.append({
"work_ref": work_ref,
"source_ref": source_ref,
"source_sha256": _sha256_bytes(raw),
"card_count": len(file_cards),
"pattern_counts": dict(sorted(pattern_counts.items())),
})
return {
"schema_version": "ai-flavor-inventory-v1",
"capture_schema_version": SCHEMA_VERSION,
"generated_on": generated_on or date.today().isoformat(),
"source_root_ref": source_root_ref or root.name,
"source_kind": source_kind,
"source_license": source_license,
"max_cards_per_work": max_cards,
"totals": {
"books": len(works),
"cards": len(cards),
"pattern_counts": dict(sorted(total_patterns.items())),
},
"works": works,
"cards": cards,
}
def write_yaml(path: Path, payload) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(yaml.safe_dump(payload, allow_unicode=True, sort_keys=False), encoding="utf-8")
def _file_sha256(path: Path) -> str:
return _sha256_bytes(path.read_bytes())
def _inventory_from_cards(cards: list[dict], *, generated_on: str | None = None,
source_root_ref: str | None = None,
source_kind: str = "existing_work",
source_license: str = "research_only") -> dict:
"""把单文件/反馈检测结果包装成统一 inventory 载荷,供自动落库。"""
works: dict[tuple[str, str], dict] = {}
patterns = Counter()
for card in cards:
source = card["source"]
key = (source.get("work_ref", ""), source.get("source_ref", ""))
work = works.setdefault(key, {
"work_ref": key[0], "source_ref": key[1],
"source_sha256": source["source_sha256"], "card_count": 0,
"pattern_counts": {},
})
pattern_key = card["observation"]["pattern_key"]
work["card_count"] += 1
work["pattern_counts"][pattern_key] = work["pattern_counts"].get(pattern_key, 0) + 1
patterns.update([pattern_key])
return {
"schema_version": "ai-flavor-inventory-v1",
"capture_schema_version": SCHEMA_VERSION,
"generated_on": generated_on or date.today().isoformat(),
"source_root_ref": source_root_ref,
"source_kind": source_kind,
"source_license": source_license,
"max_cards_per_work": len(cards),
"totals": {"books": len(works), "cards": len(cards),
"pattern_counts": dict(sorted(patterns.items()))},
"works": [dict(item, pattern_counts=dict(sorted(item["pattern_counts"].items())))
for item in sorted(works.values(), key=lambda value: (value["work_ref"], value["source_ref"]))],
"cards": cards,
}
def _persist_detection(*, inventory: dict, verification: dict, inventory_path: Path,
verification_path: Path, offline: bool = False) -> dict:
"""检测命令统一落库边界;失败即让 CLI 失败,不报假绿。"""
if offline:
return {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
# 延迟导入,保持纯函数测试不建立数据库连接,同时避免循环导入。
from persist_cases import persist_generated
try:
return persist_generated(
inventory=inventory,
verification=verification,
inventory_ref=f"capture://ai-flavor/{inventory_path.name}",
report_ref=f"capture://ai-flavor/{verification_path.name}",
inventory_sha256=_file_sha256(inventory_path),
report_sha256=_file_sha256(verification_path),
)
except CaseCardError:
raise
except Exception as exc:
raise CaseCardError(f"自动落库失败(事务已回滚): {type(exc).__name__}: {exc}") from exc
def _write_revalidation(path: Path, report: dict) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def _load_inventory_file(path: Path) -> dict:
"""读取 inventory 载荷,供显式重验证落库使用。"""
try:
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
except (OSError, UnicodeError, yaml.YAMLError) as exc:
raise CaseCardError(f"{path}: inventory 读取失败: {exc}") from exc
if not isinstance(data, dict) or data.get("schema_version") != "ai-flavor-inventory-v1":
raise CaseCardError(f"{path}: 不是 ai-flavor-inventory-v1 清单")
cards = data.get("cards")
if not isinstance(cards, list):
raise CaseCardError(f"{path}: inventory.cards 必须是数组")
for card in cards:
validate_card(card)
return data
def _sidecar(path: Path, suffix: str) -> Path:
return path.with_name(f"{path.stem}{suffix}")
def _detection_paths(output: Path, *, inventory_command: bool = False) -> tuple[Path, Path]:
"""为每次检测生成不可能互相覆盖的 inventory/revalidation 文件名。"""
if inventory_command:
if "inventory" in output.stem:
verification = output.with_name(
f"{output.stem.replace('inventory', 'revalidation', 1)}{output.suffix}"
)
else:
verification = output.with_name(f"{output.stem}.revalidation{output.suffix}")
if verification == output:
verification = output.with_name(f"{output.stem}.revalidation{output.suffix}")
return output, verification
return output.with_suffix(".inventory.json"), output.with_suffix(".revalidation.json")
def _run_detection_persistence(*, cards: list[dict], inventory: dict,
inventory_path: Path, verification_path: Path,
source_root: Path | None = None,
source_texts: dict[str, str] | None = None,
checked_on: str | None = None, offline: bool = False) -> dict:
"""所有检测命令共用:生成回执文件后立即写入数据库。"""
inventory_path.parent.mkdir(parents=True, exist_ok=True)
inventory_path.write_text(json.dumps(inventory, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
verification = build_revalidation_report(
cards, source_root=source_root, source_texts=source_texts, checked_on=checked_on,
)
# 把逻辑回执引用纳入报告哈希:同一输出路径重跑幂等,换一次检测产物则留下新批次;
# 不把本机绝对路径写进正式库。
verification["report_ref"] = f"capture://ai-flavor/{verification_path.name}"
_write_revalidation(verification_path, verification)
if offline:
return {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
return _persist_detection(
inventory=inventory, verification=verification,
inventory_path=inventory_path, verification_path=verification_path,
)
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="发现并验证 AI 味案例卡")
sub = parser.add_subparsers(dest="command", required=True)
scan = sub.add_parser("scan")
scan.add_argument("path", type=Path)
scan.add_argument("--work-ref")
scan.add_argument("--license", dest="source_license", default="research_only", choices=sorted(LICENSES))
scan.add_argument("--source-kind", default="existing_work", choices=["existing_work", "public_domain", "synthetic"])
scan.add_argument("--output", type=Path, required=True)
scan.add_argument("--max-cards", type=int, default=500)
scan.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
feedback = sub.add_parser("feedback")
feedback.add_argument("--text-file", type=Path, required=True)
feedback.add_argument("--work-ref", required=True)
feedback.add_argument("--run-ref", required=True)
feedback.add_argument("--issue", required=True)
feedback.add_argument("--license", dest="source_license", default="owned", choices=sorted(LICENSES))
feedback.add_argument("--output", type=Path, required=True)
feedback.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
validate = sub.add_parser("validate")
validate.add_argument("path", type=Path)
revalidate = sub.add_parser("revalidate")
revalidate.add_argument("cards", type=Path, help="案例卡 YAML/JSON 或 inventory 报告")
revalidate.add_argument("--source-root", type=Path,
help="受控来源根目录;不传则只能验证卡片中的绝对 source_ref")
revalidate.add_argument("--output", type=Path, required=True)
revalidate.add_argument("--checked-on")
revalidate.add_argument("--offline", action="store_true", help="只生成回执,不自动写 muse-example")
propose = sub.add_parser("propose-rule")
propose.add_argument("--cards", type=Path, nargs="+", required=True)
propose.add_argument("--rule-id", required=True)
propose.add_argument("--name", required=True)
propose.add_argument("--fix-hint", default="待独立评审决定")
propose.add_argument("--layer", default="lexical", choices=["mechanical", "lexical", "structural", "density", "semantic"])
propose.add_argument("--carrier-scope", default="all", choices=["narration", "dialogue", "monologue", "in_text_carrier", "all"])
propose.add_argument("--trigger-type", default="model_judgment", choices=["regex", "handler", "density", "model_judgment"])
propose.add_argument("--pattern")
propose.add_argument("--criteria")
propose.add_argument("--handler")
propose.add_argument("--carve-out", action="append", default=[])
propose.add_argument("--function-check", action="append", default=[])
propose.add_argument("--verification", type=Path, required=True,
help="revalidate 命令生成的来源重验证回执")
propose.add_argument("--output", type=Path, required=True)
inventory = sub.add_parser("inventory")
inventory.add_argument("root", type=Path)
inventory.add_argument("--glob", default="*.txt")
inventory.add_argument("--source-root-ref")
inventory.add_argument("--work-ref-prefix", default="ref:")
inventory.add_argument("--license", dest="source_license", default="research_only", choices=sorted(LICENSES))
inventory.add_argument("--source-kind", default="existing_work", choices=["existing_work", "public_domain", "synthetic"])
inventory.add_argument("--output", type=Path, required=True)
inventory.add_argument("--max-cards", type=int, default=5000)
inventory.add_argument("--generated-on")
inventory.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
return parser
def main(argv: list[str] | None = None) -> int:
args = _parser().parse_args(argv)
try:
if args.command == "scan":
cards = capture_file(args.path, source_license=args.source_license, source_kind=args.source_kind,
work_ref=args.work_ref, max_cards=args.max_cards)
# 单文件扫描也遵守脱敏来源合同;重验证仍通过受控 source_root 读取真实文件。
cards = [copy.deepcopy(card) for card in cards]
for card in cards:
card["source"]["source_ref"] = args.path.name
inventory = _inventory_from_cards(cards, source_root_ref=args.path.parent.name,
source_kind=args.source_kind, source_license=args.source_license)
inventory_path, verification_path = _detection_paths(args.output)
result = _run_detection_persistence(
cards=cards, inventory=inventory, inventory_path=inventory_path,
verification_path=verification_path, source_root=args.path.parent,
checked_on=date.today().isoformat(), offline=args.offline,
)
# 保留原有 YAML 作为人工复核输入,但它不是正式权威。
write_yaml(args.output, {"schema_version": SCHEMA_VERSION, "cards": cards})
print(json.dumps({"cards": len(cards), "output": str(args.output),
"inventory": str(inventory_path), "revalidation": str(verification_path),
"persistence": result}, ensure_ascii=False))
elif args.command == "feedback":
text = args.text_file.read_text(encoding="utf-8")
card = capture_feedback(text, work_ref=args.work_ref,
run_ref=args.run_ref, issue=args.issue, source_license=args.source_license)
inventory = _inventory_from_cards([card], source_root_ref="live_feedback",
source_kind="creation_feedback", source_license=args.source_license)
inventory_path, verification_path = _detection_paths(args.output)
result = _run_detection_persistence(
cards=[card], inventory=inventory, inventory_path=inventory_path,
verification_path=verification_path,
source_texts={card["id"]: text}, checked_on=date.today().isoformat(), offline=args.offline,
)
write_yaml(args.output, {"schema_version": SCHEMA_VERSION, "cards": [card]})
print(json.dumps({"cards": 1, "output": str(args.output),
"inventory": str(inventory_path), "revalidation": str(verification_path),
"persistence": result}, ensure_ascii=False))
elif args.command == "validate":
cards = load_bundle([args.path])
print(json.dumps({"valid": True, "cards": len(cards)}, ensure_ascii=False))
elif args.command == "revalidate":
cards = load_bundle([args.cards])
report = build_revalidation_report(
cards,
source_root=args.source_root,
checked_on=args.checked_on,
)
report["report_ref"] = f"capture://ai-flavor/{args.output.name}"
_write_revalidation(args.output, report)
try:
inventory = _load_inventory_file(args.cards)
except CaseCardError:
inventory = _inventory_from_cards(
cards,
source_root_ref="manual-revalidate",
source_kind=cards[0]["source"].get("kind", "existing_work") if cards else "existing_work",
source_license=cards[0]["source"].get("license", "research_only") if cards else "research_only",
)
if not args.offline:
# 手工重验证也是检测运行;默认自动追加批次/逐卡回执,--offline 才不写库。
inventory_path = args.cards.with_name(f"{args.cards.stem}.inventory.json")
if args.cards.name.endswith(".inventory.json"):
inventory_path = args.cards
# 单卡 YAML 不是正式 inventory;先写 sidecar,保证自动落库有稳定的
# inventory hash 和可恢复引用,不要求用户再调用导入脚本。
if inventory_path != args.cards:
inventory_path.parent.mkdir(parents=True, exist_ok=True)
inventory_path.write_text(
json.dumps(inventory, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
verification_path = args.output
persistence = _persist_detection(
inventory=inventory, verification=report,
inventory_path=inventory_path, verification_path=verification_path,
)
else:
persistence = {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
print(json.dumps({
"cards": report["totals"]["cards"],
"verified": report["totals"]["verified"],
"stale": report["totals"]["stale"],
"unavailable": report["totals"]["unavailable"],
"card_mismatch": report["totals"]["card_mismatch"],
"usable": report["usable"],
"output": str(args.output),
"persistence": persistence,
}, ensure_ascii=False))
elif args.command == "propose-rule":
cards = load_bundle(args.cards)
verification = load_verification(args.verification)
trigger = {"type": args.trigger_type}
if args.trigger_type == "regex":
if not args.pattern:
raise CaseCardError("regex 候选必须提供 --pattern")
trigger["pattern"] = args.pattern
elif args.trigger_type == "handler":
if not args.handler:
raise CaseCardError("handler 候选必须提供 --handler")
trigger["handler"] = args.handler
elif args.trigger_type == "density":
if not args.pattern:
raise CaseCardError("density 候选必须提供 --pattern")
trigger.update({"pattern": args.pattern, "window_chars": 500, "min_hits": 3})
else:
trigger["criteria"] = args.criteria or "待独立功能判断"
rule = propose_rule(
cards, rule_id=args.rule_id, name=args.name, fix_hint=args.fix_hint,
verification=verification, layer=args.layer, carrier_scope=args.carrier_scope,
trigger=trigger, carve_out=args.carve_out, function_check=args.function_check,
)
write_yaml(args.output, rule)
print(json.dumps({"status": rule["status"], "cards": len(cards), "output": str(args.output)}, ensure_ascii=False))
else:
report = build_inventory(
args.root,
source_license=args.source_license,
source_kind=args.source_kind,
glob=args.glob,
work_ref_prefix=args.work_ref_prefix,
source_root_ref=args.source_root_ref,
max_cards=args.max_cards,
generated_on=args.generated_on,
)
inventory_path, verification_path = _detection_paths(args.output, inventory_command=True)
persistence = _run_detection_persistence(
cards=report["cards"], inventory=report,
inventory_path=args.output, verification_path=verification_path,
source_root=args.root, checked_on=report["generated_on"], offline=args.offline,
)
print(json.dumps({"books": report["totals"]["books"], "cards": report["totals"]["cards"],
"output": str(args.output), "revalidation": str(verification_path),
"persistence": persistence}, ensure_ascii=False))
return 0
except (CaseCardError, OSError, UnicodeError) as exc:
print(f"CASE_CARD_CONTRACT_FAILED: {exc}")
return 2
if __name__ == "__main__":
raise SystemExit(main())