1080 lines
52 KiB
Python
1080 lines
52 KiB
Python
#!/usr/bin/env python3
|
||
"""AI 味案例卡的确定性发现、自动落库与候选规则构造。
|
||
|
||
扫描/回填/创作反馈命令默认在检测完成后自动写入 agent-example 的
|
||
``muse-example``;``--offline`` 只用于明确的离线合同测试或恢复准备。
|
||
脚本不调用模型;语义判断由人工或独立评审补上。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import copy
|
||
import hashlib
|
||
import json
|
||
import re
|
||
import sys
|
||
from collections import Counter
|
||
from datetime import date
|
||
from pathlib import Path
|
||
from typing import Iterable
|
||
|
||
import yaml
|
||
|
||
|
||
_AGENT_ROOT = Path(__file__).resolve().parents[4]
|
||
_HUMANIZATION_SRC = _AGENT_ROOT / "humanization" / "src"
|
||
if str(_HUMANIZATION_SRC) not in sys.path:
|
||
sys.path.insert(0, str(_HUMANIZATION_SRC))
|
||
|
||
|
||
# 作为 CLI 执行时也注册稳定模块名,自动落库模块复用同一份合同异常类型,
|
||
# 避免失败路径被重复 import 变成未捕获 traceback。
|
||
if __name__ == "__main__":
|
||
sys.modules.setdefault("capture_cases", sys.modules[__name__])
|
||
|
||
|
||
SCHEMA_VERSION = "ai-flavor-case-v1"
|
||
CARD_TYPE = "ai_flavor_case"
|
||
REVALIDATION_SCHEMA_VERSION = "ai-flavor-revalidation-v1"
|
||
STATES = {"shadow", "canonical", "rejected", "archived"}
|
||
LABELS = {"unclassified", "sf", "snf", "boundary", "regression"}
|
||
LAYERS = {"mechanical", "lexical", "structural", "density", "semantic", "unknown"}
|
||
CARRIERS = {"narration", "dialogue", "monologue", "in_text_carrier", "mixed", "unknown"}
|
||
CAPTURE_MODES = {"backfill", "live_feedback", "synthetic", "review_import"}
|
||
REVALIDATION_STATUSES = {"verified", "stale", "unavailable", "card_mismatch"}
|
||
LICENSES = {"owned", "licensed", "public_domain", "research_only", "unauthorized", "synthetic"}
|
||
NON_REPO_LICENSES = {"research_only", "unauthorized"}
|
||
STOREABLE_LICENSES = LICENSES - NON_REPO_LICENSES
|
||
|
||
|
||
class CaseCardError(ValueError):
|
||
"""案例卡合同或升级门禁失败。"""
|
||
|
||
|
||
PATTERNS = (
|
||
{
|
||
"key": "lexical.meta_disclaimer",
|
||
"layer": "lexical",
|
||
"pattern": r"(?:值得注意的是|值得一提的是|不言而喻|众所周知|换句话说)",
|
||
"diagnosis": "叙述者元话语可能没有新增信息;必须检查是否承担声线或节奏功能。",
|
||
},
|
||
{
|
||
"key": "lexical.stock_micro_expression",
|
||
"layer": "lexical",
|
||
"pattern": r"(?:嘴角(?:微微|轻轻|悄然)?(?:上扬|勾起)|眼中闪过(?:一丝|一抹)?(?:光|精光|异彩)|眸光(?:微闪|深邃))",
|
||
"diagnosis": "库存微表情候选;必须检查是否是角色签名动作或场景独有反应。",
|
||
},
|
||
{
|
||
"key": "lexical.abstract_atmosphere",
|
||
"layer": "lexical",
|
||
"pattern": r"(?:一股[^。!?\n]{0,24}气息[^。!?\n]{0,24}(?:弥漫|袭来|扑面而来)|空气仿佛凝固)",
|
||
"diagnosis": "抽象气氛候选;必须检查感官细节、因果和场面功能,不能看到词就删除。",
|
||
},
|
||
)
|
||
|
||
|
||
def _sha256_bytes(data: bytes) -> str:
|
||
return hashlib.sha256(data).hexdigest()
|
||
|
||
|
||
def text_sha256(text: str) -> str:
|
||
return _sha256_bytes(text.encode("utf-8"))
|
||
|
||
|
||
def _resolve_source_path(source_ref: str, source_root: Path | None = None) -> Path | None:
|
||
"""把脱敏来源引用解析为受控本地文件;URI 和无法落到文件的引用返回 None。"""
|
||
if not source_ref or "://" in source_ref:
|
||
return None
|
||
ref = Path(source_ref).expanduser()
|
||
if ref.is_absolute():
|
||
resolved = ref.resolve()
|
||
if source_root is not None:
|
||
try:
|
||
resolved.relative_to(source_root.expanduser().resolve())
|
||
except ValueError:
|
||
return None
|
||
return resolved
|
||
if source_root is None:
|
||
return None
|
||
root = source_root.expanduser().resolve()
|
||
candidates = [root / ref]
|
||
# inventory 默认把 source_root_ref(例如“小说清单”)写进 source_ref;
|
||
# 调用方也可以直接把 source_root 指到该目录,因此兼容两种入口。
|
||
if ref.parts and ref.parts[0] in {root.name, root.parent.name}:
|
||
candidates.append(root / Path(*ref.parts[1:]))
|
||
for candidate in candidates:
|
||
resolved = candidate.resolve()
|
||
try:
|
||
resolved.relative_to(root)
|
||
except ValueError:
|
||
continue
|
||
if resolved.is_file():
|
||
return resolved
|
||
return None
|
||
|
||
|
||
def _source_snapshot(*, source_ref: str, source_root: Path | None = None,
|
||
source_path: Path | None = None, source_text: str | None = None) -> dict:
|
||
"""读取一次来源,供单卡和批量重验证复用。"""
|
||
if source_text is not None and source_path is not None:
|
||
raise CaseCardError("source_text 与 source_path 不能同时提供")
|
||
if source_text is not None:
|
||
return {"status": "loaded", "actual_sha256": text_sha256(source_text), "text": source_text, "reason": ""}
|
||
path = source_path.expanduser().resolve() if source_path is not None else _resolve_source_path(source_ref, source_root)
|
||
if path is None:
|
||
return {"status": "unavailable", "actual_sha256": None, "text": None, "reason": "source_ref_not_file"}
|
||
try:
|
||
raw = path.read_bytes()
|
||
except OSError:
|
||
return {"status": "unavailable", "actual_sha256": None, "text": None, "reason": "source_not_found"}
|
||
actual_sha256 = _sha256_bytes(raw)
|
||
try:
|
||
text = raw.decode("utf-8")
|
||
except UnicodeError:
|
||
return {"status": "unavailable", "actual_sha256": actual_sha256, "text": None,
|
||
"reason": "source_not_utf8"}
|
||
return {"status": "loaded", "actual_sha256": actual_sha256, "text": text, "reason": ""}
|
||
|
||
|
||
def _card_anchor_reason(card: dict, text: str) -> str | None:
|
||
"""在全文哈希相同后,再检查卡片的位置和命中是否仍自洽。"""
|
||
source = card["source"]
|
||
location = source.get("location") or {}
|
||
start, end = location.get("char_start"), location.get("char_end")
|
||
if not isinstance(start, int) or not isinstance(end, int) or start < 0 or end < start or end > len(text):
|
||
return "card_anchor_out_of_range"
|
||
if text_sha256(text[start:end]) != source["excerpt_sha256"]:
|
||
return "card_excerpt_hash_mismatch"
|
||
surface_location = source.get("surface_location")
|
||
if surface_location:
|
||
surface_start, surface_end = surface_location.get("char_start"), surface_location.get("char_end")
|
||
expected_surface = card.get("observation", {}).get("surface", "")
|
||
if (not isinstance(surface_start, int) or not isinstance(surface_end, int)
|
||
or surface_start < 0 or surface_end < surface_start or surface_end > len(text)):
|
||
return "card_surface_anchor_out_of_range"
|
||
if text[surface_start:surface_end] != expected_surface:
|
||
return "card_surface_mismatch"
|
||
return None
|
||
|
||
|
||
def _revalidation_result(card: dict, snapshot: dict, *, checked_on: str) -> dict:
|
||
source = card["source"]
|
||
result = {
|
||
"card_id": card["id"],
|
||
"work_ref": source.get("work_ref", ""),
|
||
"source_ref": source.get("source_ref", ""),
|
||
"expected_source_sha256": source["source_sha256"],
|
||
"actual_source_sha256": snapshot.get("actual_sha256"),
|
||
"status": snapshot["status"],
|
||
"reason": snapshot.get("reason", ""),
|
||
"checked_on": checked_on,
|
||
}
|
||
if snapshot["status"] == "loaded":
|
||
if snapshot["actual_sha256"] != source["source_sha256"]:
|
||
result["status"] = "stale"
|
||
result["reason"] = "source_hash_mismatch"
|
||
else:
|
||
reason = _card_anchor_reason(card, snapshot["text"])
|
||
if reason:
|
||
result["status"] = "card_mismatch"
|
||
result["reason"] = reason
|
||
else:
|
||
result["status"] = "verified"
|
||
result["reason"] = "source_hash_and_anchor_match"
|
||
return result
|
||
|
||
|
||
def revalidate_card(card: dict, *, source_root: Path | None = None, source_path: Path | None = None,
|
||
source_text: str | None = None, checked_on: str | None = None) -> dict:
|
||
"""重新读取来源并返回不可变验证结果;不会自动改写案例卡状态。"""
|
||
validate_card(card)
|
||
snapshot = _source_snapshot(source_ref=card["source"].get("source_ref", ""),
|
||
source_root=source_root, source_path=source_path, source_text=source_text)
|
||
return _revalidation_result(card, snapshot, checked_on=checked_on or date.today().isoformat())
|
||
|
||
|
||
def revalidate_cards(cards: list[dict], *, source_root: Path | None = None,
|
||
checked_on: str | None = None) -> list[dict]:
|
||
"""批量重验证;同一来源只读取一次。"""
|
||
checked_on = checked_on or date.today().isoformat()
|
||
snapshots: dict[str, dict] = {}
|
||
results = []
|
||
for card in cards:
|
||
validate_card(card)
|
||
source_ref = card["source"].get("source_ref", "")
|
||
if source_ref not in snapshots:
|
||
snapshots[source_ref] = _source_snapshot(source_ref=source_ref, source_root=source_root)
|
||
results.append(_revalidation_result(card, snapshots[source_ref], checked_on=checked_on))
|
||
return results
|
||
|
||
|
||
def build_revalidation_report(cards: list[dict], *, source_root: Path | None = None,
|
||
checked_on: str | None = None,
|
||
source_texts: dict[str, str] | None = None) -> dict:
|
||
if source_texts is None:
|
||
results = revalidate_cards(cards, source_root=source_root, checked_on=checked_on)
|
||
else:
|
||
results = []
|
||
for card in cards:
|
||
if card["id"] not in source_texts:
|
||
raise CaseCardError(f"缺少案例卡 {card['id']} 的反馈正文")
|
||
results.append(revalidate_card(card, source_text=source_texts[card["id"]], checked_on=checked_on))
|
||
counts = Counter(result["status"] for result in results)
|
||
works: dict[tuple[str, str], dict] = {}
|
||
for result in results:
|
||
key = (result["work_ref"], result["source_ref"])
|
||
work = works.setdefault(key, {
|
||
"work_ref": result["work_ref"], "source_ref": result["source_ref"],
|
||
"expected_source_sha256": result["expected_source_sha256"],
|
||
"actual_source_sha256": result["actual_source_sha256"], "card_count": 0,
|
||
"status_counts": {},
|
||
})
|
||
work["card_count"] += 1
|
||
work["status_counts"][result["status"]] = work["status_counts"].get(result["status"], 0) + 1
|
||
if work["actual_source_sha256"] is None and result["actual_source_sha256"] is not None:
|
||
work["actual_source_sha256"] = result["actual_source_sha256"]
|
||
return {
|
||
"schema_version": REVALIDATION_SCHEMA_VERSION,
|
||
"checked_on": checked_on or date.today().isoformat(),
|
||
"source_root_ref": source_root.name if source_root else None,
|
||
"usable": bool(results) and all(result["status"] == "verified" for result in results),
|
||
"totals": {"cards": len(results), **{status: counts.get(status, 0) for status in sorted(REVALIDATION_STATUSES)}},
|
||
"works": list(sorted(works.values(), key=lambda item: (item["work_ref"], item["source_ref"]))),
|
||
"cards": results,
|
||
}
|
||
|
||
|
||
def _verification_results(verification) -> dict[str, dict]:
|
||
if not isinstance(verification, dict):
|
||
return {}
|
||
if verification.get("card_id"):
|
||
return {verification["card_id"]: verification}
|
||
items = verification.get("cards")
|
||
if not isinstance(items, list):
|
||
return {}
|
||
return {item.get("card_id"): item for item in items if isinstance(item, dict) and item.get("card_id")}
|
||
|
||
|
||
def require_verified(card: dict, verification) -> dict:
|
||
"""升级或消费前的 fail-closed 门:只接受本卡的 verified 回执。"""
|
||
validate_card(card)
|
||
result = _verification_results(verification).get(card["id"])
|
||
if (not result or result.get("status") != "verified"
|
||
or result.get("expected_source_sha256") != card["source"]["source_sha256"]
|
||
or result.get("actual_source_sha256") != card["source"]["source_sha256"]):
|
||
status = result.get("status") if result else "missing"
|
||
raise CaseCardError(f"案例卡 {card['id']} 来源未通过重验证: {status}")
|
||
return result
|
||
|
||
|
||
def require_verified_cards(cards: list[dict], verification) -> list[dict]:
|
||
return [require_verified(card, verification) for card in cards]
|
||
|
||
|
||
def _line_col(text: str, offset: int) -> tuple[int, int]:
|
||
line_start = text.count("\n", 0, offset) + 1
|
||
last_newline = text.rfind("\n", 0, offset)
|
||
return line_start, offset - (last_newline + 1)
|
||
|
||
|
||
def _location(text: str, start: int, end: int) -> dict:
|
||
line_start, col_start = _line_col(text, start)
|
||
line_end, col_end = _line_col(text, end)
|
||
return {
|
||
"line_start": line_start,
|
||
"line_end": line_end,
|
||
"char_start": start,
|
||
"char_end": end,
|
||
"column_start": col_start,
|
||
"column_end": col_end,
|
||
}
|
||
|
||
|
||
def _context(text: str, start: int, end: int) -> str:
|
||
"""取命中所在行及相邻一行,供有权保存的来源做人工复核。"""
|
||
line_start = text.rfind("\n", 0, start) + 1
|
||
line_end = text.find("\n", end)
|
||
if line_end < 0:
|
||
line_end = len(text)
|
||
previous_start = text.rfind("\n", 0, max(0, line_start - 1)) + 1
|
||
next_end = text.find("\n", line_end + 1)
|
||
if next_end < 0:
|
||
next_end = len(text)
|
||
return text[previous_start:next_end].strip("\n")
|
||
|
||
|
||
def _evidence_window(text: str, start: int, end: int, max_chars: int = 180) -> tuple[int, int]:
|
||
"""取命中词所在句的有限窗口,避免样例只有触发词或吞入整章。"""
|
||
left = max(text.rfind("。", 0, start), text.rfind("!", 0, start),
|
||
text.rfind("?", 0, start), text.rfind("\n", 0, start)) + 1
|
||
right_candidates = [p for p in (text.find("。", end), text.find("!", end),
|
||
text.find("?", end), text.find("\n", end)) if p >= 0]
|
||
right = min(right_candidates, default=len(text))
|
||
if right_candidates:
|
||
right += 1
|
||
if right - left > max_chars:
|
||
left = max(0, start - max_chars // 2)
|
||
right = min(len(text), max(end, start + max_chars // 2))
|
||
if right - left > max_chars:
|
||
right = left + max_chars
|
||
return left, right
|
||
|
||
|
||
def _require_string(value, field: str, *, allow_empty: bool = False) -> str:
|
||
if not isinstance(value, str) or (not allow_empty and not value.strip()):
|
||
raise CaseCardError(f"{field} 必须是{'可为空的' if allow_empty else ''}字符串")
|
||
return value
|
||
|
||
|
||
def validate_card(card: dict) -> dict:
|
||
"""验证单卡的字段和跨字段不变量。"""
|
||
if not isinstance(card, dict):
|
||
raise CaseCardError("卡片必须是对象")
|
||
for key in ("schema_version", "id", "card_type", "state", "label", "layer", "carrier", "capture_mode", "source", "observation"):
|
||
if key not in card:
|
||
raise CaseCardError(f"卡片缺少 {key}")
|
||
if card["schema_version"] != SCHEMA_VERSION:
|
||
raise CaseCardError(f"schema_version 必须为 {SCHEMA_VERSION}")
|
||
_require_string(card["id"], "id")
|
||
if not card["id"].startswith("case-"):
|
||
raise CaseCardError("id 必须以 case- 开头")
|
||
if card["card_type"] != CARD_TYPE:
|
||
raise CaseCardError("card_type 必须为 ai_flavor_case")
|
||
for field, allowed in (("state", STATES), ("label", LABELS), ("layer", LAYERS), ("carrier", CARRIERS), ("capture_mode", CAPTURE_MODES)):
|
||
if card[field] not in allowed:
|
||
raise CaseCardError(f"{field} 取值非法: {card[field]!r}")
|
||
|
||
source = card["source"]
|
||
if not isinstance(source, dict):
|
||
raise CaseCardError("source 必须是对象")
|
||
for key in ("kind", "license", "source_sha256", "excerpt_sha256", "location"):
|
||
if key not in source:
|
||
raise CaseCardError(f"source 缺少 {key}")
|
||
if source["kind"] not in {"existing_work", "creation_feedback", "synthetic", "public_domain"}:
|
||
raise CaseCardError(f"source.kind 非法: {source['kind']!r}")
|
||
if source["license"] not in LICENSES:
|
||
raise CaseCardError(f"source.license 非法: {source['license']!r}")
|
||
for key in ("source_sha256", "excerpt_sha256"):
|
||
if not re.fullmatch(r"[0-9a-f]{64}", source[key]):
|
||
raise CaseCardError(f"source.{key} 必须是 64 位小写 SHA-256")
|
||
if source["kind"] == "existing_work" and not source.get("work_ref"):
|
||
raise CaseCardError("existing_work 必须有 work_ref")
|
||
if card["capture_mode"] == "live_feedback" and source["kind"] != "creation_feedback":
|
||
raise CaseCardError("live_feedback 的 source.kind 必须为 creation_feedback")
|
||
location = source["location"]
|
||
if not isinstance(location, dict):
|
||
raise CaseCardError("source.location 必须是对象")
|
||
for key in ("line_start", "line_end", "char_start", "char_end"):
|
||
if not isinstance(location.get(key), int) or location[key] < 0:
|
||
raise CaseCardError(f"source.location.{key} 必须是非负整数")
|
||
if location["line_end"] < location["line_start"] or location["char_end"] < location["char_start"]:
|
||
raise CaseCardError("source.location 结束位置不能早于开始位置")
|
||
|
||
_require_string(card.get("excerpt", ""), "excerpt", allow_empty=True)
|
||
_require_string(card.get("context", ""), "context", allow_empty=True)
|
||
if source["license"] in NON_REPO_LICENSES:
|
||
if card.get("excerpt") or card.get("context"):
|
||
raise CaseCardError("未授权/研究限定来源必须 hash-only")
|
||
if card["state"] != "shadow":
|
||
raise CaseCardError("未授权/研究限定来源只能保持 shadow")
|
||
if card["state"] == "canonical":
|
||
if source["license"] not in STOREABLE_LICENSES:
|
||
raise CaseCardError("canonical 卡必须来自可保存的授权来源")
|
||
if not card.get("excerpt"):
|
||
raise CaseCardError("canonical 卡必须有可审阅片段")
|
||
if card["label"] == "unclassified":
|
||
raise CaseCardError("canonical 卡必须有评审标签")
|
||
review = card.get("review")
|
||
if not isinstance(review, dict) or not review.get("reviewer") or not review.get("note"):
|
||
raise CaseCardError("canonical 卡必须有 review.reviewer 和 review.note")
|
||
if card.get("excerpt") and source["excerpt_sha256"] != text_sha256(card["excerpt"]):
|
||
raise CaseCardError("excerpt_sha256 与 excerpt 不一致")
|
||
|
||
observation = card["observation"]
|
||
if not isinstance(observation, dict):
|
||
raise CaseCardError("observation 必须是对象")
|
||
for key in ("pattern_key", "surface", "diagnosis", "function_check", "risk_if_changed", "suggested_action"):
|
||
if key not in observation:
|
||
raise CaseCardError(f"observation 缺少 {key}")
|
||
for key in ("pattern_key", "surface", "diagnosis", "risk_if_changed", "suggested_action"):
|
||
_require_string(observation[key], f"observation.{key}")
|
||
if not isinstance(observation["function_check"], list) or not observation["function_check"]:
|
||
raise CaseCardError("observation.function_check 必须是非空数组")
|
||
return card
|
||
|
||
|
||
def _card_id(source_sha: str, start: int, end: int, pattern_key: str) -> str:
|
||
raw = f"{source_sha}:{start}:{end}:{pattern_key}".encode("utf-8")
|
||
return "case-" + _sha256_bytes(raw)[:20]
|
||
|
||
|
||
def build_case_card(*, text: str, source_sha256: str, source_kind: str, source_license: str,
|
||
source_ref: str, work_ref: str | None, start: int, end: int,
|
||
pattern: dict, capture_mode: str = "backfill", feedback: dict | None = None,
|
||
excerpt_start: int | None = None, excerpt_end: int | None = None) -> dict:
|
||
if source_sha256 != text_sha256(text):
|
||
raise CaseCardError("source_sha256 与原始正文不一致")
|
||
if start < 0 or end < start or end > len(text):
|
||
raise CaseCardError("surface 命中位置越界")
|
||
evidence_start = start if excerpt_start is None else excerpt_start
|
||
evidence_end = end if excerpt_end is None else excerpt_end
|
||
if evidence_start < 0 or evidence_end < evidence_start or evidence_end > len(text):
|
||
raise CaseCardError("evidence 位置越界")
|
||
excerpt = text[evidence_start:evidence_end]
|
||
stored = source_license in STOREABLE_LICENSES
|
||
source = {
|
||
"kind": source_kind,
|
||
"license": source_license,
|
||
"source_sha256": source_sha256,
|
||
"excerpt_sha256": text_sha256(excerpt),
|
||
"source_ref": source_ref,
|
||
"location": _location(text, evidence_start, evidence_end),
|
||
"surface_location": _location(text, start, end),
|
||
}
|
||
if work_ref:
|
||
source["work_ref"] = work_ref
|
||
card = {
|
||
"schema_version": SCHEMA_VERSION,
|
||
"id": _card_id(source_sha256, start, end, pattern["key"]),
|
||
"card_type": CARD_TYPE,
|
||
"state": "shadow",
|
||
"label": "unclassified",
|
||
"layer": pattern["layer"],
|
||
"carrier": "unknown",
|
||
"capture_mode": capture_mode,
|
||
"excerpt": excerpt if stored else "",
|
||
"context": _context(text, start, end) if stored else "",
|
||
"source": source,
|
||
"observation": {
|
||
"pattern_key": pattern["key"],
|
||
"surface": text[start:end],
|
||
"diagnosis": pattern["diagnosis"],
|
||
"function_check": ["是否承担具体叙事功能", "是否是角色/场内载体的有意声线"],
|
||
"risk_if_changed": "未经上下文复核直接删除可能损失人物声音、伏笔或节奏。",
|
||
"suggested_action": "保留 shadow,补充上下文后再标注 sf/snf/boundary。",
|
||
},
|
||
}
|
||
if feedback is not None:
|
||
card["feedback"] = copy.deepcopy(feedback)
|
||
return validate_card(card)
|
||
|
||
|
||
def capture_file(path: Path, *, source_license: str = "research_only", source_kind: str = "existing_work",
|
||
work_ref: str | None = None, patterns: Iterable[dict] = PATTERNS,
|
||
max_cards: int = 500) -> list[dict]:
|
||
if isinstance(max_cards, bool) or max_cards <= 0:
|
||
raise CaseCardError("max_cards 必须为正整数")
|
||
if source_license not in LICENSES:
|
||
raise CaseCardError(f"不支持的来源许可: {source_license}")
|
||
if source_kind not in {"existing_work", "public_domain", "synthetic"}:
|
||
raise CaseCardError("文件扫描 source_kind 必须是 existing_work/public_domain/synthetic")
|
||
raw = path.read_bytes()
|
||
text = raw.decode("utf-8")
|
||
source_sha = _sha256_bytes(raw)
|
||
cards = []
|
||
seen = set()
|
||
matches = []
|
||
for pattern in patterns:
|
||
matches.extend((match.start(), pattern["key"], pattern, match)
|
||
for match in re.finditer(pattern["pattern"], text, flags=re.MULTILINE))
|
||
for _start, _key, pattern, match in sorted(matches, key=lambda item: (item[0], item[1])):
|
||
evidence_start, evidence_end = _evidence_window(text, match.start(), match.end())
|
||
card = build_case_card(
|
||
text=text, source_sha256=source_sha, source_kind=source_kind,
|
||
source_license=source_license, source_ref=str(path), work_ref=work_ref,
|
||
start=match.start(), end=match.end(), pattern=pattern,
|
||
excerpt_start=evidence_start, excerpt_end=evidence_end,
|
||
)
|
||
if card["id"] not in seen:
|
||
cards.append(card)
|
||
seen.add(card["id"])
|
||
if len(cards) >= max_cards:
|
||
return cards
|
||
return cards
|
||
|
||
|
||
def capture_feedback(text: str, *, work_ref: str, run_ref: str, issue: str,
|
||
source_license: str = "owned", location: dict | None = None) -> dict:
|
||
if not text:
|
||
raise CaseCardError("反馈正文不能为空")
|
||
if not work_ref or not run_ref or not issue:
|
||
raise CaseCardError("反馈必须有 work_ref、run_ref 和 issue")
|
||
pattern = {
|
||
"key": "feedback.manual_observation",
|
||
"layer": "unknown",
|
||
"diagnosis": "创作反馈待人工归类,不把一次事故直接固化为规则。",
|
||
}
|
||
card = build_case_card(
|
||
text=text, source_sha256=text_sha256(text), source_kind="creation_feedback",
|
||
source_license=source_license, source_ref=run_ref, work_ref=work_ref,
|
||
start=0, end=len(text), pattern=pattern, capture_mode="live_feedback",
|
||
feedback={"run_ref": run_ref, "issue": issue, "candidate_sha256": text_sha256(text)},
|
||
)
|
||
card["source"]["location"] = location or {"line_start": 1, "line_end": text.count("\n") + 1,
|
||
"char_start": 0, "char_end": len(text),
|
||
"column_start": 0, "column_end": 0}
|
||
return validate_card(card)
|
||
|
||
|
||
def annotate_card(card: dict, *, label: str, carrier: str = "unknown", reviewer: str = "", note: str = "") -> dict:
|
||
validate_card(card)
|
||
if card["state"] != "shadow":
|
||
raise CaseCardError("只有 shadow 卡可以标注")
|
||
if label not in LABELS - {"unclassified"}:
|
||
raise CaseCardError(f"标注标签非法: {label}")
|
||
out = copy.deepcopy(card)
|
||
out["label"] = label
|
||
out["carrier"] = carrier
|
||
if reviewer or note:
|
||
out["review"] = {"reviewer": reviewer, "note": note}
|
||
return validate_card(out)
|
||
|
||
|
||
def confirm_card(card: dict, *, reviewer: str, note: str, verification=None) -> dict:
|
||
validate_card(card)
|
||
if card["source"]["license"] in NON_REPO_LICENSES:
|
||
raise CaseCardError("hash-only 卡不能在仓库内确认")
|
||
require_verified(card, verification)
|
||
out = copy.deepcopy(card)
|
||
out["state"] = "canonical"
|
||
out["review"] = {"reviewer": reviewer, "note": note}
|
||
return validate_card(out)
|
||
|
||
|
||
def project_sample(card: dict, *, verification=None) -> dict:
|
||
validate_card(card)
|
||
if card["state"] != "canonical":
|
||
raise CaseCardError("只有 canonical 卡可以投影样例")
|
||
require_verified(card, verification)
|
||
if card["carrier"] == "unknown":
|
||
raise CaseCardError("canonical 卡投影样例前必须确认 carrier,不能用 unknown")
|
||
source_map = {
|
||
"owned": "hand_written",
|
||
"licensed": "licensed",
|
||
"public_domain": "public_domain",
|
||
"synthetic": "synthetic",
|
||
}
|
||
return {
|
||
"id": "sample-" + card["id"],
|
||
"type": card["label"],
|
||
"rules": list(card.get("rule_candidate_ids", [])),
|
||
"carrier": card["carrier"],
|
||
"source": source_map[card["source"]["license"]],
|
||
"text": card["excerpt"],
|
||
"note": card["observation"]["diagnosis"],
|
||
"case_card_id": card["id"],
|
||
"source_ref": card["source"].get("source_ref", ""),
|
||
"source_license": card["source"]["license"],
|
||
}
|
||
|
||
|
||
def propose_rule(cards: list[dict], *, rule_id: str, name: str, fix_hint: str,
|
||
verification=None, layer: str = "lexical", carrier_scope: str = "all",
|
||
trigger: dict | None = None, carve_out: list[str] | None = None,
|
||
function_check: list[str] | None = None) -> dict:
|
||
if not cards:
|
||
raise CaseCardError("规则候选至少需要一张案例卡")
|
||
for card in cards:
|
||
validate_card(card)
|
||
# 候选仍可保持 candidate,但其证据必须先证明对应来源没有漂移。
|
||
require_verified_cards(cards, verification)
|
||
works = {card["source"].get("work_ref") for card in cards if card["source"].get("work_ref")}
|
||
if len(works) < 2:
|
||
raise CaseCardError("规则候选至少需要两个不同来源作品")
|
||
labels = {card["label"] for card in cards}
|
||
if "sf" not in labels or not labels.intersection({"snf", "boundary", "regression"}):
|
||
raise CaseCardError("规则候选必须同时有 sf 与 snf/boundary/regression 证据")
|
||
if layer not in LAYERS - {"unknown"}:
|
||
raise CaseCardError(f"规则 layer 非法: {layer}")
|
||
if carrier_scope not in {"narration", "dialogue", "monologue", "in_text_carrier", "all"}:
|
||
raise CaseCardError(f"规则 carrier_scope 非法: {carrier_scope}")
|
||
trigger = copy.deepcopy(trigger or {"type": "model_judgment", "criteria": "待独立功能判断"})
|
||
trigger_type = trigger.get("type")
|
||
if trigger_type == "regex" and not trigger.get("pattern"):
|
||
raise CaseCardError("regex 候选必须提供 pattern")
|
||
if trigger_type == "handler" and not trigger.get("handler"):
|
||
raise CaseCardError("handler 候选必须提供 handler")
|
||
if trigger_type == "density" and (
|
||
not trigger.get("pattern") or not trigger.get("window_chars") or not trigger.get("min_hits")):
|
||
raise CaseCardError("density 候选必须提供 pattern/window_chars/min_hits")
|
||
if trigger_type == "model_judgment" and not trigger.get("criteria"):
|
||
raise CaseCardError("model_judgment 候选必须提供 criteria")
|
||
function_check = list(function_check or ["是否承担具体叙事功能", "是否属于角色/场内载体的有意写法"])
|
||
labels = {card["label"] for card in cards}
|
||
samples = {stype: [] for stype in ("sf", "snf", "boundary", "regression")}
|
||
for card in cards:
|
||
if card["label"] in samples:
|
||
samples[card["label"]].append("sample-" + card["id"])
|
||
rule = {
|
||
"id": rule_id,
|
||
"name": name,
|
||
"layer": layer,
|
||
"carrier_scope": carrier_scope,
|
||
"trigger": trigger,
|
||
"carve_out": list(carve_out or []),
|
||
"default_disposition": "candidate",
|
||
"function_check": function_check,
|
||
"fix_hint": fix_hint,
|
||
"samples": samples,
|
||
"version": 1,
|
||
"status": "candidate",
|
||
"case_card_ids": [card["id"] for card in cards],
|
||
"source_work_refs": sorted(works),
|
||
"evidence": "由多来源案例卡归纳;四类样例引用为待投影的 sample-case-*,待回放和独立评审。",
|
||
}
|
||
# 候选也必须是可装载的完整规则合同;状态仍固定为 candidate。
|
||
try:
|
||
from deai.schemas import validate
|
||
validate(rule, "rule")
|
||
except ValueError as exc:
|
||
raise CaseCardError(f"规则候选合同不完整: {exc}") from exc
|
||
return rule
|
||
|
||
|
||
def load_bundle(paths: Iterable[Path]) -> list[dict]:
|
||
cards = []
|
||
seen = set()
|
||
for path in paths:
|
||
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
||
if isinstance(data, dict) and isinstance(data.get("cards"), list):
|
||
items = data["cards"]
|
||
elif isinstance(data, dict) and isinstance(data.get("card"), dict):
|
||
# annotate/confirm 回执包装可直接作为下一生命周期命令的输入。
|
||
items = [data["card"]]
|
||
else:
|
||
items = data
|
||
if not isinstance(items, list):
|
||
raise CaseCardError(f"{path}: cards 必须是数组或单卡回执")
|
||
for card in items:
|
||
validate_card(card)
|
||
if card["id"] in seen:
|
||
raise CaseCardError(f"案例卡 id 重复: {card['id']}")
|
||
seen.add(card["id"])
|
||
cards.append(card)
|
||
return cards
|
||
|
||
|
||
def load_verification(path: Path) -> dict:
|
||
"""读取 revalidate 产出的 JSON/YAML 回执,并做最小结构门禁。"""
|
||
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
||
if not isinstance(data, dict) or data.get("schema_version") != REVALIDATION_SCHEMA_VERSION:
|
||
raise CaseCardError(f"{path}: 不是 {REVALIDATION_SCHEMA_VERSION} 回执")
|
||
if not isinstance(data.get("cards"), list):
|
||
raise CaseCardError(f"{path}: 回执缺少 cards 数组")
|
||
for result in data["cards"]:
|
||
if not isinstance(result, dict) or result.get("status") not in REVALIDATION_STATUSES:
|
||
raise CaseCardError(f"{path}: 回执包含非法结果")
|
||
return data
|
||
|
||
|
||
def build_inventory(root: Path, *, source_license: str = "research_only",
|
||
source_kind: str = "existing_work", glob: str = "*.txt",
|
||
work_ref_prefix: str = "ref:", source_root_ref: str | None = None,
|
||
max_cards: int = 5000, generated_on: str | None = None) -> dict:
|
||
"""扫描目录并生成可持久化的汇总报告及完整 hash-only 卡清单。
|
||
|
||
报告保留每张卡的稳定 ID、模式、位置和来源哈希;当来源不是可保存许可时,
|
||
``capture_file`` 已将片段与上下文清空。``source_root_ref`` 只作为脱敏后的
|
||
相对引用写入报告,绝不把本机绝对路径带入共享资产。
|
||
"""
|
||
if max_cards <= 0:
|
||
raise CaseCardError("max_cards 必须为正整数")
|
||
if source_license not in LICENSES:
|
||
raise CaseCardError(f"不支持的来源许可: {source_license}")
|
||
if source_kind not in {"existing_work", "public_domain", "synthetic"}:
|
||
raise CaseCardError("目录扫描 source_kind 必须是 existing_work/public_domain/synthetic")
|
||
root = root.expanduser().resolve()
|
||
if not root.is_dir():
|
||
raise CaseCardError(f"扫描根目录不存在或不是目录: {root}")
|
||
paths = sorted(path for path in root.rglob(glob) if path.is_file())
|
||
if not paths:
|
||
raise CaseCardError(f"扫描根目录没有匹配 {glob!r} 的文件: {root}")
|
||
|
||
cards: list[dict] = []
|
||
works: list[dict] = []
|
||
total_patterns: Counter[str] = Counter()
|
||
seen_ids: set[str] = set()
|
||
prefix = source_root_ref.rstrip("/") if source_root_ref else ""
|
||
for path in paths:
|
||
relative = path.relative_to(root).as_posix()
|
||
work_name = str(Path(relative).with_suffix("")).replace("\\", "/")
|
||
work_ref = f"{work_ref_prefix}{work_name}"
|
||
raw = path.read_bytes()
|
||
source_ref = f"{prefix}/{relative}" if prefix else relative
|
||
file_cards = capture_file(
|
||
path,
|
||
source_license=source_license,
|
||
source_kind=source_kind,
|
||
work_ref=work_ref,
|
||
max_cards=max_cards,
|
||
)
|
||
pattern_counts = Counter(card["observation"]["pattern_key"] for card in file_cards)
|
||
total_patterns.update(pattern_counts)
|
||
for card in file_cards:
|
||
# 不改变卡片 ID;只将机器路径替换为报告中的相对来源引用。
|
||
card = copy.deepcopy(card)
|
||
card["source"]["source_ref"] = source_ref
|
||
validate_card(card)
|
||
if card["id"] in seen_ids:
|
||
raise CaseCardError(f"目录扫描产生重复案例卡 id: {card['id']}")
|
||
seen_ids.add(card["id"])
|
||
cards.append(card)
|
||
works.append({
|
||
"work_ref": work_ref,
|
||
"source_ref": source_ref,
|
||
"source_sha256": _sha256_bytes(raw),
|
||
"card_count": len(file_cards),
|
||
"pattern_counts": dict(sorted(pattern_counts.items())),
|
||
})
|
||
|
||
return {
|
||
"schema_version": "ai-flavor-inventory-v1",
|
||
"capture_schema_version": SCHEMA_VERSION,
|
||
"generated_on": generated_on or date.today().isoformat(),
|
||
"source_root_ref": source_root_ref or root.name,
|
||
"source_kind": source_kind,
|
||
"source_license": source_license,
|
||
"max_cards_per_work": max_cards,
|
||
"totals": {
|
||
"books": len(works),
|
||
"cards": len(cards),
|
||
"pattern_counts": dict(sorted(total_patterns.items())),
|
||
},
|
||
"works": works,
|
||
"cards": cards,
|
||
}
|
||
|
||
|
||
def write_yaml(path: Path, payload) -> None:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(yaml.safe_dump(payload, allow_unicode=True, sort_keys=False), encoding="utf-8")
|
||
|
||
|
||
def _file_sha256(path: Path) -> str:
|
||
return _sha256_bytes(path.read_bytes())
|
||
|
||
|
||
def _inventory_from_cards(cards: list[dict], *, generated_on: str | None = None,
|
||
source_root_ref: str | None = None,
|
||
source_kind: str = "existing_work",
|
||
source_license: str = "research_only") -> dict:
|
||
"""把单文件/反馈检测结果包装成统一 inventory 载荷,供自动落库。"""
|
||
works: dict[tuple[str, str], dict] = {}
|
||
patterns = Counter()
|
||
for card in cards:
|
||
source = card["source"]
|
||
key = (source.get("work_ref", ""), source.get("source_ref", ""))
|
||
work = works.setdefault(key, {
|
||
"work_ref": key[0], "source_ref": key[1],
|
||
"source_sha256": source["source_sha256"], "card_count": 0,
|
||
"pattern_counts": {},
|
||
})
|
||
pattern_key = card["observation"]["pattern_key"]
|
||
work["card_count"] += 1
|
||
work["pattern_counts"][pattern_key] = work["pattern_counts"].get(pattern_key, 0) + 1
|
||
patterns.update([pattern_key])
|
||
return {
|
||
"schema_version": "ai-flavor-inventory-v1",
|
||
"capture_schema_version": SCHEMA_VERSION,
|
||
"generated_on": generated_on or date.today().isoformat(),
|
||
"source_root_ref": source_root_ref,
|
||
"source_kind": source_kind,
|
||
"source_license": source_license,
|
||
"max_cards_per_work": len(cards),
|
||
"totals": {"books": len(works), "cards": len(cards),
|
||
"pattern_counts": dict(sorted(patterns.items()))},
|
||
"works": [dict(item, pattern_counts=dict(sorted(item["pattern_counts"].items())))
|
||
for item in sorted(works.values(), key=lambda value: (value["work_ref"], value["source_ref"]))],
|
||
"cards": cards,
|
||
}
|
||
|
||
|
||
def _persist_detection(*, inventory: dict, verification: dict, inventory_path: Path,
|
||
verification_path: Path, offline: bool = False) -> dict:
|
||
"""检测命令统一落库边界;失败即让 CLI 失败,不报假绿。"""
|
||
if offline:
|
||
return {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
|
||
# 延迟导入,保持纯函数测试不建立数据库连接,同时避免循环导入。
|
||
from persist_cases import persist_generated
|
||
|
||
try:
|
||
return persist_generated(
|
||
inventory=inventory,
|
||
verification=verification,
|
||
inventory_ref=f"capture://ai-flavor/{inventory_path.name}",
|
||
report_ref=f"capture://ai-flavor/{verification_path.name}",
|
||
inventory_sha256=_file_sha256(inventory_path),
|
||
report_sha256=_file_sha256(verification_path),
|
||
)
|
||
except CaseCardError:
|
||
raise
|
||
except Exception as exc:
|
||
raise CaseCardError(f"自动落库失败(事务已回滚): {type(exc).__name__}: {exc}") from exc
|
||
|
||
|
||
def _write_revalidation(path: Path, report: dict) -> None:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||
|
||
|
||
def _load_inventory_file(path: Path) -> dict:
|
||
"""读取 inventory 载荷,供显式重验证落库使用。"""
|
||
try:
|
||
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
||
except (OSError, UnicodeError, yaml.YAMLError) as exc:
|
||
raise CaseCardError(f"{path}: inventory 读取失败: {exc}") from exc
|
||
if not isinstance(data, dict) or data.get("schema_version") != "ai-flavor-inventory-v1":
|
||
raise CaseCardError(f"{path}: 不是 ai-flavor-inventory-v1 清单")
|
||
cards = data.get("cards")
|
||
if not isinstance(cards, list):
|
||
raise CaseCardError(f"{path}: inventory.cards 必须是数组")
|
||
for card in cards:
|
||
validate_card(card)
|
||
return data
|
||
|
||
|
||
def _sidecar(path: Path, suffix: str) -> Path:
|
||
return path.with_name(f"{path.stem}{suffix}")
|
||
|
||
|
||
def _detection_paths(output: Path, *, inventory_command: bool = False) -> tuple[Path, Path]:
|
||
"""为每次检测生成不可能互相覆盖的 inventory/revalidation 文件名。"""
|
||
if inventory_command:
|
||
if "inventory" in output.stem:
|
||
verification = output.with_name(
|
||
f"{output.stem.replace('inventory', 'revalidation', 1)}{output.suffix}"
|
||
)
|
||
else:
|
||
verification = output.with_name(f"{output.stem}.revalidation{output.suffix}")
|
||
if verification == output:
|
||
verification = output.with_name(f"{output.stem}.revalidation{output.suffix}")
|
||
return output, verification
|
||
return output.with_suffix(".inventory.json"), output.with_suffix(".revalidation.json")
|
||
|
||
|
||
def _run_detection_persistence(*, cards: list[dict], inventory: dict,
|
||
inventory_path: Path, verification_path: Path,
|
||
source_root: Path | None = None,
|
||
source_texts: dict[str, str] | None = None,
|
||
checked_on: str | None = None, offline: bool = False) -> dict:
|
||
"""所有检测命令共用:生成回执文件后立即写入数据库。"""
|
||
inventory_path.parent.mkdir(parents=True, exist_ok=True)
|
||
inventory_path.write_text(json.dumps(inventory, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||
verification = build_revalidation_report(
|
||
cards, source_root=source_root, source_texts=source_texts, checked_on=checked_on,
|
||
)
|
||
# 把逻辑回执引用纳入报告哈希:同一输出路径重跑幂等,换一次检测产物则留下新批次;
|
||
# 不把本机绝对路径写进正式库。
|
||
verification["report_ref"] = f"capture://ai-flavor/{verification_path.name}"
|
||
_write_revalidation(verification_path, verification)
|
||
if offline:
|
||
return {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
|
||
return _persist_detection(
|
||
inventory=inventory, verification=verification,
|
||
inventory_path=inventory_path, verification_path=verification_path,
|
||
)
|
||
|
||
|
||
def _parser() -> argparse.ArgumentParser:
|
||
parser = argparse.ArgumentParser(description="发现并验证 AI 味案例卡")
|
||
sub = parser.add_subparsers(dest="command", required=True)
|
||
scan = sub.add_parser("scan")
|
||
scan.add_argument("path", type=Path)
|
||
scan.add_argument("--work-ref")
|
||
scan.add_argument("--license", dest="source_license", default="research_only", choices=sorted(LICENSES))
|
||
scan.add_argument("--source-kind", default="existing_work", choices=["existing_work", "public_domain", "synthetic"])
|
||
scan.add_argument("--output", type=Path, required=True)
|
||
scan.add_argument("--max-cards", type=int, default=500)
|
||
scan.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
|
||
feedback = sub.add_parser("feedback")
|
||
feedback.add_argument("--text-file", type=Path, required=True)
|
||
feedback.add_argument("--work-ref", required=True)
|
||
feedback.add_argument("--run-ref", required=True)
|
||
feedback.add_argument("--issue", required=True)
|
||
feedback.add_argument("--license", dest="source_license", default="owned", choices=sorted(LICENSES))
|
||
feedback.add_argument("--output", type=Path, required=True)
|
||
feedback.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
|
||
validate = sub.add_parser("validate")
|
||
validate.add_argument("path", type=Path)
|
||
revalidate = sub.add_parser("revalidate")
|
||
revalidate.add_argument("cards", type=Path, help="案例卡 YAML/JSON 或 inventory 报告")
|
||
revalidate.add_argument("--source-root", type=Path,
|
||
help="受控来源根目录;不传则只能验证卡片中的绝对 source_ref")
|
||
revalidate.add_argument("--output", type=Path, required=True)
|
||
revalidate.add_argument("--checked-on")
|
||
revalidate.add_argument("--offline", action="store_true", help="只生成回执,不自动写 muse-example")
|
||
propose = sub.add_parser("propose-rule")
|
||
propose.add_argument("--cards", type=Path, nargs="+", required=True)
|
||
propose.add_argument("--rule-id", required=True)
|
||
propose.add_argument("--name", required=True)
|
||
propose.add_argument("--fix-hint", default="待独立评审决定")
|
||
propose.add_argument("--layer", default="lexical", choices=["mechanical", "lexical", "structural", "density", "semantic"])
|
||
propose.add_argument("--carrier-scope", default="all", choices=["narration", "dialogue", "monologue", "in_text_carrier", "all"])
|
||
propose.add_argument("--trigger-type", default="model_judgment", choices=["regex", "handler", "density", "model_judgment"])
|
||
propose.add_argument("--pattern")
|
||
propose.add_argument("--criteria")
|
||
propose.add_argument("--handler")
|
||
propose.add_argument("--carve-out", action="append", default=[])
|
||
propose.add_argument("--function-check", action="append", default=[])
|
||
propose.add_argument("--verification", type=Path, required=True,
|
||
help="revalidate 命令生成的来源重验证回执")
|
||
propose.add_argument("--output", type=Path, required=True)
|
||
inventory = sub.add_parser("inventory")
|
||
inventory.add_argument("root", type=Path)
|
||
inventory.add_argument("--glob", default="*.txt")
|
||
inventory.add_argument("--source-root-ref")
|
||
inventory.add_argument("--work-ref-prefix", default="ref:")
|
||
inventory.add_argument("--license", dest="source_license", default="research_only", choices=sorted(LICENSES))
|
||
inventory.add_argument("--source-kind", default="existing_work", choices=["existing_work", "public_domain", "synthetic"])
|
||
inventory.add_argument("--output", type=Path, required=True)
|
||
inventory.add_argument("--max-cards", type=int, default=5000)
|
||
inventory.add_argument("--generated-on")
|
||
inventory.add_argument("--offline", action="store_true", help="只生成文件,不自动写 muse-example")
|
||
return parser
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
args = _parser().parse_args(argv)
|
||
try:
|
||
if args.command == "scan":
|
||
cards = capture_file(args.path, source_license=args.source_license, source_kind=args.source_kind,
|
||
work_ref=args.work_ref, max_cards=args.max_cards)
|
||
# 单文件扫描也遵守脱敏来源合同;重验证仍通过受控 source_root 读取真实文件。
|
||
cards = [copy.deepcopy(card) for card in cards]
|
||
for card in cards:
|
||
card["source"]["source_ref"] = args.path.name
|
||
inventory = _inventory_from_cards(cards, source_root_ref=args.path.parent.name,
|
||
source_kind=args.source_kind, source_license=args.source_license)
|
||
inventory_path, verification_path = _detection_paths(args.output)
|
||
result = _run_detection_persistence(
|
||
cards=cards, inventory=inventory, inventory_path=inventory_path,
|
||
verification_path=verification_path, source_root=args.path.parent,
|
||
checked_on=date.today().isoformat(), offline=args.offline,
|
||
)
|
||
# 保留原有 YAML 作为人工复核输入,但它不是正式权威。
|
||
write_yaml(args.output, {"schema_version": SCHEMA_VERSION, "cards": cards})
|
||
print(json.dumps({"cards": len(cards), "output": str(args.output),
|
||
"inventory": str(inventory_path), "revalidation": str(verification_path),
|
||
"persistence": result}, ensure_ascii=False))
|
||
elif args.command == "feedback":
|
||
text = args.text_file.read_text(encoding="utf-8")
|
||
card = capture_feedback(text, work_ref=args.work_ref,
|
||
run_ref=args.run_ref, issue=args.issue, source_license=args.source_license)
|
||
inventory = _inventory_from_cards([card], source_root_ref="live_feedback",
|
||
source_kind="creation_feedback", source_license=args.source_license)
|
||
inventory_path, verification_path = _detection_paths(args.output)
|
||
result = _run_detection_persistence(
|
||
cards=[card], inventory=inventory, inventory_path=inventory_path,
|
||
verification_path=verification_path,
|
||
source_texts={card["id"]: text}, checked_on=date.today().isoformat(), offline=args.offline,
|
||
)
|
||
write_yaml(args.output, {"schema_version": SCHEMA_VERSION, "cards": [card]})
|
||
print(json.dumps({"cards": 1, "output": str(args.output),
|
||
"inventory": str(inventory_path), "revalidation": str(verification_path),
|
||
"persistence": result}, ensure_ascii=False))
|
||
elif args.command == "validate":
|
||
cards = load_bundle([args.path])
|
||
print(json.dumps({"valid": True, "cards": len(cards)}, ensure_ascii=False))
|
||
elif args.command == "revalidate":
|
||
cards = load_bundle([args.cards])
|
||
report = build_revalidation_report(
|
||
cards,
|
||
source_root=args.source_root,
|
||
checked_on=args.checked_on,
|
||
)
|
||
report["report_ref"] = f"capture://ai-flavor/{args.output.name}"
|
||
_write_revalidation(args.output, report)
|
||
try:
|
||
inventory = _load_inventory_file(args.cards)
|
||
except CaseCardError:
|
||
inventory = _inventory_from_cards(
|
||
cards,
|
||
source_root_ref="manual-revalidate",
|
||
source_kind=cards[0]["source"].get("kind", "existing_work") if cards else "existing_work",
|
||
source_license=cards[0]["source"].get("license", "research_only") if cards else "research_only",
|
||
)
|
||
if not args.offline:
|
||
# 手工重验证也是检测运行;默认自动追加批次/逐卡回执,--offline 才不写库。
|
||
inventory_path = args.cards.with_name(f"{args.cards.stem}.inventory.json")
|
||
if args.cards.name.endswith(".inventory.json"):
|
||
inventory_path = args.cards
|
||
# 单卡 YAML 不是正式 inventory;先写 sidecar,保证自动落库有稳定的
|
||
# inventory hash 和可恢复引用,不要求用户再调用导入脚本。
|
||
if inventory_path != args.cards:
|
||
inventory_path.parent.mkdir(parents=True, exist_ok=True)
|
||
inventory_path.write_text(
|
||
json.dumps(inventory, ensure_ascii=False, indent=2) + "\n",
|
||
encoding="utf-8",
|
||
)
|
||
verification_path = args.output
|
||
persistence = _persist_detection(
|
||
inventory=inventory, verification=report,
|
||
inventory_path=inventory_path, verification_path=verification_path,
|
||
)
|
||
else:
|
||
persistence = {"status": "offline", "reason": "显式 --offline,未写 muse-example"}
|
||
print(json.dumps({
|
||
"cards": report["totals"]["cards"],
|
||
"verified": report["totals"]["verified"],
|
||
"stale": report["totals"]["stale"],
|
||
"unavailable": report["totals"]["unavailable"],
|
||
"card_mismatch": report["totals"]["card_mismatch"],
|
||
"usable": report["usable"],
|
||
"output": str(args.output),
|
||
"persistence": persistence,
|
||
}, ensure_ascii=False))
|
||
elif args.command == "propose-rule":
|
||
cards = load_bundle(args.cards)
|
||
verification = load_verification(args.verification)
|
||
trigger = {"type": args.trigger_type}
|
||
if args.trigger_type == "regex":
|
||
if not args.pattern:
|
||
raise CaseCardError("regex 候选必须提供 --pattern")
|
||
trigger["pattern"] = args.pattern
|
||
elif args.trigger_type == "handler":
|
||
if not args.handler:
|
||
raise CaseCardError("handler 候选必须提供 --handler")
|
||
trigger["handler"] = args.handler
|
||
elif args.trigger_type == "density":
|
||
if not args.pattern:
|
||
raise CaseCardError("density 候选必须提供 --pattern")
|
||
trigger.update({"pattern": args.pattern, "window_chars": 500, "min_hits": 3})
|
||
else:
|
||
trigger["criteria"] = args.criteria or "待独立功能判断"
|
||
rule = propose_rule(
|
||
cards, rule_id=args.rule_id, name=args.name, fix_hint=args.fix_hint,
|
||
verification=verification, layer=args.layer, carrier_scope=args.carrier_scope,
|
||
trigger=trigger, carve_out=args.carve_out, function_check=args.function_check,
|
||
)
|
||
write_yaml(args.output, rule)
|
||
print(json.dumps({"status": rule["status"], "cards": len(cards), "output": str(args.output)}, ensure_ascii=False))
|
||
else:
|
||
report = build_inventory(
|
||
args.root,
|
||
source_license=args.source_license,
|
||
source_kind=args.source_kind,
|
||
glob=args.glob,
|
||
work_ref_prefix=args.work_ref_prefix,
|
||
source_root_ref=args.source_root_ref,
|
||
max_cards=args.max_cards,
|
||
generated_on=args.generated_on,
|
||
)
|
||
inventory_path, verification_path = _detection_paths(args.output, inventory_command=True)
|
||
persistence = _run_detection_persistence(
|
||
cards=report["cards"], inventory=report,
|
||
inventory_path=args.output, verification_path=verification_path,
|
||
source_root=args.root, checked_on=report["generated_on"], offline=args.offline,
|
||
)
|
||
print(json.dumps({"books": report["totals"]["books"], "cards": report["totals"]["cards"],
|
||
"output": str(args.output), "revalidation": str(verification_path),
|
||
"persistence": persistence}, ensure_ascii=False))
|
||
return 0
|
||
except (CaseCardError, OSError, UnicodeError) as exc:
|
||
print(f"CASE_CARD_CONTRACT_FAILED: {exc}")
|
||
return 2
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|