214 lines
7.3 KiB
Python
214 lines
7.3 KiB
Python
#!/usr/bin/env python3
|
|
"""回放快照的内容级泄露审计。
|
|
|
|
审计只接收结构化目标事实和已经冻结的快照,不读取正文、不调用模型、不写数据库。
|
|
审计结果只包含路径、哈希和原因,避免把目标章事实回显到最终报告。
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
import json
|
|
from typing import Any, Iterator, Mapping
|
|
|
|
from build_snapshot import normalize_chapter, normalize_chapter_range, sha256_value
|
|
|
|
|
|
STATUS_READY = "ready"
|
|
STATUS_INVALID_SNAPSHOT = "invalid_snapshot"
|
|
STATUS_INVALID_INPUT = "invalid_audit_input"
|
|
|
|
_CHAPTER_KEYS = frozenset(
|
|
{"chapter", "chapter_no", "order_no", "章", "章号", "from_order", "to_order"}
|
|
)
|
|
|
|
|
|
def _chapter_bounds(record: Mapping[str, Any]) -> tuple[int, int] | None:
|
|
"""按快照冻结规则解析记录的绝对章号或完整章区间。"""
|
|
|
|
for key in ("chapter", "chapter_no", "order_no", "章", "章号"):
|
|
if key in record:
|
|
return normalize_chapter_range(record[key])
|
|
if "from_order" in record or "to_order" in record:
|
|
start = record.get("from_order")
|
|
end = record.get("to_order")
|
|
if start is None or end is None:
|
|
return None
|
|
return normalize_chapter_range(f"{start}-{end}")
|
|
return None
|
|
|
|
|
|
def _walk(value: Any, path: str = "$") -> Iterator[tuple[str, Any]]:
|
|
"""以稳定顺序遍历快照,供审计定位泄露字段。"""
|
|
|
|
yield path, value
|
|
if isinstance(value, Mapping):
|
|
for key in sorted(value, key=lambda item: str(item)):
|
|
yield from _walk(value[key], f"{path}.{key}")
|
|
elif isinstance(value, list):
|
|
for index, item in enumerate(value):
|
|
yield from _walk(item, f"{path}[{index}]")
|
|
|
|
|
|
def _validate_target_facts(
|
|
target_facts: Mapping[str, Any], as_of: int, target: int
|
|
) -> tuple[list[dict[str, Any]], list[str]]:
|
|
"""校验目标事实登记,不把原始事实文本放进错误信息。"""
|
|
|
|
errors: list[str] = []
|
|
registered_target = normalize_chapter(target_facts.get("targetChapter"))
|
|
if registered_target != target:
|
|
errors.append("targetFacts.targetChapter 必须与 targetChapter 一致")
|
|
|
|
raw_facts = target_facts.get("forbiddenFacts")
|
|
if not isinstance(raw_facts, list):
|
|
errors.append("targetFacts.forbiddenFacts 必须是数组")
|
|
return [], errors
|
|
|
|
facts: list[dict[str, Any]] = []
|
|
seen_ids: set[str] = set()
|
|
for index, item in enumerate(raw_facts):
|
|
if not isinstance(item, Mapping):
|
|
errors.append(f"targetFacts.forbiddenFacts[{index}] 必须是对象")
|
|
continue
|
|
fact_id = str(item.get("id") or "")
|
|
text = item.get("text")
|
|
first_chapter = normalize_chapter(item.get("firstChapter"))
|
|
if not fact_id or fact_id in seen_ids:
|
|
errors.append(f"targetFacts.forbiddenFacts[{index}].id 缺失或重复")
|
|
if not isinstance(text, str) or not text.strip():
|
|
errors.append(f"targetFacts.forbiddenFacts[{index}].text 必须是非空字符串")
|
|
if first_chapter is None or first_chapter <= as_of:
|
|
errors.append(f"targetFacts.forbiddenFacts[{index}].firstChapter 必须晚于 as_of")
|
|
if fact_id and isinstance(text, str) and text.strip() and first_chapter is not None:
|
|
seen_ids.add(fact_id)
|
|
facts.append(
|
|
{
|
|
"id": fact_id,
|
|
"text": text,
|
|
"firstChapter": first_chapter,
|
|
"factHash": sha256_value(text),
|
|
}
|
|
)
|
|
return facts, errors
|
|
|
|
|
|
def _finding(
|
|
*,
|
|
reason: str,
|
|
path: str,
|
|
value: Any,
|
|
fact: Mapping[str, Any] | None = None,
|
|
) -> dict[str, Any]:
|
|
"""构造不回显事实文本的审计发现。"""
|
|
|
|
result: dict[str, Any] = {
|
|
"reason": reason,
|
|
"path": path,
|
|
"valueHash": sha256_value(value),
|
|
}
|
|
if fact is not None:
|
|
result["factId"] = fact["id"]
|
|
result["factHash"] = fact["factHash"]
|
|
return result
|
|
|
|
|
|
def audit_snapshot(
|
|
snapshot: Mapping[str, Any],
|
|
target_facts: Mapping[str, Any],
|
|
*,
|
|
as_of: int,
|
|
target: int,
|
|
) -> dict[str, Any]:
|
|
"""检查快照是否含有未来记录或目标章内容级事实。"""
|
|
|
|
normalized_as_of = normalize_chapter(as_of)
|
|
normalized_target = normalize_chapter(target)
|
|
if normalized_as_of is None or normalized_target != normalized_as_of + 1:
|
|
return {
|
|
"status": STATUS_INVALID_INPUT,
|
|
"ok": False,
|
|
"errors": ["as_of/target 必须是连续章号"],
|
|
"warnings": [],
|
|
"findingCount": 0,
|
|
"findings": [],
|
|
}
|
|
if not isinstance(snapshot, Mapping) or not isinstance(target_facts, Mapping):
|
|
return {
|
|
"status": STATUS_INVALID_INPUT,
|
|
"ok": False,
|
|
"errors": ["snapshot/targetFacts 必须是对象"],
|
|
"warnings": [],
|
|
"findingCount": 0,
|
|
"findings": [],
|
|
}
|
|
|
|
facts, errors = _validate_target_facts(target_facts, normalized_as_of, normalized_target)
|
|
if errors:
|
|
return {
|
|
"status": STATUS_INVALID_INPUT,
|
|
"ok": False,
|
|
"errors": errors,
|
|
"warnings": [],
|
|
"findingCount": 0,
|
|
"findings": [],
|
|
}
|
|
|
|
findings: list[dict[str, Any]] = []
|
|
for path, value in _walk(copy.deepcopy(dict(snapshot))):
|
|
if isinstance(value, Mapping) and _CHAPTER_KEYS.intersection(value):
|
|
bounds = _chapter_bounds(value)
|
|
if bounds is None:
|
|
findings.append(
|
|
_finding(reason="missing_or_invalid_chapter", path=path, value=value)
|
|
)
|
|
elif bounds[1] > normalized_as_of:
|
|
findings.append(
|
|
_finding(reason="future_or_crosses_as_of", path=path, value=value)
|
|
)
|
|
if isinstance(value, str):
|
|
for fact in facts:
|
|
if fact["text"] in value:
|
|
findings.append(
|
|
_finding(reason="target_fact_match", path=path, value=value, fact=fact)
|
|
)
|
|
|
|
return {
|
|
"status": STATUS_INVALID_SNAPSHOT if findings else STATUS_READY,
|
|
"ok": not findings,
|
|
"errors": [],
|
|
"warnings": [],
|
|
"findingCount": len(findings),
|
|
"findings": findings,
|
|
}
|
|
|
|
|
|
def _parse_args() -> Any:
|
|
"""命令行参数解析保留给后续独立审计调用。"""
|
|
|
|
import argparse
|
|
|
|
parser = argparse.ArgumentParser(description="审计冻结快照的内容级目标事实泄露")
|
|
parser.add_argument("--snapshot", required=True)
|
|
parser.add_argument("--target-facts", required=True)
|
|
parser.add_argument("--as-of", type=int, required=True)
|
|
parser.add_argument("--target", type=int, required=True)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> int:
|
|
"""执行一次文件级审计,只输出安全审计结果。"""
|
|
|
|
args = _parse_args()
|
|
with open(args.snapshot, encoding="utf-8") as handle:
|
|
snapshot = json.load(handle)
|
|
with open(args.target_facts, encoding="utf-8") as handle:
|
|
target_facts = json.load(handle)
|
|
result = audit_snapshot(snapshot, target_facts, as_of=args.as_of, target=args.target)
|
|
print(json.dumps(result, ensure_ascii=False, sort_keys=True))
|
|
return 0 if result["ok"] else 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|