cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,149 @@
1
+ """InjecMEM defense — memory-injection pattern detection (Section 0.3).
2
+
3
+ A single crafted interaction must not be able to poison the Trace.
4
+ High-risk patterns quarantine the offending fact (stored, hash-chained,
5
+ but never active nor retrievable into prompt context). Medium-risk
6
+ patterns flag provenance for downstream audit.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from dataclasses import dataclass
13
+
14
+ HIGH_RISK = [
15
+ ("ignore_instructions", re.compile(
16
+ r"ignore\s+(?:all\s+|any\s+|the\s+|those\s+)?(?:previous|prior|above|earlier|preceding|past)\s+"
17
+ r"(?:instructions?|rules?|prompts?|directives?|messages?)", re.I)),
18
+ ("disregard_context", re.compile(
19
+ r"disregard\s+(?:the\s+|all\s+|any\s+)?(?:above|previous|prior|earlier|context)", re.I)),
20
+ ("system_prompt_probe", re.compile(
21
+ r"(?:reveal|show|print|repeat|output|expose|leak)\s+(?:your\s+|the\s+|this\s+)?"
22
+ r"(?:system|developer|initial)\s+(?:prompt|instructions?|message)", re.I)),
23
+ ("identity_override", re.compile(
24
+ r"you\s+(?:are\s+now|must\s+now|will\s+now|from\s+now\s+on)\s+(?:act\s+as|become|behave\s+as|forget)", re.I)),
25
+ ("jailbreak", re.compile(
26
+ r"\b(?:jailbreak|DAN\s+mode|developer\s+mode|do\s+anything\s+now)\b", re.I)),
27
+ ("exfiltration", re.compile(
28
+ r"\b(?:exfiltrate|upload|send)\s+(?:the\s+|all\s+|every\s+)?(?:memory|memories|database|records?|secrets?|api\s+keys?|credentials?|passwords?|tokens?)", re.I)),
29
+ ("credential_capture", re.compile(
30
+ r"\b(?:api\s+key|secret\s+key|access\s+token|password)\s*(?:is|:|=)\s*\S+", re.I)),
31
+ ]
32
+
33
+ MEDIUM_RISK = [
34
+ ("always_respond", re.compile(
35
+ r"always\s+(?:respond|reply|answer|say|output)\s+(?:with|in)\b", re.I)),
36
+ ("must_remember", re.compile(
37
+ r"remember\s+that\s+you\s+(?:must|always|never|are\s+no\s+longer)", re.I)),
38
+ ("memory_overwrite", re.compile(
39
+ r"(?:overwrite|replace|wipe|erase|delete)\s+(?:all\s+|the\s+|your\s+)?memory", re.I)),
40
+ ("role_injection", re.compile(
41
+ r"\b(?:from\s+now\s+on|for\s+all\s+future\s+(?:sessions?|turns?|conversations?))\b[^.]{0,80}?"
42
+ r"(?:you\s+are|act\s+as|always)", re.I)),
43
+ ]
44
+
45
+ BENIGN_EXCEPTIONS = re.compile(
46
+ r"\b(?:never\s+ignore|don'?t\s+ignore|do\s+not\s+ignore)\b", re.I)
47
+
48
+
49
+ @dataclass
50
+ class InjectionVerdict:
51
+ risk: str # "none" | "medium" | "high"
52
+ rules: list[str]
53
+ quarantined: bool
54
+ note: str = ""
55
+
56
+
57
+ def scan(text: str, quarantine_high: bool = True) -> InjectionVerdict:
58
+ hits_high = [name for name, rx in HIGH_RISK if rx.search(text)]
59
+ hits_med = [name for name, rx in MEDIUM_RISK if rx.search(text)]
60
+ # negations like "never ignore previous instructions" (user quoting policy)
61
+ if BENIGN_EXCEPTIONS.search(text):
62
+ hits_high = [r for r in hits_high if r != "ignore_instructions"]
63
+ if hits_high:
64
+ return InjectionVerdict(
65
+ risk="high", rules=hits_high, quarantined=quarantine_high,
66
+ note="InjecMEM high-risk pattern; fact quarantined (stored for audit, never active).")
67
+ if hits_med:
68
+ return InjectionVerdict(
69
+ risk="medium", rules=hits_med, quarantined=False,
70
+ note="InjecMEM medium-risk pattern; committed with audit flag.")
71
+ return InjectionVerdict(risk="none", rules=[], quarantined=False)
72
+
73
+
74
+ # ---------------------------------------------------------------------------
75
+ # Second-order defense (MINJA, arXiv:2503.03704).
76
+ #
77
+ # MINJA showed that an attacker does NOT need write access to the memory
78
+ # bank: craft a query whose *retrieved* answer gets written back by the
79
+ # agent itself, and the poison enters the Trace through the front door.
80
+ # Countermeasure: quarantined source text is itself a tainted corpus — any
81
+ # subsequent ingest that substantially overlaps it (even when punctuation
82
+ # / ordering edits defeat the regex patterns) is quarantined too
83
+ # ("contagion guard"), so poison cannot launder itself back into active
84
+ # memory through re-ingestion loops. Deep paraphrase laundering is a
85
+ # documented limitation (see docs/SECURITY.md).
86
+ # ---------------------------------------------------------------------------
87
+
88
+ _TOKEN_RE = re.compile(r"[a-z0-9']+")
89
+ _SENTENCE_RE = re.compile(r"(?<=[.!?;])\s+|\n+")
90
+
91
+
92
+ def _token_set(text: str) -> frozenset[str]:
93
+ return frozenset(_TOKEN_RE.findall(text.lower()))
94
+
95
+
96
+ def _sentences(text: str) -> list[str]:
97
+ return [s.strip() for s in _SENTENCE_RE.split(text) if s and s.strip()]
98
+
99
+
100
+ def jaccard(a: frozenset[str], b: frozenset[str]) -> float:
101
+ if not a or not b:
102
+ return 0.0
103
+ inter = len(a & b)
104
+ return inter / (len(a) + len(b) - inter)
105
+
106
+
107
+ def contagion_scan(text: str, quarantined_texts, threshold: float = 0.50,
108
+ quarantine: bool = True) -> InjectionVerdict | None:
109
+ """Return a HIGH-risk verdict if `text` substantially overlaps any
110
+ already-quarantined source text (MINJA re-ingestion loop). None if clean.
111
+
112
+ Sentence-to-sentence token Jaccard plus a verbatim-substring shortcut,
113
+ so punctuation/ordering edits that defeat the regex patterns are still
114
+ caught. Pure set arithmetic — no LLM calls, write path stays μ=0.
115
+ """
116
+ new_sents = _sentences(text)
117
+ if not new_sents or not quarantined_texts:
118
+ return None
119
+ tainted_sents: list[frozenset[str]] = []
120
+ tainted_raw: list[str] = []
121
+ for qt in quarantined_texts:
122
+ for s in _sentences(qt):
123
+ toks = _token_set(s)
124
+ if len(toks) >= 4:
125
+ tainted_sents.append(toks)
126
+ tainted_raw.append(s)
127
+ if not tainted_sents:
128
+ return None
129
+ best_sim = 0.0
130
+ for ns in new_sents:
131
+ nt = _token_set(ns)
132
+ if not nt:
133
+ continue
134
+ low = ns.lower()
135
+ # verbatim-quote shortcut (the classic MINJA write-back)
136
+ for raw in tainted_raw:
137
+ if len(raw) >= 25 and raw.lower() in low:
138
+ return InjectionVerdict(
139
+ risk="high", rules=["minja_contagion"], quarantined=quarantine,
140
+ note=("Second-order injection (MINJA): verbatim quote of "
141
+ "quarantined source; quarantined by contagion."))
142
+ for tt in tainted_sents:
143
+ best_sim = max(best_sim, jaccard(nt, tt))
144
+ if best_sim >= threshold:
145
+ return InjectionVerdict(
146
+ risk="high", rules=["minja_contagion"], quarantined=quarantine,
147
+ note=(f"Second-order injection (MINJA): {best_sim:.0%} token "
148
+ "overlap with quarantined source; quarantined by contagion."))
149
+ return None
@@ -0,0 +1,154 @@
1
+ """MIND-style retrieval diversity defense against InjecMEM.
2
+
3
+ arXiv:2608.23471 (InjecMEM, August 2026) attack taxonomy:
4
+ * Retriever-agnostic anchors — high-recall topical cues that
5
+ guarantee the poisoned record surfaces in top-k
6
+ * Adversarial commands optimized via gradient-based coordinate
7
+ search (Multi-GCG)
8
+ * Achieves 76.6% conditional ASR when poisoned records are retrieved
9
+
10
+ arXiv:2607.28103 (MIND) defense:
11
+ * Intent-aware Information Bottleneck extracts compact intent-
12
+ behavior representations from multi-turn trajectories
13
+ * Detects poisoned memories by measuring deviation between initial
14
+ user intent and subsequent behavior — exactly the signal InjecMEM
15
+ exploits
16
+ * MIND achieves the lowest ASR (19.57%) while preserving task
17
+ accuracy and running 20.6% faster than LLM auditing
18
+
19
+ We can't ship the full MIND (it requires a learned intent encoder).
20
+ But we CAN ship the SECONDARY signal MIND relies on: retrieval
21
+ diversity scoring.
22
+
23
+ InjecMEM relies on centroid anchors that CLUSTER in embedding space —
24
+ the attacker wants multiple poisoned records to surface so the LLM
25
+ sees the malicious instruction reinforced. If the top-k results are
26
+ all too similar (low intra-result diversity), that's a strong signal
27
+ of anchor-based poisoning.
28
+
29
+ This module exposes:
30
+ * retrieval_diversity(facts, embedder) — mean pairwise cosine sim
31
+ among the top-k results. High mean = low diversity = suspect.
32
+ * mind_check(facts, embedder, threshold) — returns a MINDVerdict
33
+ with .flagged=True if diversity is too low.
34
+ * augment_provenance(result, mind_verdict) — stamps the retrieval
35
+ result's provenance with the diversity score so downstream audit
36
+ dashboards can surface flagged retrievals.
37
+
38
+ This is μ=0 compatible — pure embedding math, no learned weights, no
39
+ LLM call. The existing InjecMEM (regex + contagion) and MINJA
40
+ (taint) defenses remain unchanged; MIND diversity is a third layer
41
+ that catches what they miss: gradient-optimized adversarial text
42
+ that is semantically coherent until retrieved.
43
+ """
44
+ from __future__ import annotations
45
+
46
+ from dataclasses import dataclass, field
47
+ from typing import Iterable
48
+
49
+ import numpy as np
50
+
51
+ from cortexm.bridge.rerank import fact_nl
52
+ from cortexm.trace.fact import Fact
53
+
54
+
55
+ @dataclass
56
+ class MINDVerdict:
57
+ """MIND diversity check verdict.
58
+
59
+ diversity: float in [0,1] — mean pairwise cosine sim among top-k.
60
+ 1.0 = all facts identical (maximally suspect)
61
+ 0.0 = all facts orthogonal (maximally diverse)
62
+ flagged: True if diversity > threshold (low diversity = suspect)
63
+ reason: human-readable explanation
64
+ """
65
+ diversity: float = 0.0
66
+ flagged: bool = False
67
+ threshold: float = 0.85
68
+ n_facts: int = 0
69
+ reason: str = ""
70
+ fact_ids: list[str] = field(default_factory=list)
71
+
72
+
73
+ def _cosine_matrix(embs: np.ndarray) -> np.ndarray:
74
+ """All-pairs cosine sim. Assumes embs are L2-normalized (HashingEmbedder
75
+ guarantees this). Falls back to manual normalization for safety.
76
+ """
77
+ norms = np.linalg.norm(embs, axis=1, keepdims=True)
78
+ norms = np.where(norms == 0, 1.0, norms)
79
+ embs_n = embs / norms
80
+ return embs_n @ embs_n.T
81
+
82
+
83
+ def retrieval_diversity(facts: list[Fact], embedder,
84
+ use_fact_nl: bool = True) -> float:
85
+ """Mean pairwise cosine similarity among the top-k retrieved facts.
86
+
87
+ Lower = more diverse = healthier.
88
+ Higher = more clustered = suspect (possible InjecMEM anchor).
89
+ """
90
+ if len(facts) < 2:
91
+ return 0.0
92
+ embs = []
93
+ for f in facts:
94
+ try:
95
+ text = fact_nl(f) if use_fact_nl else f.value
96
+ embs.append(embedder.embed(text))
97
+ except Exception:
98
+ v = np.zeros(getattr(embedder, "dims", 768), dtype=np.float32)
99
+ for i, ch in enumerate((f.value or "")[:768]):
100
+ v[i] = (ord(ch) % 7 - 3) / 3.0
101
+ embs.append(v)
102
+ mat = _cosine_matrix(np.stack(embs))
103
+ n = mat.shape[0]
104
+ iu = np.triu_indices(n, k=1)
105
+ return float(np.mean(mat[iu]))
106
+
107
+
108
+ def mind_check(facts: list[Fact], embedder,
109
+ threshold: float = 0.85,
110
+ min_facts: int = 2,
111
+ flag_on_low_diversity: bool = True) -> MINDVerdict:
112
+ """Run the MIND diversity check on a retrieval result."""
113
+ if len(facts) < min_facts:
114
+ return MINDVerdict(
115
+ diversity=0.0, flagged=False, threshold=threshold,
116
+ n_facts=len(facts),
117
+ reason=f"too few facts ({len(facts)} < {min_facts}) to "
118
+ f"compute diversity",
119
+ fact_ids=[f.id for f in facts])
120
+ div = retrieval_diversity(facts, embedder)
121
+ flagged = flag_on_low_diversity and div > threshold
122
+ reason = (
123
+ f"diversity={div:.3f} (threshold={threshold:.3f}) — "
124
+ + ("FLAGGED: low diversity suggests possible InjecMEM "
125
+ "anchor-based poisoning; recommend audit."
126
+ if flagged else
127
+ "OK: retrieval set is sufficiently diverse.")
128
+ )
129
+ return MINDVerdict(
130
+ diversity=div, flagged=flagged, threshold=threshold,
131
+ n_facts=len(facts), reason=reason,
132
+ fact_ids=[f.id for f in facts])
133
+
134
+
135
+ def augment_provenance(result, verdict: MINDVerdict) -> None:
136
+ """Stamp a RetrievalResult's provenance with the MIND diversity score."""
137
+ try:
138
+ if hasattr(result, "provenance") and isinstance(
139
+ result.provenance, dict):
140
+ result.provenance["mind_diversity"] = round(verdict.diversity, 4)
141
+ result.provenance["mind_flagged"] = verdict.flagged
142
+ result.provenance["mind_threshold"] = verdict.threshold
143
+ if verdict.flagged:
144
+ result.provenance["mind_reason"] = verdict.reason
145
+ except Exception:
146
+ pass
147
+
148
+
149
+ __all__ = [
150
+ "MINDVerdict",
151
+ "retrieval_diversity",
152
+ "mind_check",
153
+ "augment_provenance",
154
+ ]
@@ -0,0 +1,265 @@
1
+ """PII detection, redaction, and reversible tokenization (GDPR/CCPA).
2
+
3
+ Enterprises cannot ship a memory layer that spreads personal data across
4
+ every table. This module runs on the WRITE path (before extraction), so
5
+ raw PII never reaches the Trace, the Palace, or the vector codec — the
6
+ deterministic extractor only ever sees surrogate tokens.
7
+
8
+ Modes:
9
+ off — pass-through (development default; μ=0 benchmarks use this)
10
+ redact — replace spans with typed tokens ``«PII:EMAIL:7f3a»``; the
11
+ mapping lives in a vault (encrypted at rest when a master
12
+ key is configured) so authorised re-identification stays
13
+ possible (e.g. DSAR subject-access requests)
14
+ block — refuse the message entirely (strict exfiltration guard)
15
+
16
+ Detectors are regex + checksum based (Luhn for cards, mod-97 for IBAN,
17
+ SSN area/group rules) — zero LLM calls, keeping the μ=0 protocol intact.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+ import secrets
24
+ from dataclasses import dataclass, field
25
+
26
+ # ---------------------------------------------------------------- detectors
27
+ _LUHN_OK: set[str] = set() # memoized card prefixes that pass Luhn
28
+
29
+
30
+ def _luhn_ok(digits: str) -> bool:
31
+ total, alt = 0, False
32
+ for ch in reversed(digits):
33
+ d = ord(ch) - 48
34
+ if alt:
35
+ d *= 2
36
+ if d > 9:
37
+ d -= 9
38
+ total += d
39
+ alt = not alt
40
+ return total % 10 == 0
41
+
42
+
43
+ def _iban_ok(s: str) -> bool:
44
+ s = re.sub(r"\s", "", s).upper()
45
+ if len(s) < 15 or not s[:2].isalpha() or not s[2:].isdigit() or len(s) > 34:
46
+ return False
47
+ rearranged = s[4:] + s[:4]
48
+ total = 0
49
+ for ch in rearranged:
50
+ total = (total * (36 if ch.isalpha() else 10) +
51
+ (ord(ch) - (55 if ch.isalpha() else 48))) % 97
52
+ return total == 1
53
+
54
+
55
+ def _ssn_ok(s: str) -> bool:
56
+ digits = re.sub(r"\D", "", s)
57
+ if len(digits) != 9:
58
+ return False
59
+ area, group, serial = int(digits[0:3]), int(digits[3:5]), int(digits[5:9])
60
+ return not (area in (0, 666) or area >= 900) and group != 0 and serial != 0
61
+
62
+
63
+ @dataclass
64
+ class Span:
65
+ kind: str
66
+ start: int
67
+ end: int
68
+ text: str
69
+ token: str = ""
70
+
71
+
72
+ DETECTORS: list[tuple[str, re.Pattern]] = [
73
+ # order matters: earlier patterns claim overlapping spans first
74
+ ("EMAIL", re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")),
75
+ ("IBAN", re.compile(r"\b[A-Z]{2}\d{2}(?:\s?[A-Z0-9]{4}){2,7}\b")),
76
+ ("CREDIT_CARD", re.compile(r"\b(?:\d[ -]?){13,19}\b")),
77
+ ("PHONE", re.compile(r"(?:\+\d{1,3}[\s.-]?)?(?:\(\d{2,4}\)[\s.-]?)?"
78
+ r"\d{3,4}[\s.-]?\d{3,4}(?:[\s.-]?\d{2,4})?\b")),
79
+ ("SSN", re.compile(r"\b\d{3}-\d{2}-\d{4}\b")),
80
+ ("IP", re.compile(r"\b(?:(?:25[0-5]|2[0-4]\d|1?\d?\d)\.){3}"
81
+ r"(?:25[0-5]|2[0-4]\d|1?\d?\d)\b")),
82
+ ("API_KEY", re.compile(r"\b(?:sk-[A-Za-z0-9]{20,}|ghp_[A-Za-z0-9]{36}|"
83
+ r"gho_[A-Za-z0-9]{36}|xox[baprs]-[A-Za-z0-9-]{10,}|"
84
+ r"AKIA[0-9A-Z]{16}|AIza[0-9A-Za-z_-]{35})\b")),
85
+ ("PASSPORT", re.compile(r"\b[A-Z]{1,2}\d{6,9}\b(?![\w-])")),
86
+ ]
87
+
88
+ _CHECKS = {"CREDIT_CARD": lambda t: _luhn_ok(re.sub(r"[ -]", "", t)),
89
+ "IBAN": _iban_ok,
90
+ "SSN": _ssn_ok,
91
+ "IP": lambda t: True}
92
+
93
+
94
+ def scan(text: str) -> list[Span]:
95
+ """All validated PII spans in ``text``, left-to-right, non-overlapping."""
96
+ out: list[Span] = []
97
+ taken: list[tuple[int, int]] = []
98
+
99
+ def overlaps(s: int, e: int) -> bool:
100
+ return any(not (e <= a or s >= b) for a, b in taken)
101
+
102
+ for kind, rx in DETECTORS:
103
+ for m in rx.finditer(text):
104
+ if overlaps(m.start(), m.end()):
105
+ continue
106
+ raw = m.group(0)
107
+ check = _CHECKS.get(kind)
108
+ # card-shaped spans are claimed even on checksum failure —
109
+ # otherwise their digits get reinterpreted as a phone number
110
+ claim = kind == "CREDIT_CARD"
111
+ if claim:
112
+ taken.append((m.start(), m.end()))
113
+ if check and not check(raw):
114
+ continue
115
+ # phone sanity: needs >= 7 digits total and a separator/space or +
116
+ if kind == "PHONE":
117
+ digits = re.sub(r"\D", "", raw)
118
+ if len(digits) < 7 or len(digits) > 15:
119
+ continue
120
+ if not re.search(r"[+\s().-]", raw) and len(digits) <= 9 \
121
+ and not raw.startswith("+"):
122
+ continue # bare short number — probably an id, not a phone
123
+ out.append(Span(kind, m.start(), m.end(), raw))
124
+ taken.append((m.start(), m.end()))
125
+ out.sort(key=lambda s: s.start)
126
+ return out
127
+
128
+
129
+ # ---------------------------------------------------------------- vault
130
+ class PIIVault:
131
+ """Reversible token vault. ``token -> original`` lives in the kv store,
132
+ optionally encrypted with the master key (AES-256-GCM). Crypto-shredding
133
+ the vault key renders every token permanently unrecoverable."""
134
+
135
+ def __init__(self, store, cipher=None) -> None:
136
+ self.store = store # TraceStore (kv table)
137
+ self.cipher = cipher # cortexm.security.crypto.AESGCMCipher | None
138
+ self._counter_key = "pii:vault:counter"
139
+
140
+ def _next_id(self) -> str:
141
+ cur = int(self.store.kv_get(self._counter_key, "0") or "0")
142
+ self.store.kv_set(self._counter_key, str(cur + 1))
143
+ return f"{cur + 1:04x}"
144
+
145
+ def tokenize(self, kind: str, original: str) -> str:
146
+ """Store original, return stable token. Deterministic per original
147
+ (same value -> same token) so contradictions still resolve."""
148
+ digest = None
149
+ raw_index = self.store.kv_get("pii:vault:index") or "{}"
150
+ import json
151
+ try:
152
+ index = json.loads(raw_index)
153
+ except Exception:
154
+ index = {}
155
+ # index maps sha256-ish hash -> token id (avoid storing plaintext key)
156
+ from cortexm.security.hashes import HashProvider
157
+ h = HashProvider("blake2b").hash_text(original)[:24]
158
+ if h in index:
159
+ return f"«PII:{kind}:{index[h]}»"
160
+ tid = self._next_id()
161
+ index[h] = tid
162
+ self.store.kv_set("pii:vault:index", json.dumps(index))
163
+ payload = original
164
+ if self.cipher is not None:
165
+ payload = self.cipher.encrypt_str(original)
166
+ self.store.kv_set(f"pii:vault:{tid}", payload)
167
+ return f"«PII:{kind}:{tid}»"
168
+
169
+ def resolve(self, token: str) -> str | None:
170
+ """Token -> original (DSAR / subject-access path)."""
171
+ m = re.fullmatch(r"«PII:(\w+):([0-9a-f]+)»", token)
172
+ if not m:
173
+ return None
174
+ payload = self.store.kv_get(f"pii:vault:{m.group(2)}")
175
+ if payload is None:
176
+ return None
177
+ if self.cipher is not None and not payload.startswith("«enc:"):
178
+ return None # vault not encrypted but cipher set — mismatch
179
+ if self.cipher is not None:
180
+ return self.cipher.decrypt_str(payload)
181
+ return payload
182
+
183
+ def crypto_shred(self) -> int:
184
+ """Destroy recoverability of every vault entry (GDPR erasure).
185
+ Returns number of tokens shredded."""
186
+ import json
187
+ raw_index = self.store.kv_get("pii:vault:index") or "{}"
188
+ try:
189
+ index = json.loads(raw_index)
190
+ except Exception:
191
+ index = {}
192
+ n = 0
193
+ for _h, tid in index.items():
194
+ if self.store.kv_get(f"pii:vault:{tid}") is not None:
195
+ n += 1
196
+ self.store.kv_set(f"pii:vault:{tid}",
197
+ "«shredded»" if self.cipher is None
198
+ else self.cipher.encrypt_str("«shredded»"))
199
+ # rotate the index so old tokens cannot be re-linked
200
+ self.store.kv_set("pii:vault:index", "{}")
201
+ self.store.kv_set(self._counter_key, "0")
202
+ return n
203
+
204
+ def stats(self) -> dict:
205
+ import json
206
+ raw_index = self.store.kv_get("pii:vault:index") or "{}"
207
+ try:
208
+ n = len(json.loads(raw_index))
209
+ except Exception:
210
+ n = 0
211
+ return {"vault_entries": n, "encrypted": self.cipher is not None}
212
+
213
+
214
+ # ---------------------------------------------------------------- engine
215
+ @dataclass
216
+ class PIIResult:
217
+ mode: str
218
+ blocked: bool = False
219
+ redacted_text: str = ""
220
+ spans: list[Span] = field(default_factory=list)
221
+ tokens: list[str] = field(default_factory=list)
222
+
223
+
224
+ class PIIGuard:
225
+ """Write-path PII policy engine."""
226
+
227
+ def __init__(self, mode: str = "off", vault: PIIVault | None = None) -> None:
228
+ if mode not in ("off", "redact", "block", "tag"):
229
+ raise ValueError(f"unknown pii mode {mode!r}")
230
+ self.mode = mode
231
+ self.vault = vault
232
+
233
+ def process(self, text: str) -> PIIResult:
234
+ res = PIIResult(mode=self.mode, redacted_text=text)
235
+ if self.mode == "off" or not text:
236
+ return res
237
+ res.spans = scan(text)
238
+ if not res.spans:
239
+ return res
240
+ if self.mode == "block":
241
+ res.blocked = True
242
+ return res
243
+ if self.mode == "tag":
244
+ res.tokens = [f"{s.kind}" for s in res.spans]
245
+ return res
246
+ # redact: right-to-left replacement so indices stay valid
247
+ out = text
248
+ for s in reversed(res.spans):
249
+ if self.vault is not None:
250
+ token = self.vault.tokenize(s.kind, s.text)
251
+ s.token = token
252
+ res.tokens.append(token)
253
+ out = out[:s.start] + token + out[s.end:]
254
+ else:
255
+ out = out[:s.start] + f"«PII:{s.kind}»" + out[s.end:]
256
+ res.redacted_text = out
257
+ return res
258
+
259
+
260
+ def redact_inplace(text: str) -> str:
261
+ """Stateless helper: mask without vault (irreversible)."""
262
+ out = text
263
+ for s in reversed(scan(text)):
264
+ out = out[:s.start] + f"«PII:{s.kind}»" + out[s.end:]
265
+ return out