torusguard 0.9.5 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/.torusguard/.manifest.json +98 -79
  2. package/.torusguard/config/torusguard.json +1 -1
  3. package/.torusguard/references/csharp-security.md +41 -0
  4. package/.torusguard/references/go-security.md +41 -0
  5. package/.torusguard/references/java-security.md +40 -0
  6. package/.torusguard/references/polyglot-security-matrix.md +25 -0
  7. package/.torusguard/references/rust-security.md +40 -0
  8. package/.torusguard/rules/custom/.gitkeep +1 -0
  9. package/.torusguard/rules/custom/README.md +30 -0
  10. package/.torusguard/runs/report-latest.html +328 -0
  11. package/.torusguard/schemas/golden-recipe.schema.json +75 -0
  12. package/.torusguard/schemas/memory-context.schema.json +95 -0
  13. package/.torusguard/schemas/memory-event.schema.json +70 -0
  14. package/.torusguard/schemas/memory-pattern.schema.json +65 -0
  15. package/.torusguard/scripts/diff_guard.py +202 -15
  16. package/.torusguard/scripts/finding_scorer.py +199 -86
  17. package/.torusguard/scripts/html_reporter.py +628 -0
  18. package/.torusguard/scripts/manifest_builder.py +7 -5
  19. package/.torusguard/scripts/memory_engine.py +1334 -0
  20. package/.torusguard/scripts/monorepo_detector.py +123 -10
  21. package/.torusguard/scripts/rules_sync.py +321 -0
  22. package/.torusguard/scripts/run_manager.py +171 -89
  23. package/.torusguard/scripts/stack_detect.py +458 -26
  24. package/.torusguard/skills/torusguard/SKILL.md +4 -3
  25. package/.torusguard/skills/torusguard/bootstrap.py +344 -56
  26. package/.torusguard/skills/torusguard/references/csharp-security.md +41 -0
  27. package/.torusguard/skills/torusguard/references/go-security.md +41 -0
  28. package/.torusguard/skills/torusguard/references/java-security.md +40 -0
  29. package/.torusguard/skills/torusguard/references/polyglot-security-matrix.md +25 -0
  30. package/.torusguard/skills/torusguard/references/rust-security.md +40 -0
  31. package/.torusguard/skills/torusguard-audit/SKILL.md +25 -25
  32. package/.torusguard/skills/torusguard-harden/SKILL.md +3 -3
  33. package/.torusguard/skills/torusguard-recheck/SKILL.md +3 -3
  34. package/.torusguard/workflows/memory.md +53 -0
  35. package/README.md +307 -401
  36. package/bin/torusguard.js +204 -6
  37. package/package.json +1 -1
  38. package/skills/torusguard/SKILL.md +4 -3
  39. package/skills/torusguard/bootstrap.py +68 -4
  40. package/skills/torusguard/payload/.manifest.json +98 -79
  41. package/skills/torusguard/payload/config/torusguard.json +1 -1
  42. package/skills/torusguard/payload/references/csharp-security.md +41 -0
  43. package/skills/torusguard/payload/references/go-security.md +41 -0
  44. package/skills/torusguard/payload/references/java-security.md +40 -0
  45. package/skills/torusguard/payload/references/polyglot-security-matrix.md +25 -0
  46. package/skills/torusguard/payload/references/rust-security.md +40 -0
  47. package/skills/torusguard/payload/rules/custom/.gitkeep +1 -0
  48. package/skills/torusguard/payload/rules/custom/README.md +30 -0
  49. package/skills/torusguard/payload/schemas/golden-recipe.schema.json +75 -0
  50. package/skills/torusguard/payload/schemas/memory-context.schema.json +95 -0
  51. package/skills/torusguard/payload/schemas/memory-event.schema.json +70 -0
  52. package/skills/torusguard/payload/schemas/memory-pattern.schema.json +65 -0
  53. package/skills/torusguard/payload/scripts/diff_guard.py +202 -15
  54. package/skills/torusguard/payload/scripts/finding_scorer.py +199 -86
  55. package/skills/torusguard/payload/scripts/html_reporter.py +628 -0
  56. package/skills/torusguard/payload/scripts/manifest_builder.py +7 -5
  57. package/skills/torusguard/payload/scripts/memory_engine.py +1334 -0
  58. package/skills/torusguard/payload/scripts/monorepo_detector.py +123 -10
  59. package/skills/torusguard/payload/scripts/rules_sync.py +321 -0
  60. package/skills/torusguard/payload/scripts/run_manager.py +171 -89
  61. package/skills/torusguard/payload/scripts/stack_detect.py +458 -26
  62. package/skills/torusguard/payload/skills/torusguard/SKILL.md +4 -3
  63. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +344 -56
  64. package/skills/torusguard/payload/skills/torusguard/references/csharp-security.md +41 -0
  65. package/skills/torusguard/payload/skills/torusguard/references/go-security.md +41 -0
  66. package/skills/torusguard/payload/skills/torusguard/references/java-security.md +40 -0
  67. package/skills/torusguard/payload/skills/torusguard/references/polyglot-security-matrix.md +25 -0
  68. package/skills/torusguard/payload/skills/torusguard/references/rust-security.md +40 -0
  69. package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +25 -25
  70. package/skills/torusguard/payload/skills/torusguard-harden/SKILL.md +3 -3
  71. package/skills/torusguard/payload/skills/torusguard-recheck/SKILL.md +3 -3
  72. package/skills/torusguard/payload/workflows/memory.md +53 -0
  73. package/skills/torusguard-audit/SKILL.md +25 -25
  74. package/skills/torusguard-harden/SKILL.md +3 -3
  75. package/skills/torusguard-recheck/SKILL.md +3 -3
  76. package/.torusguard/scripts/__pycache__/diff_guard.cpython-311.pyc +0 -0
  77. package/.torusguard/scripts/__pycache__/monorepo_detector.cpython-311.pyc +0 -0
  78. package/skills/torusguard/__pycache__/bootstrap.cpython-311.pyc +0 -0
  79. package/skills/torusguard/payload/scripts/__pycache__/diff_guard.cpython-311.pyc +0 -0
  80. package/skills/torusguard/payload/scripts/__pycache__/monorepo_detector.cpython-311.pyc +0 -0
@@ -0,0 +1,1334 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ TorusGuard Adaptive Security Memory Engine (v1.0.0)
4
+ Zero-dependency persistent intelligence layer for local-first security guardrails.
5
+ Manages raw event logging, pattern distillation, confidence amplification/decay,
6
+ and token-budgeted context window computation for AI agent prompts.
7
+ """
8
+
9
+ import os
10
+ import sys
11
+ import json
12
+ import datetime
13
+ import hashlib
14
+ import uuid
15
+ import argparse
16
+ from pathlib import Path
17
+ from typing import Dict, List, Any, Optional, Tuple
18
+
19
+ # Ensure UTF-8 stdout/stderr on Windows consoles
20
+ if sys.stdout and hasattr(sys.stdout, "reconfigure"):
21
+ try:
22
+ getattr(sys.stdout, "reconfigure")(encoding="utf-8", errors="replace")
23
+ except Exception:
24
+ pass
25
+ if sys.stderr and hasattr(sys.stderr, "reconfigure"):
26
+ try:
27
+ getattr(sys.stderr, "reconfigure")(encoding="utf-8", errors="replace")
28
+ except Exception:
29
+ pass
30
+
31
+ VERSION = "1.1.0"
32
+ DEFAULT_TOKEN_BUDGET = 2000
33
+ DEFAULT_TTL_DAYS = 90
34
+ DEFAULT_DECAY_RATE = 0.15
35
+
36
+
37
+ def _utc_now() -> datetime.datetime:
38
+ """Return timezone-aware current UTC datetime."""
39
+ return datetime.datetime.now(datetime.timezone.utc)
40
+
41
+
42
+ def _utc_now_iso() -> str:
43
+ """Return ISO 8601 formatted UTC timestamp string ending with 'Z'."""
44
+ return datetime.datetime.now(datetime.timezone.utc).isoformat().replace("+00:00", "Z")
45
+
46
+
47
+ def _parse_iso_utc(ts_str: str) -> datetime.datetime:
48
+ """Parse ISO timestamp string and guarantee timezone-aware UTC datetime."""
49
+ clean = ts_str.replace("Z", "+00:00")
50
+ try:
51
+ dt = datetime.datetime.fromisoformat(clean)
52
+ except ValueError:
53
+ dt = datetime.datetime.fromisoformat(clean.split(".")[0] + "+00:00")
54
+ if dt.tzinfo is None:
55
+ dt = dt.replace(tzinfo=datetime.timezone.utc)
56
+ return dt
57
+
58
+
59
+ def find_project_root(start_dir: Optional[str] = None) -> Path:
60
+ """Detect project root directory by searching for standard repo root markers."""
61
+ current = Path(start_dir or os.getcwd()).resolve()
62
+ markers = [".git", "package.json", "pyproject.toml", "manage.py", "Pipfile", "requirements.txt", ".torusguard"]
63
+
64
+ for m in markers:
65
+ if (current / m).exists():
66
+ return current
67
+
68
+ for parent in current.parents:
69
+ for m in markers:
70
+ if (parent / m).exists():
71
+ return parent
72
+
73
+ return current
74
+
75
+
76
+ def get_memory_paths(root_dir: Optional[Path] = None) -> Dict[str, Path]:
77
+ """Return all key paths within the .torusguard/memory subsystem."""
78
+ base = Path(root_dir or find_project_root()).resolve()
79
+ torusguard_dir = base / ".torusguard"
80
+ memory_dir = torusguard_dir / "memory"
81
+ events_dir = memory_dir / "events"
82
+
83
+ return {
84
+ "root": base,
85
+ "torusguard": torusguard_dir,
86
+ "memory": memory_dir,
87
+ "events": events_dir,
88
+ "patterns": memory_dir / "patterns.json",
89
+ "context": memory_dir / "context.json",
90
+ "profile": memory_dir / "profile.json",
91
+ "decay": memory_dir / "decay.json",
92
+ "compacted": events_dir / "compacted_archive.json",
93
+ "gitignore": memory_dir / ".gitignore"
94
+ }
95
+
96
+
97
+ def ensure_memory_structure(root_dir: Optional[Path] = None) -> Dict[str, Path]:
98
+ """Initialize memory directory layout with privacy isolation."""
99
+ paths = get_memory_paths(root_dir)
100
+ paths["memory"].mkdir(parents=True, exist_ok=True)
101
+ paths["events"].mkdir(parents=True, exist_ok=True)
102
+
103
+ # Privacy belt-and-suspenders: ignore everything inside .torusguard/memory/
104
+ if not paths["gitignore"].exists():
105
+ paths["gitignore"].write_text("*\n", encoding="utf-8")
106
+
107
+ gitkeep = paths["events"] / ".gitkeep"
108
+ if not gitkeep.exists():
109
+ gitkeep.write_text("", encoding="utf-8")
110
+
111
+ if not paths["decay"].exists():
112
+ decay_init = {
113
+ "default_ttl_days": DEFAULT_TTL_DAYS,
114
+ "decay_rate": DEFAULT_DECAY_RATE,
115
+ "last_decay_run": None
116
+ }
117
+ paths["decay"].write_text(json.dumps(decay_init, indent=2), encoding="utf-8")
118
+
119
+ if not paths["patterns"].exists():
120
+ paths["patterns"].write_text("[]", encoding="utf-8")
121
+
122
+ if not paths["profile"].exists():
123
+ paths["profile"].write_text(json.dumps({
124
+ "stack": [],
125
+ "total_events": 0,
126
+ "active_patterns_count": 0,
127
+ "fix_rate_percentage": None,
128
+ "top_vulnerabilities": [],
129
+ "last_updated": _utc_now_iso()
130
+ }, indent=2), encoding="utf-8")
131
+
132
+ if not paths["context"].exists():
133
+ initial_context = {
134
+ "version": VERSION,
135
+ "generated_at": _utc_now_iso(),
136
+ "token_estimate": 0,
137
+ "max_token_budget": DEFAULT_TOKEN_BUDGET,
138
+ "project_profile": {
139
+ "stack": [],
140
+ "total_events": 0,
141
+ "active_patterns_count": 0,
142
+ "fix_rate_percentage": None,
143
+ "top_vulnerabilities": []
144
+ },
145
+ "cards": []
146
+ }
147
+ paths["context"].write_text(json.dumps(initial_context, indent=2), encoding="utf-8")
148
+
149
+ return paths
150
+
151
+
152
+ def estimate_tokens(obj: Any) -> int:
153
+ """Rough conservative token estimation for JSON payloads (1 token ≈ 4 chars)."""
154
+ serialized = json.dumps(obj, separators=(",", ":"))
155
+ return max(1, len(serialized) // 4)
156
+
157
+
158
+ def record_event(
159
+ event_type: str,
160
+ data: Dict[str, Any],
161
+ root_dir: Optional[Path] = None
162
+ ) -> Dict[str, Any]:
163
+ """
164
+ Append an individual raw event to .torusguard/memory/events/.
165
+ Event types:
166
+ - audit_finding
167
+ - fix_applied
168
+ - fix_verified
169
+ - false_positive
170
+ - pattern_learned
171
+ - stack_changed
172
+ """
173
+ valid_types = {
174
+ "audit_finding", "fix_applied", "fix_verified",
175
+ "false_positive", "pattern_learned", "stack_changed"
176
+ }
177
+ if event_type not in valid_types:
178
+ raise ValueError(f"Invalid event_type: {event_type}. Must be one of {valid_types}")
179
+
180
+ paths = ensure_memory_structure(root_dir)
181
+ now_utc = _utc_now()
182
+ timestamp_iso = _utc_now_iso()
183
+ event_id = f"evt-{now_utc.strftime('%Y%m%d%H%M%S')}-{uuid.uuid4().hex[:8]}"
184
+
185
+ # Sanitize file_path to be relative to project root
186
+ raw_file = data.get("file_path")
187
+ clean_file = None
188
+ if raw_file:
189
+ try:
190
+ rel = Path(raw_file).resolve().relative_to(paths["root"].resolve())
191
+ clean_file = str(rel).replace("\\", "/")
192
+ except Exception:
193
+ clean_file = str(raw_file).replace("\\", "/")
194
+
195
+ event = {
196
+ "event_id": event_id,
197
+ "event_type": event_type,
198
+ "timestamp": timestamp_iso,
199
+ "version": VERSION,
200
+ "rule_id": data.get("rule_id"),
201
+ "file_path": clean_file,
202
+ "line_number": data.get("line_number"),
203
+ "severity": data.get("severity"),
204
+ "confidence_score": data.get("confidence_score"),
205
+ "code_hash": data.get("code_hash"),
206
+ "fix_strategy": data.get("fix_strategy"),
207
+ "verification_result": data.get("verification_result"),
208
+ "suppression_reason": data.get("suppression_reason"),
209
+ "metadata": data.get("metadata", {})
210
+ }
211
+
212
+ # Write event file with timestamp prefix for chronological directory ordering
213
+ filename = f"{now_utc.strftime('%Y%m%d_%H%M%S')}_{event_id}.json"
214
+ event_path = paths["events"] / filename
215
+ event_path.write_text(json.dumps(event, indent=2), encoding="utf-8")
216
+
217
+ return event
218
+
219
+
220
+ def load_all_events(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
221
+ """Load all raw events from memory/events/ including compacted archive."""
222
+ paths = ensure_memory_structure(root_dir)
223
+ events: List[Dict[str, Any]] = []
224
+
225
+ # 1. Load compacted archive if present
226
+ if paths["compacted"].exists():
227
+ try:
228
+ with open(paths["compacted"], "r", encoding="utf-8") as f:
229
+ archived = json.load(f)
230
+ if isinstance(archived, list):
231
+ events.extend(archived)
232
+ except Exception:
233
+ pass
234
+
235
+ # 2. Load individual event files
236
+ if paths["events"].is_dir():
237
+ for item in sorted(paths["events"].glob("*.json")):
238
+ if item.name == "compacted_archive.json":
239
+ continue
240
+ try:
241
+ with open(item, "r", encoding="utf-8") as f:
242
+ evt = json.load(f)
243
+ if isinstance(evt, dict) and "event_id" in evt:
244
+ events.append(evt)
245
+ except Exception:
246
+ continue
247
+
248
+ # Deduplicate by event_id
249
+ seen_ids = set()
250
+ deduped = []
251
+ for evt in events:
252
+ eid = evt.get("event_id")
253
+ if eid and eid not in seen_ids:
254
+ seen_ids.add(eid)
255
+ deduped.append(evt)
256
+
257
+ # Sort chronologically
258
+ deduped.sort(key=lambda x: x.get("timestamp", ""))
259
+ return deduped
260
+
261
+
262
+ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
263
+ """
264
+ Distill raw events into actionable, deduplicated security patterns.
265
+ Pattern types:
266
+ - recurring_fix: Repeated successful fixes for a rule
267
+ - common_vulnerability: Vulnerabilities appearing frequently
268
+ - false_positive_class: Suppressed rules/files
269
+ - regression_watch: Rules/files that regressed or re-appeared
270
+ - security_idiom: Project-specific established remediation practices
271
+ """
272
+ paths = ensure_memory_structure(root_dir)
273
+ events = load_all_events(root_dir)
274
+ patterns: List[Dict[str, Any]] = []
275
+
276
+ # Preserve existing golden_fix_recipe patterns
277
+ if paths["patterns"].exists():
278
+ try:
279
+ prev_patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
280
+ for p in prev_patterns:
281
+ if p.get("pattern_type") == "golden_fix_recipe":
282
+ patterns.append(p)
283
+ except Exception:
284
+ pass
285
+
286
+ if not events and not patterns:
287
+ paths["patterns"].write_text("[]", encoding="utf-8")
288
+ compute_context_window(root_dir=root_dir)
289
+ return patterns
290
+
291
+ # Grouping indices
292
+ rule_findings: Dict[str, List[Dict[str, Any]]] = {}
293
+ rule_fixes: Dict[Tuple[str, str], List[Dict[str, Any]]] = {}
294
+ false_positives: Dict[str, List[Dict[str, Any]]] = {}
295
+ regressions: Dict[str, List[Dict[str, Any]]] = {}
296
+
297
+ for evt in events:
298
+ etype = evt.get("event_type")
299
+ rule_id = evt.get("rule_id")
300
+ if not rule_id:
301
+ continue
302
+
303
+ if etype == "audit_finding":
304
+ rule_findings.setdefault(rule_id, []).append(evt)
305
+ elif etype == "fix_applied":
306
+ strat = evt.get("fix_strategy") or "standard_remediation"
307
+ rule_fixes.setdefault((rule_id, strat), []).append(evt)
308
+ elif etype == "fix_verified":
309
+ strat = evt.get("fix_strategy") or "standard_remediation"
310
+ vres = evt.get("verification_result")
311
+ if vres == "regressed":
312
+ regressions.setdefault(rule_id, []).append(evt)
313
+ else:
314
+ rule_fixes.setdefault((rule_id, strat), []).append(evt)
315
+ elif etype == "false_positive":
316
+ false_positives.setdefault(rule_id, []).append(evt)
317
+
318
+ pat_idx = 1
319
+
320
+ # 1. Distill Recurring Fixes & Security Idioms
321
+ for (rule_id, fix_strat), fix_evts in rule_fixes.items():
322
+ occurrences = len(fix_evts)
323
+ affected_files = sorted(list({str(e["file_path"]) for e in fix_evts if e.get("file_path")}))
324
+ verified_count = sum(1 for e in fix_evts if e.get("verification_result") == "fixed")
325
+
326
+ # Confidence amplification logic
327
+ base_confidence = min(95, 50 + (occurrences * 10) + (verified_count * 10))
328
+ # Multi-file bonus
329
+ if len(affected_files) > 1:
330
+ base_confidence = min(98, base_confidence + 5)
331
+
332
+ pattern_type = "security_idiom" if occurrences >= 3 and verified_count >= 2 else "recurring_fix"
333
+ timestamps: List[str] = [str(e["timestamp"]) for e in fix_evts if e.get("timestamp")]
334
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
335
+ last_seen = max(timestamps) if timestamps else first_seen
336
+
337
+ patterns.append({
338
+ "pattern_id": f"PAT-{pat_idx:03d}",
339
+ "rule_id": rule_id,
340
+ "pattern_type": pattern_type,
341
+ "description": f"Verified remediation strategy for {rule_id}: {fix_strat}",
342
+ "fix_strategy": fix_strat,
343
+ "confidence": base_confidence,
344
+ "occurrences": occurrences,
345
+ "affected_files": affected_files,
346
+ "first_seen": first_seen,
347
+ "last_seen": last_seen,
348
+ "decay_checkpoint": last_seen,
349
+ "source_events": [e.get("event_id") for e in fix_evts if e.get("event_id")]
350
+ })
351
+ pat_idx += 1
352
+
353
+ # 2. Distill Common Vulnerabilities
354
+ for rule_id, find_evts in rule_findings.items():
355
+ occurrences = len(find_evts)
356
+ if occurrences >= 2:
357
+ affected_files = sorted(list({str(e["file_path"]) for e in find_evts if e.get("file_path")}))
358
+ timestamps: List[str] = [str(e["timestamp"]) for e in find_evts if e.get("timestamp")]
359
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
360
+ last_seen = max(timestamps) if timestamps else first_seen
361
+
362
+ confidence = min(90, 55 + (occurrences * 5))
363
+ patterns.append({
364
+ "pattern_id": f"PAT-{pat_idx:03d}",
365
+ "rule_id": rule_id,
366
+ "pattern_type": "common_vulnerability",
367
+ "description": f"Frequent vulnerability pattern detected across project for {rule_id}",
368
+ "fix_strategy": None,
369
+ "confidence": confidence,
370
+ "occurrences": occurrences,
371
+ "affected_files": affected_files,
372
+ "first_seen": first_seen,
373
+ "last_seen": last_seen,
374
+ "decay_checkpoint": last_seen,
375
+ "source_events": [e.get("event_id") for e in find_evts if e.get("event_id")]
376
+ })
377
+ pat_idx += 1
378
+
379
+ # 3. Distill False Positive Classes
380
+ for rule_id, fp_evts in false_positives.items():
381
+ occurrences = len(fp_evts)
382
+ affected_files = sorted(list({str(e["file_path"]) for e in fp_evts if e.get("file_path")}))
383
+ reasons = [e.get("suppression_reason") for e in fp_evts if e.get("suppression_reason")]
384
+ summary_reason = reasons[-1] if reasons else "Suppressed by team policy"
385
+ timestamps: List[str] = [str(e["timestamp"]) for e in fp_evts if e.get("timestamp")]
386
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
387
+ last_seen = max(timestamps) if timestamps else first_seen
388
+
389
+ patterns.append({
390
+ "pattern_id": f"PAT-{pat_idx:03d}",
391
+ "rule_id": rule_id,
392
+ "pattern_type": "false_positive_class",
393
+ "description": f"Rule {rule_id} marked as false positive ({summary_reason})",
394
+ "fix_strategy": None,
395
+ "confidence": 90,
396
+ "occurrences": occurrences,
397
+ "affected_files": affected_files,
398
+ "first_seen": first_seen,
399
+ "last_seen": last_seen,
400
+ "decay_checkpoint": last_seen,
401
+ "source_events": [e.get("event_id") for e in fp_evts if e.get("event_id")]
402
+ })
403
+ pat_idx += 1
404
+
405
+ # 4. Distill Regression Watch Entries
406
+ for rule_id, reg_evts in regressions.items():
407
+ occurrences = len(reg_evts)
408
+ affected_files = sorted(list({str(e["file_path"]) for e in reg_evts if e.get("file_path")}))
409
+ timestamps: List[str] = [str(e["timestamp"]) for e in reg_evts if e.get("timestamp")]
410
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
411
+ last_seen = max(timestamps) if timestamps else first_seen
412
+
413
+ patterns.append({
414
+ "pattern_id": f"PAT-{pat_idx:03d}",
415
+ "rule_id": rule_id,
416
+ "pattern_type": "regression_watch",
417
+ "description": f"High risk regression watch: {rule_id} re-occurred or failed verification",
418
+ "fix_strategy": None,
419
+ "confidence": 88,
420
+ "occurrences": occurrences,
421
+ "affected_files": affected_files,
422
+ "first_seen": first_seen,
423
+ "last_seen": last_seen,
424
+ "decay_checkpoint": last_seen,
425
+ "source_events": [e.get("event_id") for e in reg_evts if e.get("event_id")]
426
+ })
427
+ pat_idx += 1
428
+
429
+ # Save patterns
430
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
431
+
432
+ # Update profile and pre-computed context window
433
+ get_project_profile(root_dir=root_dir)
434
+ compute_context_window(root_dir=root_dir)
435
+
436
+ return patterns
437
+
438
+
439
+ def decay_stale_entries(
440
+ ttl_days: int = DEFAULT_TTL_DAYS,
441
+ decay_rate: float = DEFAULT_DECAY_RATE,
442
+ root_dir: Optional[Path] = None
443
+ ) -> int:
444
+ """
445
+ Apply TTL decay to patterns not reconfirmed within ttl_days.
446
+ Reduces confidence to prevent stale architectural advice.
447
+ """
448
+ paths = ensure_memory_structure(root_dir)
449
+ now = _utc_now()
450
+
451
+ # Read config from decay.json if available
452
+ try:
453
+ if paths["decay"].exists():
454
+ cfg = json.loads(paths["decay"].read_text(encoding="utf-8"))
455
+ ttl_days = cfg.get("default_ttl_days", ttl_days)
456
+ decay_rate = cfg.get("decay_rate", decay_rate)
457
+ except Exception:
458
+ pass
459
+
460
+ if not paths["patterns"].exists():
461
+ return 0
462
+
463
+ try:
464
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
465
+ except Exception:
466
+ return 0
467
+
468
+ decayed_count = 0
469
+ for pat in patterns:
470
+ chk_str = pat.get("decay_checkpoint") or pat.get("last_seen")
471
+ if not chk_str:
472
+ continue
473
+ try:
474
+ last_dt = _parse_iso_utc(chk_str)
475
+ days_elapsed = (now - last_dt).days
476
+
477
+ if days_elapsed >= ttl_days:
478
+ old_conf = pat.get("confidence", 50)
479
+ reduction = max(5, int(old_conf * decay_rate))
480
+ new_conf = max(10, old_conf - reduction)
481
+ if new_conf != old_conf:
482
+ pat["confidence"] = new_conf
483
+ pat["decay_checkpoint"] = _utc_now_iso()
484
+ decayed_count += 1
485
+ except Exception:
486
+ continue
487
+
488
+ if decayed_count > 0:
489
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
490
+ compute_context_window(root_dir=root_dir)
491
+
492
+ # Record decay run in decay.json
493
+ try:
494
+ decay_cfg = {
495
+ "default_ttl_days": ttl_days,
496
+ "decay_rate": decay_rate,
497
+ "last_decay_run": now.isoformat() + "Z",
498
+ "last_decayed_count": decayed_count
499
+ }
500
+ paths["decay"].write_text(json.dumps(decay_cfg, indent=2), encoding="utf-8")
501
+ except Exception:
502
+ pass
503
+
504
+ return decayed_count
505
+
506
+
507
+ def get_project_profile(root_dir: Optional[Path] = None) -> Dict[str, Any]:
508
+ """Compute and update the project's security DNA profile."""
509
+ paths = ensure_memory_structure(root_dir)
510
+ events = load_all_events(root_dir)
511
+
512
+ # Detect stack from torusguard.json if available
513
+ detected_stack: List[str] = []
514
+ config_file = paths["torusguard"] / "config" / "torusguard.json"
515
+ if config_file.exists():
516
+ try:
517
+ cfg = json.loads(config_file.read_text(encoding="utf-8"))
518
+ stk = cfg.get("detected_stack", {})
519
+ for k in ("language", "framework", "data_layer"):
520
+ v = stk.get(k)
521
+ if v and v not in ("None", "Unknown"):
522
+ detected_stack.append(v)
523
+ except Exception:
524
+ pass
525
+
526
+ # Count statistics
527
+ findings_count = 0
528
+ fixes_count = 0
529
+ verified_fixed_count = 0
530
+ vuln_counter: Dict[str, int] = {}
531
+
532
+ for evt in events:
533
+ etype = evt.get("event_type")
534
+ rid = evt.get("rule_id")
535
+ if etype == "audit_finding":
536
+ findings_count += 1
537
+ if rid:
538
+ vuln_counter[rid] = vuln_counter.get(rid, 0) + 1
539
+ elif etype == "fix_applied":
540
+ fixes_count += 1
541
+ elif etype == "fix_verified":
542
+ if evt.get("verification_result") == "fixed":
543
+ verified_fixed_count += 1
544
+
545
+ fix_rate = None
546
+ if findings_count > 0:
547
+ fix_rate = round((verified_fixed_count / findings_count) * 100, 1)
548
+
549
+ top_vulns = [item[0] for item in sorted(vuln_counter.items(), key=lambda x: x[1], reverse=True)[:5]]
550
+
551
+ # Count active patterns
552
+ active_patterns_count = 0
553
+ if paths["patterns"].exists():
554
+ try:
555
+ pats = json.loads(paths["patterns"].read_text(encoding="utf-8"))
556
+ active_patterns_count = len(pats)
557
+ except Exception:
558
+ pass
559
+
560
+ profile = {
561
+ "stack": detected_stack,
562
+ "total_events": len(events),
563
+ "active_patterns_count": active_patterns_count,
564
+ "findings_count": findings_count,
565
+ "fixes_applied_count": fixes_count,
566
+ "fixes_verified_count": verified_fixed_count,
567
+ "fix_rate_percentage": fix_rate,
568
+ "top_vulnerabilities": top_vulns,
569
+ "last_updated": _utc_now_iso()
570
+ }
571
+ paths["profile"].write_text(json.dumps(profile, indent=2), encoding="utf-8")
572
+ return profile
573
+
574
+
575
+ def compute_proximity_score(
576
+ pattern: Dict[str, Any],
577
+ target_file: Optional[str] = None,
578
+ target_rule_id: Optional[str] = None
579
+ ) -> int:
580
+ """
581
+ Compute file proximity and rule relevance score (0-100) for a memory pattern against a target query.
582
+ Enables file-scoped and rule-scoped context ranking.
583
+ """
584
+ score = 0
585
+ if not target_file and not target_rule_id:
586
+ return score
587
+
588
+ pat_rule = pattern.get("rule_id", "")
589
+ affected_files = [str(f).replace("\\", "/") for f in pattern.get("affected_files", [])]
590
+
591
+ # 1. Rule ID relevance: exact match = +40, family match (e.g. TG-DB-) = +20
592
+ if target_rule_id:
593
+ if pat_rule == target_rule_id:
594
+ score += 40
595
+ elif pat_rule and pat_rule.split("-")[:2] == target_rule_id.split("-")[:2]:
596
+ score += 20
597
+
598
+ # 2. File path relevance
599
+ if target_file:
600
+ norm_target = str(target_file).replace("\\", "/")
601
+ target_path = Path(norm_target)
602
+ target_ext = target_path.suffix.lower()
603
+ target_parts = set(p.lower() for p in target_path.parts if p not in (".", ".."))
604
+
605
+ best_file_score = 0
606
+ for aff in affected_files:
607
+ aff_score = 0
608
+ aff_path = Path(aff)
609
+ if aff == norm_target:
610
+ best_file_score = 50
611
+ break
612
+ # Same directory or subpath
613
+ if aff_path.parent == target_path.parent and str(target_path.parent) not in (".", ""):
614
+ aff_score = max(aff_score, 30)
615
+ elif any(part in target_parts for part in (p.lower() for p in aff_path.parts if p not in (".", ".."))):
616
+ aff_score = max(aff_score, 15)
617
+ # Extension match
618
+ if target_ext and aff_path.suffix.lower() == target_ext:
619
+ aff_score = max(aff_score, aff_score + 10)
620
+
621
+ if aff_score > best_file_score:
622
+ best_file_score = aff_score
623
+
624
+ # Also check file_type in recipe or pattern metadata
625
+ pat_file_type = pattern.get("file_type") or pattern.get("recipe_data", {}).get("file_type")
626
+ if pat_file_type and target_ext and pat_file_type.lower() == target_ext and best_file_score == 0:
627
+ best_file_score = 10
628
+
629
+ score += best_file_score
630
+
631
+ return min(score, 100)
632
+
633
+
634
+ def record_golden_recipe(
635
+ rule_id: str,
636
+ before_snippet: str,
637
+ after_snippet: str,
638
+ diff_snippet: str,
639
+ file_type: str = ".py",
640
+ framework: Optional[str] = None,
641
+ description: str = "Verified AST remediation recipe",
642
+ additions: Optional[int] = None,
643
+ deletions: Optional[int] = None,
644
+ root_dir: Optional[Path] = None
645
+ ) -> Dict[str, Any]:
646
+ """
647
+ Store or update a verified Golden Fix Recipe adhering to Ponytail bounds (<=35 add, <=25 del).
648
+ """
649
+ paths = ensure_memory_structure(root_dir)
650
+
651
+ if additions is None:
652
+ additions = sum(1 for line in diff_snippet.splitlines() if line.startswith("+") and not line.startswith("+++"))
653
+ if deletions is None:
654
+ deletions = sum(1 for line in diff_snippet.splitlines() if line.startswith("-") and not line.startswith("---"))
655
+
656
+ if additions > 35 or deletions > 25:
657
+ raise ValueError(f"Recipe exceeds Ponytail bounds: +{additions}/35 add, -{deletions}/25 del")
658
+
659
+ recipe_hash = hashlib.sha256(diff_snippet.strip().encode("utf-8")).hexdigest()[:8]
660
+ recipe_id = f"recipe-{rule_id}-{recipe_hash}"
661
+ timestamp = _utc_now_iso()
662
+
663
+ recipe = {
664
+ "recipe_id": recipe_id,
665
+ "rule_id": rule_id,
666
+ "file_type": file_type,
667
+ "framework": framework,
668
+ "description": description,
669
+ "before_snippet": before_snippet.strip(),
670
+ "after_snippet": after_snippet.strip(),
671
+ "diff_snippet": diff_snippet.strip(),
672
+ "ponytail_metrics": {
673
+ "additions": additions,
674
+ "deletions": deletions
675
+ },
676
+ "verified_count": 1,
677
+ "last_verified": timestamp
678
+ }
679
+
680
+ patterns: List[Dict[str, Any]] = []
681
+ if paths["patterns"].exists():
682
+ try:
683
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
684
+ except Exception:
685
+ patterns = []
686
+
687
+ for p in patterns:
688
+ if p.get("pattern_type") == "golden_fix_recipe" and p.get("recipe_id") == recipe_id:
689
+ count = p.get("verified_count", 1) + 1
690
+ p["verified_count"] = count
691
+ recipe["verified_count"] = count
692
+ p["last_verified"] = timestamp
693
+ p["confidence"] = min(99, p.get("confidence", 85) + 5)
694
+ p["recipe_data"] = recipe
695
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
696
+ return p
697
+
698
+ recipe_pattern: Dict[str, Any] = {
699
+ "pattern_id": f"PAT-RECIPE-{recipe_id}",
700
+ "rule_id": rule_id,
701
+ "pattern_type": "golden_fix_recipe",
702
+ "recipe_id": recipe_id,
703
+ "file_type": file_type,
704
+ "framework": framework,
705
+ "description": description,
706
+ "recipe_data": recipe,
707
+ "confidence": 90,
708
+ "occurrences": 1,
709
+ "affected_files": [],
710
+ "first_seen": timestamp,
711
+ "last_seen": timestamp
712
+ }
713
+ patterns.append(recipe_pattern)
714
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
715
+ return recipe_pattern
716
+
717
+
718
+ def compute_context_window(
719
+ max_tokens: int = DEFAULT_TOKEN_BUDGET,
720
+ target_role: str = "all",
721
+ target_file: Optional[str] = None,
722
+ rule_id: Optional[str] = None,
723
+ target_rule_id: Optional[str] = None,
724
+ root_dir: Optional[Path] = None
725
+ ) -> Dict[str, Any]:
726
+ """
727
+ Build the pre-computed or role-tailored context window as structured JSON cards.
728
+ Guarantees strict token enforcement <= max_tokens with proximity and role weighting.
729
+ """
730
+ rule_id = target_rule_id or rule_id
731
+ paths = ensure_memory_structure(root_dir)
732
+ profile = get_project_profile(root_dir=root_dir)
733
+
734
+ patterns: List[Dict[str, Any]] = []
735
+ if paths["patterns"].exists():
736
+ try:
737
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
738
+ except Exception:
739
+ patterns = []
740
+
741
+ # Persona-tailored base priority matrices
742
+ role_priority_matrices = {
743
+ "all": {
744
+ "regression_watch": 95,
745
+ "golden_fix_recipe": 90,
746
+ "security_idiom": 85,
747
+ "recurring_fix": 80,
748
+ "false_positive_class": 75,
749
+ "common_vulnerability": 70
750
+ },
751
+ "auditor": {
752
+ "false_positive_class": 100,
753
+ "common_vulnerability": 95,
754
+ "regression_watch": 90,
755
+ "recurring_fix": 80,
756
+ "security_idiom": 60,
757
+ "golden_fix_recipe": 40
758
+ },
759
+ "remediator": {
760
+ "golden_fix_recipe": 115,
761
+ "security_idiom": 100,
762
+ "recurring_fix": 90,
763
+ "regression_watch": 80,
764
+ "common_vulnerability": 60,
765
+ "false_positive_class": 50
766
+ },
767
+ "reviewer": {
768
+ "regression_watch": 115,
769
+ "false_positive_class": 95,
770
+ "golden_fix_recipe": 85,
771
+ "recurring_fix": 75,
772
+ "security_idiom": 65,
773
+ "common_vulnerability": 55
774
+ }
775
+ }
776
+
777
+ type_priority = role_priority_matrices.get(target_role, role_priority_matrices["all"])
778
+
779
+ # Generate Candidate Cards
780
+ cards: List[Dict[str, Any]] = []
781
+ card_idx = 1
782
+
783
+ # 1. Project Profile Card (Always top priority)
784
+ cards.append({
785
+ "card_id": f"CARD-{card_idx:03d}",
786
+ "card_type": "profile",
787
+ "type": "profile",
788
+ "priority": 150,
789
+ "proximity_score": 0,
790
+ "title": "Project Security Posture & DNA",
791
+ "summary": f"Stack: {', '.join(profile.get('stack') or ['Generic'])}; Total Events: {profile.get('total_events', 0)}; Fix Rate: {profile.get('fix_rate_percentage')}%",
792
+ "card_data": {
793
+ "stack": profile.get("stack", []),
794
+ "total_events": profile.get("total_events", 0),
795
+ "fix_rate_percentage": profile.get("fix_rate_percentage"),
796
+ "top_vulnerabilities": profile.get("top_vulnerabilities", []),
797
+ "target_role": target_role
798
+ }
799
+ })
800
+ card_idx += 1
801
+
802
+ # Score patterns: base priority + proximity score
803
+ scored_patterns = []
804
+ for pat in patterns:
805
+ ptype = pat.get("pattern_type", "pattern")
806
+ base_prio = type_priority.get(ptype, 50)
807
+ proximity = compute_proximity_score(pat, target_file=target_file, target_rule_id=rule_id)
808
+ effective_prio = base_prio + proximity
809
+ scored_patterns.append((effective_prio, proximity, pat))
810
+
811
+ # Sort patterns by effective priority -> confidence -> occurrences
812
+ scored_patterns.sort(
813
+ key=lambda x: (
814
+ x[0],
815
+ x[2].get("confidence", 0),
816
+ x[2].get("occurrences", 0)
817
+ ),
818
+ reverse=True
819
+ )
820
+
821
+ for eff_prio, prox_score, pat in scored_patterns:
822
+ ptype = pat.get("pattern_type", "pattern")
823
+ card_type = "pattern"
824
+ title = f"[{pat.get('rule_id')}] {pat.get('pattern_type')}: {pat.get('description', '')[:60]}"
825
+ summary = pat.get("description", "")
826
+ card_data: Dict[str, Any] = {
827
+ "rule_id": pat.get("rule_id"),
828
+ "pattern_type": pat.get("pattern_type"),
829
+ "confidence": pat.get("confidence"),
830
+ "occurrences": pat.get("occurrences"),
831
+ "affected_files": pat.get("affected_files", [])[:5]
832
+ }
833
+
834
+ if ptype == "regression_watch":
835
+ card_type = "regression_watch"
836
+ elif ptype == "false_positive_class":
837
+ card_type = "false_positive_suppression"
838
+ elif ptype == "common_vulnerability":
839
+ card_type = "common_vulnerability"
840
+ elif ptype == "security_idiom":
841
+ card_type = "fix_idiom"
842
+ card_data["fix_strategy"] = pat.get("fix_strategy")
843
+ elif ptype == "golden_fix_recipe":
844
+ card_type = "golden_recipe"
845
+ recipe_obj = pat.get("recipe_data", {})
846
+ title = f"[{pat.get('rule_id')}] Golden Recipe: {pat.get('description', '')[:50]}"
847
+ card_data["diff_snippet"] = recipe_obj.get("diff_snippet", "")
848
+ card_data["ponytail_metrics"] = recipe_obj.get("ponytail_metrics", {})
849
+ card_data["framework"] = recipe_obj.get("framework")
850
+
851
+ cards.append({
852
+ "card_id": f"CARD-{card_idx:03d}",
853
+ "card_type": card_type,
854
+ "type": card_type,
855
+ "priority": eff_prio,
856
+ "proximity_score": prox_score,
857
+ "title": title,
858
+ "summary": summary,
859
+ "card_data": card_data
860
+ })
861
+ card_idx += 1
862
+
863
+ # Build Context Window Payload
864
+ context = {
865
+ "version": VERSION,
866
+ "generated_at": _utc_now_iso(),
867
+ "role": target_role,
868
+ "target_role": target_role,
869
+ "target_query": {
870
+ "file": target_file,
871
+ "rule_id": rule_id
872
+ },
873
+ "target_file": target_file,
874
+ "rule_id": rule_id,
875
+ "token_estimate": 0,
876
+ "max_token_budget": max_tokens,
877
+ "project_profile": {
878
+ "stack": profile.get("stack", []),
879
+ "total_events": profile.get("total_events", 0),
880
+ "active_patterns_count": profile.get("active_patterns_count", 0),
881
+ "fix_rate_percentage": profile.get("fix_rate_percentage"),
882
+ "top_vulnerabilities": profile.get("top_vulnerabilities", [])
883
+ },
884
+ "cards": cards
885
+ }
886
+
887
+ # Token budget enforcement: Drop lowest-priority cards until within limit
888
+ while len(context["cards"]) > 1 and estimate_tokens(context) > max_tokens:
889
+ context["cards"].pop()
890
+
891
+ context["token_estimate"] = estimate_tokens(context)
892
+
893
+ # Persist default context.json if generating for role="all" without filters
894
+ if target_role == "all" and not target_file and not rule_id:
895
+ paths["context"].write_text(json.dumps(context, indent=2), encoding="utf-8")
896
+
897
+ return context
898
+
899
+
900
+ def get_context(
901
+ target_role: str = "all",
902
+ target_file: Optional[str] = None,
903
+ rule_id: Optional[str] = None,
904
+ max_tokens: int = DEFAULT_TOKEN_BUDGET,
905
+ root_dir: Optional[Path] = None
906
+ ) -> Dict[str, Any]:
907
+ """Retrieve or compute persona-tailored and file-scoped context window."""
908
+ if target_role == "all" and not target_file and not rule_id:
909
+ paths = ensure_memory_structure(root_dir)
910
+ if paths["context"].exists():
911
+ try:
912
+ data = json.loads(paths["context"].read_text(encoding="utf-8"))
913
+ if data.get("version") == VERSION:
914
+ return data
915
+ except Exception:
916
+ pass
917
+
918
+ return compute_context_window(
919
+ max_tokens=max_tokens,
920
+ target_role=target_role,
921
+ target_file=target_file,
922
+ rule_id=rule_id,
923
+ root_dir=root_dir
924
+ )
925
+
926
+
927
+ def record_false_positive(
928
+ rule_id: str,
929
+ file_path: Optional[str] = None,
930
+ reason: str = "False positive verified by user",
931
+ root_dir: Optional[Path] = None
932
+ ) -> Dict[str, Any]:
933
+ """Record a false positive suppression and update patterns."""
934
+ evt = record_event(
935
+ "false_positive",
936
+ {
937
+ "rule_id": rule_id,
938
+ "file_path": file_path,
939
+ "suppression_reason": reason
940
+ },
941
+ root_dir=root_dir
942
+ )
943
+ distill_patterns(root_dir=root_dir)
944
+ return evt
945
+
946
+
947
+ def export_memory(
948
+ target_path: Optional[Any] = None,
949
+ target_file: Optional[Any] = None,
950
+ sanitized: bool = False,
951
+ root_dir: Optional[Path] = None
952
+ ) -> Dict[str, Any]:
953
+ """
954
+ Export memory to an external file for team sharing.
955
+ Supports sanitized mode for safe repository version control (strips absolute paths & secrets).
956
+ """
957
+ paths = ensure_memory_structure(root_dir)
958
+ events = load_all_events(root_dir)
959
+ profile = get_project_profile(root_dir=root_dir)
960
+
961
+ patterns: List[Dict[str, Any]] = []
962
+ if paths["patterns"].exists():
963
+ try:
964
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
965
+ except Exception:
966
+ pass
967
+
968
+ decay_cfg: Dict[str, Any] = {}
969
+ if paths["decay"].exists():
970
+ try:
971
+ decay_cfg = json.loads(paths["decay"].read_text(encoding="utf-8"))
972
+ except Exception:
973
+ pass
974
+
975
+ if sanitized:
976
+ sanitized_events = []
977
+ for e in events:
978
+ ce = dict(e)
979
+ if "file_path" in ce and ce["file_path"]:
980
+ ce["file_path"] = Path(ce["file_path"]).name
981
+ sanitized_events.append(ce)
982
+ events = sanitized_events
983
+
984
+ sanitized_patterns = []
985
+ for p in patterns:
986
+ cp = dict(p)
987
+ if "affected_files" in cp:
988
+ cp["affected_files"] = [Path(f).name for f in cp["affected_files"]]
989
+ sanitized_patterns.append(cp)
990
+ patterns = sanitized_patterns
991
+
992
+ export_payload = {
993
+ "format": "torusguard-memory-bundle",
994
+ "schema_version": VERSION,
995
+ "sanitized": sanitized,
996
+ "exported_at": _utc_now_iso(),
997
+ "project_profile": profile,
998
+ "decay_config": decay_cfg,
999
+ "patterns": patterns,
1000
+ "events": events
1001
+ }
1002
+
1003
+ dest = target_file or target_path or "torusguard-memory-export.json"
1004
+ out_file = Path(dest).resolve()
1005
+ out_file.parent.mkdir(parents=True, exist_ok=True)
1006
+ out_file.write_text(json.dumps(export_payload, indent=2), encoding="utf-8")
1007
+
1008
+ return {
1009
+ "target_path": str(out_file),
1010
+ "sanitized": sanitized,
1011
+ "exported_events_count": len(events),
1012
+ "exported_patterns_count": len(patterns)
1013
+ }
1014
+
1015
+
1016
+ def import_memory(source_path: str, merge: bool = True, root_dir: Optional[Path] = None) -> Dict[str, Any]:
1017
+ """Import and optionally merge external memory events and patterns."""
1018
+ paths = ensure_memory_structure(root_dir)
1019
+ src = Path(source_path).resolve()
1020
+ if not src.is_file():
1021
+ raise FileNotFoundError(f"Export file not found: {source_path}")
1022
+
1023
+ with open(src, "r", encoding="utf-8") as f:
1024
+ payload = json.load(f)
1025
+
1026
+ imported_events = payload.get("events", [])
1027
+ if not isinstance(imported_events, list):
1028
+ raise ValueError("Malformed import payload: 'events' must be a list")
1029
+
1030
+ count = 0
1031
+ for evt in imported_events:
1032
+ eid = evt.get("event_id")
1033
+ if not eid:
1034
+ continue
1035
+ filename = f"imported_{eid}.json"
1036
+ dest = paths["events"] / filename
1037
+ if not dest.exists() or not merge:
1038
+ dest.write_text(json.dumps(evt, indent=2), encoding="utf-8")
1039
+ count += 1
1040
+
1041
+ distill_patterns(root_dir=root_dir)
1042
+
1043
+ return {
1044
+ "source_path": str(src),
1045
+ "imported_events_count": count,
1046
+ "active_patterns_count": len(json.loads(paths["patterns"].read_text(encoding="utf-8")))
1047
+ }
1048
+
1049
+
1050
+ def compact_events(older_than_days: int = 30, root_dir: Optional[Path] = None) -> int:
1051
+ """Archive events older than older_than_days into compacted_archive.json."""
1052
+ paths = ensure_memory_structure(root_dir)
1053
+ cutoff = _utc_now() - datetime.timedelta(days=older_than_days)
1054
+
1055
+ archived: List[Dict[str, Any]] = []
1056
+ archived_ids = set()
1057
+
1058
+ if paths["compacted"].exists():
1059
+ try:
1060
+ archived = json.loads(paths["compacted"].read_text(encoding="utf-8"))
1061
+ for item in archived:
1062
+ eid = item.get("event_id")
1063
+ if eid:
1064
+ archived_ids.add(eid)
1065
+ except Exception:
1066
+ archived = []
1067
+
1068
+ compacted_count = 0
1069
+ for item in paths["events"].glob("*.json"):
1070
+ if item.name == "compacted_archive.json":
1071
+ continue
1072
+ try:
1073
+ evt = json.loads(item.read_text(encoding="utf-8"))
1074
+ ts_str = evt.get("timestamp")
1075
+ if not ts_str:
1076
+ continue
1077
+ ts = _parse_iso_utc(ts_str)
1078
+ if ts < cutoff:
1079
+ eid = evt.get("event_id")
1080
+ if eid and eid not in archived_ids:
1081
+ archived.append(evt)
1082
+ archived_ids.add(eid)
1083
+ item.unlink(missing_ok=True)
1084
+ compacted_count += 1
1085
+ except Exception:
1086
+ continue
1087
+
1088
+ if compacted_count > 0 or not paths["compacted"].exists():
1089
+ paths["compacted"].write_text(json.dumps(archived, indent=2), encoding="utf-8")
1090
+
1091
+ return compacted_count
1092
+
1093
+
1094
+ def install_git_hook(root_dir: Optional[Path] = None) -> Dict[str, Any]:
1095
+ """Install a pre-commit git hook running TorusGuard diff_guard."""
1096
+ base = Path(root_dir or find_project_root()).resolve()
1097
+ git_dir = base / ".git"
1098
+ if not git_dir.exists():
1099
+ raise RuntimeError(f"Not a git repository: {base}")
1100
+
1101
+ hooks_dir = git_dir / "hooks"
1102
+ hooks_dir.mkdir(parents=True, exist_ok=True)
1103
+ pre_commit_hook = hooks_dir / "pre-commit"
1104
+
1105
+ hook_content = (
1106
+ "#!/bin/sh\n"
1107
+ "# TorusGuard Autonomous Security & Regression Guard\n"
1108
+ "python .torusguard/scripts/diff_guard.py --pre-commit\n"
1109
+ "EXIT_CODE=$?\n"
1110
+ "if [ $EXIT_CODE -ne 0 ]; then\n"
1111
+ " echo \"[BLOCKED] Commit rejected by TorusGuard diff security guardrails.\"\n"
1112
+ " exit $EXIT_CODE\n"
1113
+ "fi\n"
1114
+ "exit 0\n"
1115
+ )
1116
+
1117
+ pre_commit_hook.write_text(hook_content, encoding="utf-8")
1118
+ try:
1119
+ os.chmod(pre_commit_hook, 0o755)
1120
+ except Exception:
1121
+ pass
1122
+
1123
+ return {
1124
+ "status": "installed",
1125
+ "hook_path": str(pre_commit_hook)
1126
+ }
1127
+
1128
+
1129
+ def uninstall_git_hook(root_dir: Optional[Path] = None) -> Dict[str, Any]:
1130
+ """Uninstall the TorusGuard pre-commit git hook."""
1131
+ base = Path(root_dir or find_project_root()).resolve()
1132
+ pre_commit_hook = base / ".git" / "hooks" / "pre-commit"
1133
+ if pre_commit_hook.exists():
1134
+ content = pre_commit_hook.read_text(encoding="utf-8", errors="replace")
1135
+ if "TorusGuard" in content:
1136
+ pre_commit_hook.unlink()
1137
+ return {"status": "uninstalled", "hook_path": str(pre_commit_hook)}
1138
+ else:
1139
+ return {"status": "skipped", "reason": "Pre-commit hook does not belong to TorusGuard"}
1140
+ return {"status": "not_found"}
1141
+
1142
+
1143
+ def learn_from_git(commit_range: str = "HEAD~10..HEAD", root_dir: Optional[Path] = None) -> Dict[str, Any]:
1144
+ """Ingest developer security fixes from git commits into local memory."""
1145
+ import subprocess
1146
+ base = Path(root_dir or find_project_root()).resolve()
1147
+ learned_events = []
1148
+
1149
+ try:
1150
+ res = subprocess.run(
1151
+ ["git", "log", "-n", "10", "--pretty=format:%H|||%s", "--name-only"],
1152
+ cwd=str(base),
1153
+ capture_output=True,
1154
+ text=True,
1155
+ encoding="utf-8",
1156
+ errors="replace"
1157
+ )
1158
+ output = res.stdout
1159
+ except Exception as e:
1160
+ return {"commits_scanned": 0, "learned_events_count": 0, "events_recorded": 0, "error": str(e)}
1161
+
1162
+ current_hash = ""
1163
+ current_subject = ""
1164
+ for line in output.splitlines():
1165
+ line = line.strip()
1166
+ if not line:
1167
+ continue
1168
+ if "|||" in line:
1169
+ parts = line.split("|||", 1)
1170
+ current_hash = parts[0]
1171
+ current_subject = parts[1]
1172
+ else:
1173
+ file_mod = line
1174
+ sub_lower = current_subject.lower()
1175
+ if any(k in sub_lower for k in ["security", "fix", "vuln", "cve", "sanitize", "auth", "tenant"]):
1176
+ evt = record_event(
1177
+ "pattern_learned",
1178
+ {
1179
+ "source": "git_history",
1180
+ "commit_hash": current_hash[:8],
1181
+ "commit_message": current_subject,
1182
+ "file_path": file_mod
1183
+ },
1184
+ root_dir=base
1185
+ )
1186
+ learned_events.append(evt)
1187
+
1188
+ if learned_events:
1189
+ distill_patterns(root_dir=base)
1190
+
1191
+ return {
1192
+ "commits_scanned": 10,
1193
+ "learned_events_count": len(learned_events),
1194
+ "events_recorded": len(learned_events)
1195
+ }
1196
+
1197
+
1198
+ # ─── Command Line Interface ──────────────────────────────────────────────────
1199
+ def main():
1200
+ parser = argparse.ArgumentParser(description="TorusGuard Security Memory Engine")
1201
+ parser.add_argument("--action", required=True, choices=[
1202
+ "record", "distill", "context", "profile", "decay", "fp", "export", "import", "compact", "status",
1203
+ "recipe", "hook-install", "hook-uninstall", "learn"
1204
+ ], help="Action to perform")
1205
+ parser.add_argument("--root", help="Project root directory override")
1206
+ parser.add_argument("--type", help="Event type (audit_finding, fix_applied, etc.)")
1207
+ parser.add_argument("--rule-id", help="TorusGuard rule ID (e.g., TG-DB-004)")
1208
+ parser.add_argument("--file", help="File path")
1209
+ parser.add_argument("--line", type=int, help="Line number")
1210
+ parser.add_argument("--severity", choices=["critical", "high", "medium", "low", "info"])
1211
+ parser.add_argument("--score", type=int, help="Confidence score (0-100)")
1212
+ parser.add_argument("--strategy", help="Fix strategy description")
1213
+ parser.add_argument("--result", choices=["fixed", "regressed", "partial", "not_tested"])
1214
+ parser.add_argument("--reason", help="Suppression reason for false positive")
1215
+ parser.add_argument("--role", choices=["all", "auditor", "remediator", "reviewer"], default="all", help="Target agent role for context generation")
1216
+ parser.add_argument("--target", help="Export target path")
1217
+ parser.add_argument("--source", help="Import source path")
1218
+ parser.add_argument("--sanitized", action="store_true", help="Sanitize paths & secrets for export")
1219
+ parser.add_argument("--before", help="Before code snippet for golden recipe")
1220
+ parser.add_argument("--after", help="After code snippet for golden recipe")
1221
+ parser.add_argument("--diff", help="Diff snippet for golden recipe")
1222
+ parser.add_argument("--ttl", type=int, default=DEFAULT_TTL_DAYS, help="Decay TTL in days")
1223
+ parser.add_argument("--older-than", type=int, default=30, help="Compaction age in days")
1224
+ parser.add_argument("--json", action="store_true", help="Output raw JSON")
1225
+
1226
+ args = parser.parse_args()
1227
+ root = Path(args.root).resolve() if args.root else None
1228
+
1229
+ if args.action == "record":
1230
+ if not args.type:
1231
+ print("Error: --type is required for record action", file=sys.stderr)
1232
+ sys.exit(1)
1233
+ data = {
1234
+ "rule_id": args.rule_id,
1235
+ "file_path": args.file,
1236
+ "line_number": args.line,
1237
+ "severity": args.severity,
1238
+ "confidence_score": args.score,
1239
+ "fix_strategy": args.strategy,
1240
+ "verification_result": args.result,
1241
+ "suppression_reason": args.reason
1242
+ }
1243
+ evt = record_event(args.type, data, root_dir=root)
1244
+ distill_patterns(root_dir=root)
1245
+ print(json.dumps(evt, indent=2) if args.json else f"Recorded event: {evt['event_id']}")
1246
+
1247
+ elif args.action == "distill":
1248
+ pats = distill_patterns(root_dir=root)
1249
+ if args.json:
1250
+ print(json.dumps(pats, indent=2))
1251
+ else:
1252
+ print(f"Distilled {len(pats)} active patterns.")
1253
+
1254
+ elif args.action == "context":
1255
+ ctx = get_context(
1256
+ target_role=args.role,
1257
+ target_file=args.file,
1258
+ rule_id=args.rule_id,
1259
+ root_dir=root
1260
+ )
1261
+ print(json.dumps(ctx, indent=2))
1262
+
1263
+ elif args.action == "profile":
1264
+ prof = get_project_profile(root_dir=root)
1265
+ print(json.dumps(prof, indent=2))
1266
+
1267
+ elif args.action == "decay":
1268
+ decayed = decay_stale_entries(ttl_days=args.ttl, root_dir=root)
1269
+ print(f"Decayed {decayed} patterns older than {args.ttl} days.")
1270
+
1271
+ elif args.action == "fp":
1272
+ if not args.rule_id:
1273
+ print("Error: --rule-id is required for fp action", file=sys.stderr)
1274
+ sys.exit(1)
1275
+ evt = record_false_positive(args.rule_id, file_path=args.file, reason=args.reason or "False positive", root_dir=root)
1276
+ print(f"Suppressed false positive for {args.rule_id}")
1277
+
1278
+ elif args.action == "recipe":
1279
+ if not args.rule_id or not args.diff:
1280
+ print("Error: --rule-id and --diff are required for recipe action", file=sys.stderr)
1281
+ sys.exit(1)
1282
+ rec = record_golden_recipe(
1283
+ rule_id=args.rule_id,
1284
+ before_snippet=args.before or "",
1285
+ after_snippet=args.after or "",
1286
+ diff_snippet=args.diff,
1287
+ file_type=Path(args.file).suffix if args.file else ".py",
1288
+ description=args.strategy or "Verified AST remediation recipe",
1289
+ root_dir=root
1290
+ )
1291
+ print(json.dumps(rec, indent=2) if args.json else f"Recorded golden recipe: {rec['recipe_id']}")
1292
+
1293
+ elif args.action == "hook-install":
1294
+ res_h = install_git_hook(root_dir=root)
1295
+ print(json.dumps(res_h, indent=2) if args.json else f"Git pre-commit hook installed: {res_h['hook_path']}")
1296
+
1297
+ elif args.action == "hook-uninstall":
1298
+ res_u = uninstall_git_hook(root_dir=root)
1299
+ print(json.dumps(res_u, indent=2) if args.json else f"Git pre-commit hook {res_u['status']}")
1300
+
1301
+ elif args.action == "learn":
1302
+ res_l = learn_from_git(root_dir=root)
1303
+ print(json.dumps(res_l, indent=2) if args.json else f"Learned {res_l['learned_events_count']} events from git history.")
1304
+
1305
+ elif args.action == "export":
1306
+ if not args.target:
1307
+ print("Error: --target is required for export action", file=sys.stderr)
1308
+ sys.exit(1)
1309
+ res = export_memory(args.target, sanitized=args.sanitized, root_dir=root)
1310
+ print(json.dumps(res, indent=2) if args.json else f"Exported {res['exported_events_count']} events to {res['target_path']}")
1311
+
1312
+ elif args.action == "import":
1313
+ if not args.source:
1314
+ print("Error: --source is required for import action", file=sys.stderr)
1315
+ sys.exit(1)
1316
+ res = import_memory(args.source, root_dir=root)
1317
+ print(json.dumps(res, indent=2) if args.json else f"Imported {res['imported_events_count']} events from {res['source_path']}")
1318
+
1319
+ elif args.action == "compact":
1320
+ n = compact_events(older_than_days=args.older_than, root_dir=root)
1321
+ print(f"Compacted {n} events older than {args.older_than} days.")
1322
+
1323
+ elif args.action == "status":
1324
+ prof = get_project_profile(root_dir=root)
1325
+ ctx = get_context(root_dir=root)
1326
+ print(f"Memory Status (v{VERSION}):")
1327
+ print(f" Events Recorded: {prof.get('total_events', 0)}")
1328
+ print(f" Patterns Active: {prof.get('active_patterns_count', 0)}")
1329
+ print(f" Context Estimate: {ctx.get('token_estimate', 0)} / {ctx.get('max_token_budget', DEFAULT_TOKEN_BUDGET)} tokens")
1330
+ print(f" Fix Velocity Rate: {prof.get('fix_rate_percentage')}%")
1331
+
1332
+
1333
+ if __name__ == "__main__":
1334
+ main()