torusguard 0.9.5 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/.torusguard/.manifest.json +15 -10
  2. package/.torusguard/config/torusguard.json +1 -1
  3. package/.torusguard/schemas/memory-context.schema.json +95 -0
  4. package/.torusguard/schemas/memory-event.schema.json +70 -0
  5. package/.torusguard/schemas/memory-pattern.schema.json +65 -0
  6. package/.torusguard/scripts/__pycache__/diff_guard.cpython-311.pyc +0 -0
  7. package/.torusguard/scripts/__pycache__/finding_scorer.cpython-311.pyc +0 -0
  8. package/.torusguard/scripts/__pycache__/memory_engine.cpython-311.pyc +0 -0
  9. package/.torusguard/scripts/diff_guard.py +71 -11
  10. package/.torusguard/scripts/finding_scorer.py +171 -86
  11. package/.torusguard/scripts/manifest_builder.py +2 -0
  12. package/.torusguard/scripts/memory_engine.py +912 -0
  13. package/.torusguard/scripts/run_manager.py +171 -89
  14. package/.torusguard/skills/torusguard/SKILL.md +4 -3
  15. package/.torusguard/skills/torusguard/bootstrap.py +344 -56
  16. package/.torusguard/skills/torusguard-audit/SKILL.md +25 -25
  17. package/.torusguard/skills/torusguard-harden/SKILL.md +3 -3
  18. package/.torusguard/skills/torusguard-recheck/SKILL.md +3 -3
  19. package/.torusguard/workflows/memory.md +53 -0
  20. package/README.md +11 -2
  21. package/bin/torusguard.js +99 -5
  22. package/package.json +1 -1
  23. package/skills/torusguard/SKILL.md +4 -3
  24. package/skills/torusguard/__pycache__/bootstrap.cpython-311.pyc +0 -0
  25. package/skills/torusguard/bootstrap.py +66 -2
  26. package/skills/torusguard/payload/.manifest.json +15 -10
  27. package/skills/torusguard/payload/config/torusguard.json +1 -1
  28. package/skills/torusguard/payload/schemas/memory-context.schema.json +95 -0
  29. package/skills/torusguard/payload/schemas/memory-event.schema.json +70 -0
  30. package/skills/torusguard/payload/schemas/memory-pattern.schema.json +65 -0
  31. package/skills/torusguard/payload/scripts/diff_guard.py +71 -11
  32. package/skills/torusguard/payload/scripts/finding_scorer.py +171 -86
  33. package/skills/torusguard/payload/scripts/manifest_builder.py +2 -0
  34. package/skills/torusguard/payload/scripts/memory_engine.py +912 -0
  35. package/skills/torusguard/payload/scripts/run_manager.py +171 -89
  36. package/skills/torusguard/payload/skills/torusguard/SKILL.md +4 -3
  37. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +344 -56
  38. package/skills/torusguard/payload/skills/torusguard-audit/SKILL.md +25 -25
  39. package/skills/torusguard/payload/skills/torusguard-harden/SKILL.md +3 -3
  40. package/skills/torusguard/payload/skills/torusguard-recheck/SKILL.md +3 -3
  41. package/skills/torusguard/payload/workflows/memory.md +53 -0
  42. package/skills/torusguard-audit/SKILL.md +25 -25
  43. package/skills/torusguard-harden/SKILL.md +3 -3
  44. package/skills/torusguard-recheck/SKILL.md +3 -3
  45. package/skills/torusguard/payload/scripts/__pycache__/diff_guard.cpython-311.pyc +0 -0
  46. package/skills/torusguard/payload/scripts/__pycache__/monorepo_detector.cpython-311.pyc +0 -0
@@ -0,0 +1,912 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ TorusGuard Adaptive Security Memory Engine (v1.0.0)
4
+ Zero-dependency persistent intelligence layer for local-first security guardrails.
5
+ Manages raw event logging, pattern distillation, confidence amplification/decay,
6
+ and token-budgeted context window computation for AI agent prompts.
7
+ """
8
+
9
+ import os
10
+ import sys
11
+ import json
12
+ import datetime
13
+ import hashlib
14
+ import uuid
15
+ import argparse
16
+ from pathlib import Path
17
+ from typing import Dict, List, Any, Optional, Tuple
18
+
19
+ # Ensure UTF-8 stdout/stderr on Windows consoles
20
+ if sys.stdout and hasattr(sys.stdout, "reconfigure"):
21
+ try:
22
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
23
+ except Exception:
24
+ pass
25
+ if sys.stderr and hasattr(sys.stderr, "reconfigure"):
26
+ try:
27
+ sys.stderr.reconfigure(encoding="utf-8", errors="replace")
28
+ except Exception:
29
+ pass
30
+
31
+ VERSION = "1.0.0"
32
+ DEFAULT_TOKEN_BUDGET = 2000
33
+ DEFAULT_TTL_DAYS = 90
34
+ DEFAULT_DECAY_RATE = 0.15
35
+
36
+
37
+ def find_project_root(start_dir: Optional[str] = None) -> Path:
38
+ """Detect project root directory by searching for standard repo root markers."""
39
+ current = Path(start_dir or os.getcwd()).resolve()
40
+ markers = [".git", "package.json", "pyproject.toml", "manage.py", "Pipfile", "requirements.txt", ".torusguard"]
41
+
42
+ for m in markers:
43
+ if (current / m).exists():
44
+ return current
45
+
46
+ for parent in current.parents:
47
+ for m in markers:
48
+ if (parent / m).exists():
49
+ return parent
50
+
51
+ return current
52
+
53
+
54
+ def get_memory_paths(root_dir: Optional[Path] = None) -> Dict[str, Path]:
55
+ """Return all key paths within the .torusguard/memory subsystem."""
56
+ base = Path(root_dir or find_project_root()).resolve()
57
+ torusguard_dir = base / ".torusguard"
58
+ memory_dir = torusguard_dir / "memory"
59
+ events_dir = memory_dir / "events"
60
+
61
+ return {
62
+ "root": base,
63
+ "torusguard": torusguard_dir,
64
+ "memory": memory_dir,
65
+ "events": events_dir,
66
+ "patterns": memory_dir / "patterns.json",
67
+ "context": memory_dir / "context.json",
68
+ "profile": memory_dir / "profile.json",
69
+ "decay": memory_dir / "decay.json",
70
+ "compacted": events_dir / "compacted_archive.json",
71
+ "gitignore": memory_dir / ".gitignore"
72
+ }
73
+
74
+
75
+ def ensure_memory_structure(root_dir: Optional[Path] = None) -> Dict[str, Path]:
76
+ """Initialize memory directory layout with privacy isolation."""
77
+ paths = get_memory_paths(root_dir)
78
+ paths["memory"].mkdir(parents=True, exist_ok=True)
79
+ paths["events"].mkdir(parents=True, exist_ok=True)
80
+
81
+ # Privacy belt-and-suspenders: ignore everything inside .torusguard/memory/
82
+ if not paths["gitignore"].exists():
83
+ paths["gitignore"].write_text("*\n", encoding="utf-8")
84
+
85
+ gitkeep = paths["events"] / ".gitkeep"
86
+ if not gitkeep.exists():
87
+ gitkeep.write_text("", encoding="utf-8")
88
+
89
+ if not paths["decay"].exists():
90
+ decay_init = {
91
+ "default_ttl_days": DEFAULT_TTL_DAYS,
92
+ "decay_rate": DEFAULT_DECAY_RATE,
93
+ "last_decay_run": None
94
+ }
95
+ paths["decay"].write_text(json.dumps(decay_init, indent=2), encoding="utf-8")
96
+
97
+ if not paths["patterns"].exists():
98
+ paths["patterns"].write_text("[]", encoding="utf-8")
99
+
100
+ if not paths["profile"].exists():
101
+ paths["profile"].write_text(json.dumps({
102
+ "stack": [],
103
+ "total_events": 0,
104
+ "active_patterns_count": 0,
105
+ "fix_rate_percentage": None,
106
+ "top_vulnerabilities": [],
107
+ "last_updated": datetime.datetime.utcnow().isoformat() + "Z"
108
+ }, indent=2), encoding="utf-8")
109
+
110
+ if not paths["context"].exists():
111
+ initial_context = {
112
+ "version": VERSION,
113
+ "generated_at": datetime.datetime.utcnow().isoformat() + "Z",
114
+ "token_estimate": 0,
115
+ "max_token_budget": DEFAULT_TOKEN_BUDGET,
116
+ "project_profile": {
117
+ "stack": [],
118
+ "total_events": 0,
119
+ "active_patterns_count": 0,
120
+ "fix_rate_percentage": None,
121
+ "top_vulnerabilities": []
122
+ },
123
+ "cards": []
124
+ }
125
+ paths["context"].write_text(json.dumps(initial_context, indent=2), encoding="utf-8")
126
+
127
+ return paths
128
+
129
+
130
+ def estimate_tokens(obj: Any) -> int:
131
+ """Rough conservative token estimation for JSON payloads (1 token ≈ 4 chars)."""
132
+ serialized = json.dumps(obj, separators=(",", ":"))
133
+ return max(1, len(serialized) // 4)
134
+
135
+
136
+ def record_event(
137
+ event_type: str,
138
+ data: Dict[str, Any],
139
+ root_dir: Optional[Path] = None
140
+ ) -> Dict[str, Any]:
141
+ """
142
+ Append an individual raw event to .torusguard/memory/events/.
143
+ Event types:
144
+ - audit_finding
145
+ - fix_applied
146
+ - fix_verified
147
+ - false_positive
148
+ - pattern_learned
149
+ - stack_changed
150
+ """
151
+ valid_types = {
152
+ "audit_finding", "fix_applied", "fix_verified",
153
+ "false_positive", "pattern_learned", "stack_changed"
154
+ }
155
+ if event_type not in valid_types:
156
+ raise ValueError(f"Invalid event_type: {event_type}. Must be one of {valid_types}")
157
+
158
+ paths = ensure_memory_structure(root_dir)
159
+ now_utc = datetime.datetime.utcnow()
160
+ timestamp_iso = now_utc.isoformat() + "Z"
161
+ event_id = f"evt-{now_utc.strftime('%Y%m%d%H%M%S')}-{uuid.uuid4().hex[:8]}"
162
+
163
+ # Sanitize file_path to be relative to project root
164
+ raw_file = data.get("file_path")
165
+ clean_file = None
166
+ if raw_file:
167
+ try:
168
+ rel = Path(raw_file).resolve().relative_to(paths["root"].resolve())
169
+ clean_file = str(rel).replace("\\", "/")
170
+ except Exception:
171
+ clean_file = str(raw_file).replace("\\", "/")
172
+
173
+ event = {
174
+ "event_id": event_id,
175
+ "event_type": event_type,
176
+ "timestamp": timestamp_iso,
177
+ "version": VERSION,
178
+ "rule_id": data.get("rule_id"),
179
+ "file_path": clean_file,
180
+ "line_number": data.get("line_number"),
181
+ "severity": data.get("severity"),
182
+ "confidence_score": data.get("confidence_score"),
183
+ "code_hash": data.get("code_hash"),
184
+ "fix_strategy": data.get("fix_strategy"),
185
+ "verification_result": data.get("verification_result"),
186
+ "suppression_reason": data.get("suppression_reason"),
187
+ "metadata": data.get("metadata", {})
188
+ }
189
+
190
+ # Write event file with timestamp prefix for chronological directory ordering
191
+ filename = f"{now_utc.strftime('%Y%m%d_%H%M%S')}_{event_id}.json"
192
+ event_path = paths["events"] / filename
193
+ event_path.write_text(json.dumps(event, indent=2), encoding="utf-8")
194
+
195
+ return event
196
+
197
+
198
+ def load_all_events(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
199
+ """Load all raw events from memory/events/ including compacted archive."""
200
+ paths = ensure_memory_structure(root_dir)
201
+ events: List[Dict[str, Any]] = []
202
+
203
+ # 1. Load compacted archive if present
204
+ if paths["compacted"].exists():
205
+ try:
206
+ with open(paths["compacted"], "r", encoding="utf-8") as f:
207
+ archived = json.load(f)
208
+ if isinstance(archived, list):
209
+ events.extend(archived)
210
+ except Exception:
211
+ pass
212
+
213
+ # 2. Load individual event files
214
+ if paths["events"].is_dir():
215
+ for item in sorted(paths["events"].glob("*.json")):
216
+ if item.name == "compacted_archive.json":
217
+ continue
218
+ try:
219
+ with open(item, "r", encoding="utf-8") as f:
220
+ evt = json.load(f)
221
+ if isinstance(evt, dict) and "event_id" in evt:
222
+ events.append(evt)
223
+ except Exception:
224
+ continue
225
+
226
+ # Deduplicate by event_id
227
+ seen_ids = set()
228
+ deduped = []
229
+ for evt in events:
230
+ eid = evt.get("event_id")
231
+ if eid and eid not in seen_ids:
232
+ seen_ids.add(eid)
233
+ deduped.append(evt)
234
+
235
+ # Sort chronologically
236
+ deduped.sort(key=lambda x: x.get("timestamp", ""))
237
+ return deduped
238
+
239
+
240
+ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
241
+ """
242
+ Distill raw events into actionable, deduplicated security patterns.
243
+ Pattern types:
244
+ - recurring_fix: Repeated successful fixes for a rule
245
+ - common_vulnerability: Vulnerabilities appearing frequently
246
+ - false_positive_class: Suppressed rules/files
247
+ - regression_watch: Rules/files that regressed or re-appeared
248
+ - security_idiom: Project-specific established remediation practices
249
+ """
250
+ paths = ensure_memory_structure(root_dir)
251
+ events = load_all_events(root_dir)
252
+ patterns: List[Dict[str, Any]] = []
253
+
254
+ if not events:
255
+ paths["patterns"].write_text("[]", encoding="utf-8")
256
+ compute_context_window(root_dir=root_dir)
257
+ return patterns
258
+
259
+ # Grouping indices
260
+ rule_findings: Dict[str, List[Dict[str, Any]]] = {}
261
+ rule_fixes: Dict[Tuple[str, str], List[Dict[str, Any]]] = {}
262
+ false_positives: Dict[str, List[Dict[str, Any]]] = {}
263
+ regressions: Dict[str, List[Dict[str, Any]]] = {}
264
+
265
+ for evt in events:
266
+ etype = evt.get("event_type")
267
+ rule_id = evt.get("rule_id")
268
+ if not rule_id:
269
+ continue
270
+
271
+ if etype == "audit_finding":
272
+ rule_findings.setdefault(rule_id, []).append(evt)
273
+ elif etype == "fix_applied":
274
+ strat = evt.get("fix_strategy") or "standard_remediation"
275
+ rule_fixes.setdefault((rule_id, strat), []).append(evt)
276
+ elif etype == "fix_verified":
277
+ strat = evt.get("fix_strategy") or "standard_remediation"
278
+ vres = evt.get("verification_result")
279
+ if vres == "regressed":
280
+ regressions.setdefault(rule_id, []).append(evt)
281
+ else:
282
+ rule_fixes.setdefault((rule_id, strat), []).append(evt)
283
+ elif etype == "false_positive":
284
+ false_positives.setdefault(rule_id, []).append(evt)
285
+
286
+ pat_idx = 1
287
+
288
+ # 1. Distill Recurring Fixes & Security Idioms
289
+ for (rule_id, fix_strat), fix_evts in rule_fixes.items():
290
+ occurrences = len(fix_evts)
291
+ affected_files = sorted(list({e.get("file_path") for e in fix_evts if e.get("file_path")}))
292
+ verified_count = sum(1 for e in fix_evts if e.get("verification_result") == "fixed")
293
+
294
+ # Confidence amplification logic
295
+ base_confidence = min(95, 50 + (occurrences * 10) + (verified_count * 10))
296
+ # Multi-file bonus
297
+ if len(affected_files) > 1:
298
+ base_confidence = min(98, base_confidence + 5)
299
+
300
+ pattern_type = "security_idiom" if occurrences >= 3 and verified_count >= 2 else "recurring_fix"
301
+ timestamps = [e.get("timestamp") for e in fix_evts if e.get("timestamp")]
302
+ first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
303
+ last_seen = max(timestamps) if timestamps else first_seen
304
+
305
+ patterns.append({
306
+ "pattern_id": f"PAT-{pat_idx:03d}",
307
+ "rule_id": rule_id,
308
+ "pattern_type": pattern_type,
309
+ "description": f"Verified remediation strategy for {rule_id}: {fix_strat}",
310
+ "fix_strategy": fix_strat,
311
+ "confidence": base_confidence,
312
+ "occurrences": occurrences,
313
+ "affected_files": affected_files,
314
+ "first_seen": first_seen,
315
+ "last_seen": last_seen,
316
+ "decay_checkpoint": last_seen,
317
+ "source_events": [e.get("event_id") for e in fix_evts if e.get("event_id")]
318
+ })
319
+ pat_idx += 1
320
+
321
+ # 2. Distill Common Vulnerabilities
322
+ for rule_id, find_evts in rule_findings.items():
323
+ occurrences = len(find_evts)
324
+ if occurrences >= 2:
325
+ affected_files = sorted(list({e.get("file_path") for e in find_evts if e.get("file_path")}))
326
+ timestamps = [e.get("timestamp") for e in find_evts if e.get("timestamp")]
327
+ first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
328
+ last_seen = max(timestamps) if timestamps else first_seen
329
+
330
+ confidence = min(90, 55 + (occurrences * 5))
331
+ patterns.append({
332
+ "pattern_id": f"PAT-{pat_idx:03d}",
333
+ "rule_id": rule_id,
334
+ "pattern_type": "common_vulnerability",
335
+ "description": f"Frequent vulnerability pattern detected across project for {rule_id}",
336
+ "fix_strategy": None,
337
+ "confidence": confidence,
338
+ "occurrences": occurrences,
339
+ "affected_files": affected_files,
340
+ "first_seen": first_seen,
341
+ "last_seen": last_seen,
342
+ "decay_checkpoint": last_seen,
343
+ "source_events": [e.get("event_id") for e in find_evts if e.get("event_id")]
344
+ })
345
+ pat_idx += 1
346
+
347
+ # 3. Distill False Positive Classes
348
+ for rule_id, fp_evts in false_positives.items():
349
+ occurrences = len(fp_evts)
350
+ affected_files = sorted(list({e.get("file_path") for e in fp_evts if e.get("file_path")}))
351
+ reasons = [e.get("suppression_reason") for e in fp_evts if e.get("suppression_reason")]
352
+ summary_reason = reasons[-1] if reasons else "Suppressed by team policy"
353
+ timestamps = [e.get("timestamp") for e in fp_evts if e.get("timestamp")]
354
+ first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
355
+ last_seen = max(timestamps) if timestamps else first_seen
356
+
357
+ patterns.append({
358
+ "pattern_id": f"PAT-{pat_idx:03d}",
359
+ "rule_id": rule_id,
360
+ "pattern_type": "false_positive_class",
361
+ "description": f"Rule {rule_id} marked as false positive ({summary_reason})",
362
+ "fix_strategy": None,
363
+ "confidence": 90,
364
+ "occurrences": occurrences,
365
+ "affected_files": affected_files,
366
+ "first_seen": first_seen,
367
+ "last_seen": last_seen,
368
+ "decay_checkpoint": last_seen,
369
+ "source_events": [e.get("event_id") for e in fp_evts if e.get("event_id")]
370
+ })
371
+ pat_idx += 1
372
+
373
+ # 4. Distill Regression Watch Entries
374
+ for rule_id, reg_evts in regressions.items():
375
+ occurrences = len(reg_evts)
376
+ affected_files = sorted(list({e.get("file_path") for e in reg_evts if e.get("file_path")}))
377
+ timestamps = [e.get("timestamp") for e in reg_evts if e.get("timestamp")]
378
+ first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
379
+ last_seen = max(timestamps) if timestamps else first_seen
380
+
381
+ patterns.append({
382
+ "pattern_id": f"PAT-{pat_idx:03d}",
383
+ "rule_id": rule_id,
384
+ "pattern_type": "regression_watch",
385
+ "description": f"High risk regression watch: {rule_id} re-occurred or failed verification",
386
+ "fix_strategy": None,
387
+ "confidence": 88,
388
+ "occurrences": occurrences,
389
+ "affected_files": affected_files,
390
+ "first_seen": first_seen,
391
+ "last_seen": last_seen,
392
+ "decay_checkpoint": last_seen,
393
+ "source_events": [e.get("event_id") for e in reg_evts if e.get("event_id")]
394
+ })
395
+ pat_idx += 1
396
+
397
+ # Save patterns
398
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
399
+
400
+ # Update profile and pre-computed context window
401
+ get_project_profile(root_dir=root_dir)
402
+ compute_context_window(root_dir=root_dir)
403
+
404
+ return patterns
405
+
406
+
407
+ def decay_stale_entries(
408
+ ttl_days: int = DEFAULT_TTL_DAYS,
409
+ decay_rate: float = DEFAULT_DECAY_RATE,
410
+ root_dir: Optional[Path] = None
411
+ ) -> int:
412
+ """
413
+ Apply TTL decay to patterns not reconfirmed within ttl_days.
414
+ Reduces confidence to prevent stale architectural advice.
415
+ """
416
+ paths = ensure_memory_structure(root_dir)
417
+ now = datetime.datetime.utcnow()
418
+
419
+ # Read config from decay.json if available
420
+ try:
421
+ if paths["decay"].exists():
422
+ cfg = json.loads(paths["decay"].read_text(encoding="utf-8"))
423
+ ttl_days = cfg.get("default_ttl_days", ttl_days)
424
+ decay_rate = cfg.get("decay_rate", decay_rate)
425
+ except Exception:
426
+ pass
427
+
428
+ if not paths["patterns"].exists():
429
+ return 0
430
+
431
+ try:
432
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
433
+ except Exception:
434
+ return 0
435
+
436
+ decayed_count = 0
437
+ for pat in patterns:
438
+ chk_str = pat.get("decay_checkpoint") or pat.get("last_seen")
439
+ if not chk_str:
440
+ continue
441
+ try:
442
+ # Parse ISO date string (strip Z if present)
443
+ clean_ts = chk_str.rstrip("Z")
444
+ last_dt = datetime.datetime.fromisoformat(clean_ts)
445
+ days_elapsed = (now - last_dt).days
446
+
447
+ if days_elapsed >= ttl_days:
448
+ old_conf = pat.get("confidence", 50)
449
+ reduction = max(5, int(old_conf * decay_rate))
450
+ new_conf = max(10, old_conf - reduction)
451
+ if new_conf != old_conf:
452
+ pat["confidence"] = new_conf
453
+ pat["decay_checkpoint"] = now.isoformat() + "Z"
454
+ decayed_count += 1
455
+ except Exception:
456
+ continue
457
+
458
+ if decayed_count > 0:
459
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
460
+ compute_context_window(root_dir=root_dir)
461
+
462
+ # Record decay run in decay.json
463
+ try:
464
+ decay_cfg = {
465
+ "default_ttl_days": ttl_days,
466
+ "decay_rate": decay_rate,
467
+ "last_decay_run": now.isoformat() + "Z",
468
+ "last_decayed_count": decayed_count
469
+ }
470
+ paths["decay"].write_text(json.dumps(decay_cfg, indent=2), encoding="utf-8")
471
+ except Exception:
472
+ pass
473
+
474
+ return decayed_count
475
+
476
+
477
+ def get_project_profile(root_dir: Optional[Path] = None) -> Dict[str, Any]:
478
+ """Compute and update the project's security DNA profile."""
479
+ paths = ensure_memory_structure(root_dir)
480
+ events = load_all_events(root_dir)
481
+
482
+ # Detect stack from torusguard.json if available
483
+ detected_stack: List[str] = []
484
+ config_file = paths["torusguard"] / "config" / "torusguard.json"
485
+ if config_file.exists():
486
+ try:
487
+ cfg = json.loads(config_file.read_text(encoding="utf-8"))
488
+ stk = cfg.get("detected_stack", {})
489
+ for k in ("language", "framework", "data_layer"):
490
+ v = stk.get(k)
491
+ if v and v not in ("None", "Unknown"):
492
+ detected_stack.append(v)
493
+ except Exception:
494
+ pass
495
+
496
+ # Count statistics
497
+ findings_count = 0
498
+ fixes_count = 0
499
+ verified_fixed_count = 0
500
+ vuln_counter: Dict[str, int] = {}
501
+
502
+ for evt in events:
503
+ etype = evt.get("event_type")
504
+ rid = evt.get("rule_id")
505
+ if etype == "audit_finding":
506
+ findings_count += 1
507
+ if rid:
508
+ vuln_counter[rid] = vuln_counter.get(rid, 0) + 1
509
+ elif etype == "fix_applied":
510
+ fixes_count += 1
511
+ elif etype == "fix_verified":
512
+ if evt.get("verification_result") == "fixed":
513
+ verified_fixed_count += 1
514
+
515
+ fix_rate = None
516
+ if findings_count > 0:
517
+ fix_rate = round((verified_fixed_count / findings_count) * 100, 1)
518
+
519
+ top_vulns = [item[0] for item in sorted(vuln_counter.items(), key=lambda x: x[1], reverse=True)[:5]]
520
+
521
+ # Count active patterns
522
+ active_patterns_count = 0
523
+ if paths["patterns"].exists():
524
+ try:
525
+ pats = json.loads(paths["patterns"].read_text(encoding="utf-8"))
526
+ active_patterns_count = len(pats)
527
+ except Exception:
528
+ pass
529
+
530
+ profile = {
531
+ "stack": detected_stack,
532
+ "total_events": len(events),
533
+ "active_patterns_count": active_patterns_count,
534
+ "findings_count": findings_count,
535
+ "fixes_applied_count": fixes_count,
536
+ "fixes_verified_count": verified_fixed_count,
537
+ "fix_rate_percentage": fix_rate,
538
+ "top_vulnerabilities": top_vulns,
539
+ "last_updated": datetime.datetime.utcnow().isoformat() + "Z"
540
+ }
541
+
542
+ paths["profile"].write_text(json.dumps(profile, indent=2), encoding="utf-8")
543
+ return profile
544
+
545
+
546
+ def compute_context_window(
547
+ max_tokens: int = DEFAULT_TOKEN_BUDGET,
548
+ root_dir: Optional[Path] = None
549
+ ) -> Dict[str, Any]:
550
+ """
551
+ Build the pre-computed, token-budgeted context window as structured JSON cards.
552
+ Guarantees strict token enforcement <= max_tokens.
553
+ """
554
+ paths = ensure_memory_structure(root_dir)
555
+ profile = get_project_profile(root_dir=root_dir)
556
+
557
+ patterns: List[Dict[str, Any]] = []
558
+ if paths["patterns"].exists():
559
+ try:
560
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
561
+ except Exception:
562
+ patterns = []
563
+
564
+ # Sort patterns by priority: regressions (95) -> idioms (90) -> recurring fixes (85) -> FP (80) -> common vulns (70)
565
+ type_priority = {
566
+ "regression_watch": 95,
567
+ "security_idiom": 90,
568
+ "recurring_fix": 85,
569
+ "false_positive_class": 80,
570
+ "common_vulnerability": 70
571
+ }
572
+
573
+ # Generate Candidate Cards
574
+ cards: List[Dict[str, Any]] = []
575
+ card_idx = 1
576
+
577
+ # 1. Project Profile Card (Priority: 100)
578
+ cards.append({
579
+ "card_id": f"CARD-{card_idx:03d}",
580
+ "card_type": "profile",
581
+ "priority": 100,
582
+ "title": "Project Security Posture & DNA",
583
+ "summary": f"Stack: {', '.join(profile.get('stack') or ['Generic'])}; Total Events: {profile.get('total_events', 0)}; Fix Rate: {profile.get('fix_rate_percentage')}%",
584
+ "card_data": {
585
+ "stack": profile.get("stack", []),
586
+ "total_events": profile.get("total_events", 0),
587
+ "fix_rate_percentage": profile.get("fix_rate_percentage"),
588
+ "top_vulnerabilities": profile.get("top_vulnerabilities", [])
589
+ }
590
+ })
591
+ card_idx += 1
592
+
593
+ # Sort patterns by type priority then confidence then occurrences
594
+ sorted_patterns = sorted(
595
+ patterns,
596
+ key=lambda p: (
597
+ type_priority.get(p.get("pattern_type", ""), 50),
598
+ p.get("confidence", 0),
599
+ p.get("occurrences", 0)
600
+ ),
601
+ reverse=True
602
+ )
603
+
604
+ for pat in sorted_patterns:
605
+ ptype = pat.get("pattern_type", "pattern")
606
+ card_type = "pattern"
607
+ if ptype == "regression_watch":
608
+ card_type = "regression_watch"
609
+ elif ptype == "false_positive_class":
610
+ card_type = "false_positive"
611
+ elif ptype == "security_idiom":
612
+ card_type = "fix_idiom"
613
+
614
+ cards.append({
615
+ "card_id": f"CARD-{card_idx:03d}",
616
+ "card_type": card_type,
617
+ "priority": type_priority.get(ptype, 60),
618
+ "title": f"[{pat.get('rule_id')}] {pat.get('pattern_type')}: {pat.get('description', '')[:60]}",
619
+ "summary": pat.get("description", ""),
620
+ "card_data": {
621
+ "rule_id": pat.get("rule_id"),
622
+ "pattern_type": pat.get("pattern_type"),
623
+ "fix_strategy": pat.get("fix_strategy"),
624
+ "confidence": pat.get("confidence"),
625
+ "occurrences": pat.get("occurrences"),
626
+ "affected_files": pat.get("affected_files", [])[:5]
627
+ }
628
+ })
629
+ card_idx += 1
630
+
631
+ # Build Context Window Payload
632
+ context = {
633
+ "version": VERSION,
634
+ "generated_at": datetime.datetime.utcnow().isoformat() + "Z",
635
+ "token_estimate": 0,
636
+ "max_token_budget": max_tokens,
637
+ "project_profile": {
638
+ "stack": profile.get("stack", []),
639
+ "total_events": profile.get("total_events", 0),
640
+ "active_patterns_count": profile.get("active_patterns_count", 0),
641
+ "fix_rate_percentage": profile.get("fix_rate_percentage"),
642
+ "top_vulnerabilities": profile.get("top_vulnerabilities", [])
643
+ },
644
+ "cards": cards
645
+ }
646
+
647
+ # Token budget enforcement: Drop lowest-priority cards until within limit
648
+ # Always keep at least the profile card (cards[0])
649
+ while len(context["cards"]) > 1 and estimate_tokens(context) > max_tokens:
650
+ context["cards"].pop()
651
+
652
+ context["token_estimate"] = estimate_tokens(context)
653
+
654
+ # Persist context.json
655
+ paths["context"].write_text(json.dumps(context, indent=2), encoding="utf-8")
656
+ return context
657
+
658
+
659
+ def get_context(root_dir: Optional[Path] = None) -> Dict[str, Any]:
660
+ """Retrieve the current pre-computed context window."""
661
+ paths = ensure_memory_structure(root_dir)
662
+ if paths["context"].exists():
663
+ try:
664
+ return json.loads(paths["context"].read_text(encoding="utf-8"))
665
+ except Exception:
666
+ pass
667
+
668
+ return compute_context_window(root_dir=root_dir)
669
+
670
+
671
+ def record_false_positive(
672
+ rule_id: str,
673
+ file_path: Optional[str] = None,
674
+ reason: str = "False positive verified by user",
675
+ root_dir: Optional[Path] = None
676
+ ) -> Dict[str, Any]:
677
+ """Record a false positive suppression and update patterns."""
678
+ evt = record_event(
679
+ "false_positive",
680
+ {
681
+ "rule_id": rule_id,
682
+ "file_path": file_path,
683
+ "suppression_reason": reason
684
+ },
685
+ root_dir=root_dir
686
+ )
687
+ distill_patterns(root_dir=root_dir)
688
+ return evt
689
+
690
+
691
+ def export_memory(target_path: str, root_dir: Optional[Path] = None) -> Dict[str, Any]:
692
+ """
693
+ Export memory to an external file for team sharing.
694
+ Per user choice: retain code hashes for high-fidelity matching, but sanitize absolute paths.
695
+ """
696
+ paths = ensure_memory_structure(root_dir)
697
+ events = load_all_events(root_dir)
698
+ profile = get_project_profile(root_dir=root_dir)
699
+
700
+ patterns = []
701
+ if paths["patterns"].exists():
702
+ try:
703
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
704
+ except Exception:
705
+ pass
706
+
707
+ decay_cfg = {}
708
+ if paths["decay"].exists():
709
+ try:
710
+ decay_cfg = json.loads(paths["decay"].read_text(encoding="utf-8"))
711
+ except Exception:
712
+ pass
713
+
714
+ export_payload = {
715
+ "format": "torusguard-memory-bundle",
716
+ "schema_version": "1.0.0",
717
+ "exported_at": datetime.datetime.utcnow().isoformat() + "Z",
718
+ "project_profile": profile,
719
+ "decay_config": decay_cfg,
720
+ "patterns": patterns,
721
+ "events": events
722
+ }
723
+
724
+ out_file = Path(target_path).resolve()
725
+ out_file.parent.mkdir(parents=True, exist_ok=True)
726
+ out_file.write_text(json.dumps(export_payload, indent=2), encoding="utf-8")
727
+
728
+ return {
729
+ "target_path": str(out_file),
730
+ "exported_events_count": len(events),
731
+ "exported_patterns_count": len(patterns)
732
+ }
733
+
734
+
735
+ def import_memory(source_path: str, merge: bool = True, root_dir: Optional[Path] = None) -> Dict[str, Any]:
736
+ """Import and optionally merge external memory events and patterns."""
737
+ paths = ensure_memory_structure(root_dir)
738
+ src = Path(source_path).resolve()
739
+ if not src.is_file():
740
+ raise FileNotFoundError(f"Export file not found: {source_path}")
741
+
742
+ with open(src, "r", encoding="utf-8") as f:
743
+ payload = json.load(f)
744
+
745
+ imported_events = payload.get("events", [])
746
+ if not isinstance(imported_events, list):
747
+ raise ValueError("Malformed import payload: 'events' must be a list")
748
+
749
+ count = 0
750
+ for evt in imported_events:
751
+ eid = evt.get("event_id")
752
+ if not eid:
753
+ continue
754
+ filename = f"imported_{eid}.json"
755
+ dest = paths["events"] / filename
756
+ if not dest.exists() or not merge:
757
+ dest.write_text(json.dumps(evt, indent=2), encoding="utf-8")
758
+ count += 1
759
+
760
+ distill_patterns(root_dir=root_dir)
761
+
762
+ return {
763
+ "source_path": str(src),
764
+ "imported_events_count": count,
765
+ "active_patterns_count": len(json.loads(paths["patterns"].read_text(encoding="utf-8")))
766
+ }
767
+
768
+
769
+ def compact_events(older_than_days: int = 30, root_dir: Optional[Path] = None) -> int:
770
+ """
771
+ Compact loose event JSON files older than older_than_days into compacted_archive.json.
772
+ Prevents filesystem inode saturation while preserving full history.
773
+ """
774
+ paths = ensure_memory_structure(root_dir)
775
+ cutoff = datetime.datetime.utcnow() - datetime.timedelta(days=older_than_days)
776
+
777
+ archived: List[Dict[str, Any]] = []
778
+ if paths["compacted"].exists():
779
+ try:
780
+ with open(paths["compacted"], "r", encoding="utf-8") as f:
781
+ data = json.load(f)
782
+ if isinstance(data, list):
783
+ archived = data
784
+ except Exception:
785
+ archived = []
786
+
787
+ archived_ids = {e.get("event_id") for e in archived if e.get("event_id")}
788
+ compacted_count = 0
789
+
790
+ for item in sorted(paths["events"].glob("*.json")):
791
+ if item.name == "compacted_archive.json":
792
+ continue
793
+ try:
794
+ with open(item, "r", encoding="utf-8") as f:
795
+ evt = json.load(f)
796
+ ts_str = evt.get("timestamp", "").rstrip("Z")
797
+ evt_dt = datetime.datetime.fromisoformat(ts_str) if ts_str else None
798
+ if evt_dt and evt_dt < cutoff:
799
+ eid = evt.get("event_id")
800
+ if eid and eid not in archived_ids:
801
+ archived.append(evt)
802
+ archived_ids.add(eid)
803
+ item.unlink(missing_ok=True)
804
+ compacted_count += 1
805
+ except Exception:
806
+ continue
807
+
808
+ if compacted_count > 0 or not paths["compacted"].exists():
809
+ paths["compacted"].write_text(json.dumps(archived, indent=2), encoding="utf-8")
810
+
811
+ return compacted_count
812
+
813
+
814
+ # ─── Command Line Interface ──────────────────────────────────────────────────
815
+ def main():
816
+ parser = argparse.ArgumentParser(description="TorusGuard Security Memory Engine")
817
+ parser.add_argument("--action", required=True, choices=[
818
+ "record", "distill", "context", "profile", "decay", "fp", "export", "import", "compact", "status"
819
+ ], help="Action to perform")
820
+ parser.add_argument("--root", help="Project root directory override")
821
+ parser.add_argument("--type", help="Event type (audit_finding, fix_applied, etc.)")
822
+ parser.add_argument("--rule-id", help="TorusGuard rule ID (e.g., TG-DB-004)")
823
+ parser.add_argument("--file", help="File path")
824
+ parser.add_argument("--line", type=int, help="Line number")
825
+ parser.add_argument("--severity", choices=["critical", "high", "medium", "low", "info"])
826
+ parser.add_argument("--score", type=int, help="Confidence score (0-100)")
827
+ parser.add_argument("--strategy", help="Fix strategy description")
828
+ parser.add_argument("--result", choices=["fixed", "regressed", "partial", "not_tested"])
829
+ parser.add_argument("--reason", help="Suppression reason for false positive")
830
+ parser.add_argument("--target", help="Export target path")
831
+ parser.add_argument("--source", help="Import source path")
832
+ parser.add_argument("--ttl", type=int, default=DEFAULT_TTL_DAYS, help="Decay TTL in days")
833
+ parser.add_argument("--older-than", type=int, default=30, help="Compaction age in days")
834
+ parser.add_argument("--json", action="store_true", help="Output raw JSON")
835
+
836
+ args = parser.parse_args()
837
+ root = Path(args.root).resolve() if args.root else None
838
+
839
+ if args.action == "record":
840
+ if not args.type:
841
+ print("Error: --type is required for record action", file=sys.stderr)
842
+ sys.exit(1)
843
+ data = {
844
+ "rule_id": args.rule_id,
845
+ "file_path": args.file,
846
+ "line_number": args.line,
847
+ "severity": args.severity,
848
+ "confidence_score": args.score,
849
+ "fix_strategy": args.strategy,
850
+ "verification_result": args.result,
851
+ "suppression_reason": args.reason
852
+ }
853
+ evt = record_event(args.type, data, root_dir=root)
854
+ distill_patterns(root_dir=root)
855
+ print(json.dumps(evt, indent=2) if args.json else f"Recorded event: {evt['event_id']}")
856
+
857
+ elif args.action == "distill":
858
+ pats = distill_patterns(root_dir=root)
859
+ if args.json:
860
+ print(json.dumps(pats, indent=2))
861
+ else:
862
+ print(f"Distilled {len(pats)} active patterns.")
863
+
864
+ elif args.action == "context":
865
+ ctx = get_context(root_dir=root)
866
+ print(json.dumps(ctx, indent=2))
867
+
868
+ elif args.action == "profile":
869
+ prof = get_project_profile(root_dir=root)
870
+ print(json.dumps(prof, indent=2))
871
+
872
+ elif args.action == "decay":
873
+ decayed = decay_stale_entries(ttl_days=args.ttl, root_dir=root)
874
+ print(f"Decayed {decayed} patterns older than {args.ttl} days.")
875
+
876
+ elif args.action == "fp":
877
+ if not args.rule_id:
878
+ print("Error: --rule-id is required for fp action", file=sys.stderr)
879
+ sys.exit(1)
880
+ evt = record_false_positive(args.rule_id, file_path=args.file, reason=args.reason or "False positive", root_dir=root)
881
+ print(f"Suppressed false positive for {args.rule_id}")
882
+
883
+ elif args.action == "export":
884
+ if not args.target:
885
+ print("Error: --target is required for export action", file=sys.stderr)
886
+ sys.exit(1)
887
+ res = export_memory(args.target, root_dir=root)
888
+ print(json.dumps(res, indent=2) if args.json else f"Exported {res['exported_events_count']} events to {res['target_path']}")
889
+
890
+ elif args.action == "import":
891
+ if not args.source:
892
+ print("Error: --source is required for import action", file=sys.stderr)
893
+ sys.exit(1)
894
+ res = import_memory(args.source, root_dir=root)
895
+ print(json.dumps(res, indent=2) if args.json else f"Imported {res['imported_events_count']} events from {res['source_path']}")
896
+
897
+ elif args.action == "compact":
898
+ n = compact_events(older_than_days=args.older_than, root_dir=root)
899
+ print(f"Compacted {n} events older than {args.older_than} days.")
900
+
901
+ elif args.action == "status":
902
+ prof = get_project_profile(root_dir=root)
903
+ ctx = get_context(root_dir=root)
904
+ print(f"Memory Status (v{VERSION}):")
905
+ print(f" Events Recorded: {prof.get('total_events', 0)}")
906
+ print(f" Patterns Active: {prof.get('active_patterns_count', 0)}")
907
+ print(f" Context Estimate: {ctx.get('token_estimate', 0)} / {ctx.get('max_token_budget', DEFAULT_TOKEN_BUDGET)} tokens")
908
+ print(f" Fix Velocity Rate: {prof.get('fix_rate_percentage')}%")
909
+
910
+
911
+ if __name__ == "__main__":
912
+ main()