torusguard 1.0.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/.torusguard/.manifest.json +89 -75
  2. package/.torusguard/references/csharp-security.md +41 -0
  3. package/.torusguard/references/go-security.md +41 -0
  4. package/.torusguard/references/java-security.md +40 -0
  5. package/.torusguard/references/polyglot-security-matrix.md +25 -0
  6. package/.torusguard/references/rust-security.md +40 -0
  7. package/.torusguard/rules/custom/.gitkeep +1 -0
  8. package/.torusguard/rules/custom/README.md +30 -0
  9. package/.torusguard/runs/report-latest.html +328 -0
  10. package/.torusguard/schemas/golden-recipe.schema.json +75 -0
  11. package/.torusguard/scripts/diff_guard.py +131 -4
  12. package/.torusguard/scripts/finding_scorer.py +33 -5
  13. package/.torusguard/scripts/html_reporter.py +628 -0
  14. package/.torusguard/scripts/manifest_builder.py +5 -5
  15. package/.torusguard/scripts/memory_engine.py +521 -99
  16. package/.torusguard/scripts/monorepo_detector.py +123 -10
  17. package/.torusguard/scripts/rules_sync.py +321 -0
  18. package/.torusguard/scripts/stack_detect.py +458 -26
  19. package/.torusguard/skills/torusguard/references/csharp-security.md +41 -0
  20. package/.torusguard/skills/torusguard/references/go-security.md +41 -0
  21. package/.torusguard/skills/torusguard/references/java-security.md +40 -0
  22. package/.torusguard/skills/torusguard/references/polyglot-security-matrix.md +25 -0
  23. package/.torusguard/skills/torusguard/references/rust-security.md +40 -0
  24. package/README.md +307 -410
  25. package/bin/torusguard.js +108 -4
  26. package/package.json +1 -1
  27. package/skills/torusguard/bootstrap.py +2 -2
  28. package/skills/torusguard/payload/.manifest.json +89 -75
  29. package/skills/torusguard/payload/references/csharp-security.md +41 -0
  30. package/skills/torusguard/payload/references/go-security.md +41 -0
  31. package/skills/torusguard/payload/references/java-security.md +40 -0
  32. package/skills/torusguard/payload/references/polyglot-security-matrix.md +25 -0
  33. package/skills/torusguard/payload/references/rust-security.md +40 -0
  34. package/skills/torusguard/payload/rules/custom/.gitkeep +1 -0
  35. package/skills/torusguard/payload/rules/custom/README.md +30 -0
  36. package/skills/torusguard/payload/schemas/golden-recipe.schema.json +75 -0
  37. package/skills/torusguard/payload/scripts/diff_guard.py +131 -4
  38. package/skills/torusguard/payload/scripts/finding_scorer.py +33 -5
  39. package/skills/torusguard/payload/scripts/html_reporter.py +628 -0
  40. package/skills/torusguard/payload/scripts/manifest_builder.py +5 -5
  41. package/skills/torusguard/payload/scripts/memory_engine.py +521 -99
  42. package/skills/torusguard/payload/scripts/monorepo_detector.py +123 -10
  43. package/skills/torusguard/payload/scripts/rules_sync.py +321 -0
  44. package/skills/torusguard/payload/scripts/stack_detect.py +458 -26
  45. package/skills/torusguard/payload/skills/torusguard/bootstrap.py +2 -2
  46. package/skills/torusguard/payload/skills/torusguard/references/csharp-security.md +41 -0
  47. package/skills/torusguard/payload/skills/torusguard/references/go-security.md +41 -0
  48. package/skills/torusguard/payload/skills/torusguard/references/java-security.md +40 -0
  49. package/skills/torusguard/payload/skills/torusguard/references/polyglot-security-matrix.md +25 -0
  50. package/skills/torusguard/payload/skills/torusguard/references/rust-security.md +40 -0
  51. package/.torusguard/scripts/__pycache__/diff_guard.cpython-311.pyc +0 -0
  52. package/.torusguard/scripts/__pycache__/finding_scorer.cpython-311.pyc +0 -0
  53. package/.torusguard/scripts/__pycache__/memory_engine.cpython-311.pyc +0 -0
  54. package/.torusguard/scripts/__pycache__/monorepo_detector.cpython-311.pyc +0 -0
  55. package/skills/torusguard/__pycache__/bootstrap.cpython-311.pyc +0 -0
@@ -19,21 +19,43 @@ from typing import Dict, List, Any, Optional, Tuple
19
19
  # Ensure UTF-8 stdout/stderr on Windows consoles
20
20
  if sys.stdout and hasattr(sys.stdout, "reconfigure"):
21
21
  try:
22
- sys.stdout.reconfigure(encoding="utf-8", errors="replace")
22
+ getattr(sys.stdout, "reconfigure")(encoding="utf-8", errors="replace")
23
23
  except Exception:
24
24
  pass
25
25
  if sys.stderr and hasattr(sys.stderr, "reconfigure"):
26
26
  try:
27
- sys.stderr.reconfigure(encoding="utf-8", errors="replace")
27
+ getattr(sys.stderr, "reconfigure")(encoding="utf-8", errors="replace")
28
28
  except Exception:
29
29
  pass
30
30
 
31
- VERSION = "1.0.0"
31
+ VERSION = "1.1.0"
32
32
  DEFAULT_TOKEN_BUDGET = 2000
33
33
  DEFAULT_TTL_DAYS = 90
34
34
  DEFAULT_DECAY_RATE = 0.15
35
35
 
36
36
 
37
+ def _utc_now() -> datetime.datetime:
38
+ """Return timezone-aware current UTC datetime."""
39
+ return datetime.datetime.now(datetime.timezone.utc)
40
+
41
+
42
+ def _utc_now_iso() -> str:
43
+ """Return ISO 8601 formatted UTC timestamp string ending with 'Z'."""
44
+ return datetime.datetime.now(datetime.timezone.utc).isoformat().replace("+00:00", "Z")
45
+
46
+
47
+ def _parse_iso_utc(ts_str: str) -> datetime.datetime:
48
+ """Parse ISO timestamp string and guarantee timezone-aware UTC datetime."""
49
+ clean = ts_str.replace("Z", "+00:00")
50
+ try:
51
+ dt = datetime.datetime.fromisoformat(clean)
52
+ except ValueError:
53
+ dt = datetime.datetime.fromisoformat(clean.split(".")[0] + "+00:00")
54
+ if dt.tzinfo is None:
55
+ dt = dt.replace(tzinfo=datetime.timezone.utc)
56
+ return dt
57
+
58
+
37
59
  def find_project_root(start_dir: Optional[str] = None) -> Path:
38
60
  """Detect project root directory by searching for standard repo root markers."""
39
61
  current = Path(start_dir or os.getcwd()).resolve()
@@ -104,13 +126,13 @@ def ensure_memory_structure(root_dir: Optional[Path] = None) -> Dict[str, Path]:
104
126
  "active_patterns_count": 0,
105
127
  "fix_rate_percentage": None,
106
128
  "top_vulnerabilities": [],
107
- "last_updated": datetime.datetime.utcnow().isoformat() + "Z"
129
+ "last_updated": _utc_now_iso()
108
130
  }, indent=2), encoding="utf-8")
109
131
 
110
132
  if not paths["context"].exists():
111
133
  initial_context = {
112
134
  "version": VERSION,
113
- "generated_at": datetime.datetime.utcnow().isoformat() + "Z",
135
+ "generated_at": _utc_now_iso(),
114
136
  "token_estimate": 0,
115
137
  "max_token_budget": DEFAULT_TOKEN_BUDGET,
116
138
  "project_profile": {
@@ -156,8 +178,8 @@ def record_event(
156
178
  raise ValueError(f"Invalid event_type: {event_type}. Must be one of {valid_types}")
157
179
 
158
180
  paths = ensure_memory_structure(root_dir)
159
- now_utc = datetime.datetime.utcnow()
160
- timestamp_iso = now_utc.isoformat() + "Z"
181
+ now_utc = _utc_now()
182
+ timestamp_iso = _utc_now_iso()
161
183
  event_id = f"evt-{now_utc.strftime('%Y%m%d%H%M%S')}-{uuid.uuid4().hex[:8]}"
162
184
 
163
185
  # Sanitize file_path to be relative to project root
@@ -251,7 +273,17 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
251
273
  events = load_all_events(root_dir)
252
274
  patterns: List[Dict[str, Any]] = []
253
275
 
254
- if not events:
276
+ # Preserve existing golden_fix_recipe patterns
277
+ if paths["patterns"].exists():
278
+ try:
279
+ prev_patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
280
+ for p in prev_patterns:
281
+ if p.get("pattern_type") == "golden_fix_recipe":
282
+ patterns.append(p)
283
+ except Exception:
284
+ pass
285
+
286
+ if not events and not patterns:
255
287
  paths["patterns"].write_text("[]", encoding="utf-8")
256
288
  compute_context_window(root_dir=root_dir)
257
289
  return patterns
@@ -288,7 +320,7 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
288
320
  # 1. Distill Recurring Fixes & Security Idioms
289
321
  for (rule_id, fix_strat), fix_evts in rule_fixes.items():
290
322
  occurrences = len(fix_evts)
291
- affected_files = sorted(list({e.get("file_path") for e in fix_evts if e.get("file_path")}))
323
+ affected_files = sorted(list({str(e["file_path"]) for e in fix_evts if e.get("file_path")}))
292
324
  verified_count = sum(1 for e in fix_evts if e.get("verification_result") == "fixed")
293
325
 
294
326
  # Confidence amplification logic
@@ -298,8 +330,8 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
298
330
  base_confidence = min(98, base_confidence + 5)
299
331
 
300
332
  pattern_type = "security_idiom" if occurrences >= 3 and verified_count >= 2 else "recurring_fix"
301
- timestamps = [e.get("timestamp") for e in fix_evts if e.get("timestamp")]
302
- first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
333
+ timestamps: List[str] = [str(e["timestamp"]) for e in fix_evts if e.get("timestamp")]
334
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
303
335
  last_seen = max(timestamps) if timestamps else first_seen
304
336
 
305
337
  patterns.append({
@@ -322,9 +354,9 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
322
354
  for rule_id, find_evts in rule_findings.items():
323
355
  occurrences = len(find_evts)
324
356
  if occurrences >= 2:
325
- affected_files = sorted(list({e.get("file_path") for e in find_evts if e.get("file_path")}))
326
- timestamps = [e.get("timestamp") for e in find_evts if e.get("timestamp")]
327
- first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
357
+ affected_files = sorted(list({str(e["file_path"]) for e in find_evts if e.get("file_path")}))
358
+ timestamps: List[str] = [str(e["timestamp"]) for e in find_evts if e.get("timestamp")]
359
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
328
360
  last_seen = max(timestamps) if timestamps else first_seen
329
361
 
330
362
  confidence = min(90, 55 + (occurrences * 5))
@@ -347,11 +379,11 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
347
379
  # 3. Distill False Positive Classes
348
380
  for rule_id, fp_evts in false_positives.items():
349
381
  occurrences = len(fp_evts)
350
- affected_files = sorted(list({e.get("file_path") for e in fp_evts if e.get("file_path")}))
382
+ affected_files = sorted(list({str(e["file_path"]) for e in fp_evts if e.get("file_path")}))
351
383
  reasons = [e.get("suppression_reason") for e in fp_evts if e.get("suppression_reason")]
352
384
  summary_reason = reasons[-1] if reasons else "Suppressed by team policy"
353
- timestamps = [e.get("timestamp") for e in fp_evts if e.get("timestamp")]
354
- first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
385
+ timestamps: List[str] = [str(e["timestamp"]) for e in fp_evts if e.get("timestamp")]
386
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
355
387
  last_seen = max(timestamps) if timestamps else first_seen
356
388
 
357
389
  patterns.append({
@@ -373,9 +405,9 @@ def distill_patterns(root_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
373
405
  # 4. Distill Regression Watch Entries
374
406
  for rule_id, reg_evts in regressions.items():
375
407
  occurrences = len(reg_evts)
376
- affected_files = sorted(list({e.get("file_path") for e in reg_evts if e.get("file_path")}))
377
- timestamps = [e.get("timestamp") for e in reg_evts if e.get("timestamp")]
378
- first_seen = min(timestamps) if timestamps else datetime.datetime.utcnow().isoformat() + "Z"
408
+ affected_files = sorted(list({str(e["file_path"]) for e in reg_evts if e.get("file_path")}))
409
+ timestamps: List[str] = [str(e["timestamp"]) for e in reg_evts if e.get("timestamp")]
410
+ first_seen = min(timestamps) if timestamps else _utc_now_iso()
379
411
  last_seen = max(timestamps) if timestamps else first_seen
380
412
 
381
413
  patterns.append({
@@ -414,7 +446,7 @@ def decay_stale_entries(
414
446
  Reduces confidence to prevent stale architectural advice.
415
447
  """
416
448
  paths = ensure_memory_structure(root_dir)
417
- now = datetime.datetime.utcnow()
449
+ now = _utc_now()
418
450
 
419
451
  # Read config from decay.json if available
420
452
  try:
@@ -439,9 +471,7 @@ def decay_stale_entries(
439
471
  if not chk_str:
440
472
  continue
441
473
  try:
442
- # Parse ISO date string (strip Z if present)
443
- clean_ts = chk_str.rstrip("Z")
444
- last_dt = datetime.datetime.fromisoformat(clean_ts)
474
+ last_dt = _parse_iso_utc(chk_str)
445
475
  days_elapsed = (now - last_dt).days
446
476
 
447
477
  if days_elapsed >= ttl_days:
@@ -450,7 +480,7 @@ def decay_stale_entries(
450
480
  new_conf = max(10, old_conf - reduction)
451
481
  if new_conf != old_conf:
452
482
  pat["confidence"] = new_conf
453
- pat["decay_checkpoint"] = now.isoformat() + "Z"
483
+ pat["decay_checkpoint"] = _utc_now_iso()
454
484
  decayed_count += 1
455
485
  except Exception:
456
486
  continue
@@ -536,21 +566,168 @@ def get_project_profile(root_dir: Optional[Path] = None) -> Dict[str, Any]:
536
566
  "fixes_verified_count": verified_fixed_count,
537
567
  "fix_rate_percentage": fix_rate,
538
568
  "top_vulnerabilities": top_vulns,
539
- "last_updated": datetime.datetime.utcnow().isoformat() + "Z"
569
+ "last_updated": _utc_now_iso()
540
570
  }
541
-
542
571
  paths["profile"].write_text(json.dumps(profile, indent=2), encoding="utf-8")
543
572
  return profile
544
573
 
545
574
 
575
+ def compute_proximity_score(
576
+ pattern: Dict[str, Any],
577
+ target_file: Optional[str] = None,
578
+ target_rule_id: Optional[str] = None
579
+ ) -> int:
580
+ """
581
+ Compute file proximity and rule relevance score (0-100) for a memory pattern against a target query.
582
+ Enables file-scoped and rule-scoped context ranking.
583
+ """
584
+ score = 0
585
+ if not target_file and not target_rule_id:
586
+ return score
587
+
588
+ pat_rule = pattern.get("rule_id", "")
589
+ affected_files = [str(f).replace("\\", "/") for f in pattern.get("affected_files", [])]
590
+
591
+ # 1. Rule ID relevance: exact match = +40, family match (e.g. TG-DB-) = +20
592
+ if target_rule_id:
593
+ if pat_rule == target_rule_id:
594
+ score += 40
595
+ elif pat_rule and pat_rule.split("-")[:2] == target_rule_id.split("-")[:2]:
596
+ score += 20
597
+
598
+ # 2. File path relevance
599
+ if target_file:
600
+ norm_target = str(target_file).replace("\\", "/")
601
+ target_path = Path(norm_target)
602
+ target_ext = target_path.suffix.lower()
603
+ target_parts = set(p.lower() for p in target_path.parts if p not in (".", ".."))
604
+
605
+ best_file_score = 0
606
+ for aff in affected_files:
607
+ aff_score = 0
608
+ aff_path = Path(aff)
609
+ if aff == norm_target:
610
+ best_file_score = 50
611
+ break
612
+ # Same directory or subpath
613
+ if aff_path.parent == target_path.parent and str(target_path.parent) not in (".", ""):
614
+ aff_score = max(aff_score, 30)
615
+ elif any(part in target_parts for part in (p.lower() for p in aff_path.parts if p not in (".", ".."))):
616
+ aff_score = max(aff_score, 15)
617
+ # Extension match
618
+ if target_ext and aff_path.suffix.lower() == target_ext:
619
+ aff_score = max(aff_score, aff_score + 10)
620
+
621
+ if aff_score > best_file_score:
622
+ best_file_score = aff_score
623
+
624
+ # Also check file_type in recipe or pattern metadata
625
+ pat_file_type = pattern.get("file_type") or pattern.get("recipe_data", {}).get("file_type")
626
+ if pat_file_type and target_ext and pat_file_type.lower() == target_ext and best_file_score == 0:
627
+ best_file_score = 10
628
+
629
+ score += best_file_score
630
+
631
+ return min(score, 100)
632
+
633
+
634
+ def record_golden_recipe(
635
+ rule_id: str,
636
+ before_snippet: str,
637
+ after_snippet: str,
638
+ diff_snippet: str,
639
+ file_type: str = ".py",
640
+ framework: Optional[str] = None,
641
+ description: str = "Verified AST remediation recipe",
642
+ additions: Optional[int] = None,
643
+ deletions: Optional[int] = None,
644
+ root_dir: Optional[Path] = None
645
+ ) -> Dict[str, Any]:
646
+ """
647
+ Store or update a verified Golden Fix Recipe adhering to Ponytail bounds (<=35 add, <=25 del).
648
+ """
649
+ paths = ensure_memory_structure(root_dir)
650
+
651
+ if additions is None:
652
+ additions = sum(1 for line in diff_snippet.splitlines() if line.startswith("+") and not line.startswith("+++"))
653
+ if deletions is None:
654
+ deletions = sum(1 for line in diff_snippet.splitlines() if line.startswith("-") and not line.startswith("---"))
655
+
656
+ if additions > 35 or deletions > 25:
657
+ raise ValueError(f"Recipe exceeds Ponytail bounds: +{additions}/35 add, -{deletions}/25 del")
658
+
659
+ recipe_hash = hashlib.sha256(diff_snippet.strip().encode("utf-8")).hexdigest()[:8]
660
+ recipe_id = f"recipe-{rule_id}-{recipe_hash}"
661
+ timestamp = _utc_now_iso()
662
+
663
+ recipe = {
664
+ "recipe_id": recipe_id,
665
+ "rule_id": rule_id,
666
+ "file_type": file_type,
667
+ "framework": framework,
668
+ "description": description,
669
+ "before_snippet": before_snippet.strip(),
670
+ "after_snippet": after_snippet.strip(),
671
+ "diff_snippet": diff_snippet.strip(),
672
+ "ponytail_metrics": {
673
+ "additions": additions,
674
+ "deletions": deletions
675
+ },
676
+ "verified_count": 1,
677
+ "last_verified": timestamp
678
+ }
679
+
680
+ patterns: List[Dict[str, Any]] = []
681
+ if paths["patterns"].exists():
682
+ try:
683
+ patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
684
+ except Exception:
685
+ patterns = []
686
+
687
+ for p in patterns:
688
+ if p.get("pattern_type") == "golden_fix_recipe" and p.get("recipe_id") == recipe_id:
689
+ count = p.get("verified_count", 1) + 1
690
+ p["verified_count"] = count
691
+ recipe["verified_count"] = count
692
+ p["last_verified"] = timestamp
693
+ p["confidence"] = min(99, p.get("confidence", 85) + 5)
694
+ p["recipe_data"] = recipe
695
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
696
+ return p
697
+
698
+ recipe_pattern: Dict[str, Any] = {
699
+ "pattern_id": f"PAT-RECIPE-{recipe_id}",
700
+ "rule_id": rule_id,
701
+ "pattern_type": "golden_fix_recipe",
702
+ "recipe_id": recipe_id,
703
+ "file_type": file_type,
704
+ "framework": framework,
705
+ "description": description,
706
+ "recipe_data": recipe,
707
+ "confidence": 90,
708
+ "occurrences": 1,
709
+ "affected_files": [],
710
+ "first_seen": timestamp,
711
+ "last_seen": timestamp
712
+ }
713
+ patterns.append(recipe_pattern)
714
+ paths["patterns"].write_text(json.dumps(patterns, indent=2), encoding="utf-8")
715
+ return recipe_pattern
716
+
717
+
546
718
  def compute_context_window(
547
719
  max_tokens: int = DEFAULT_TOKEN_BUDGET,
720
+ target_role: str = "all",
721
+ target_file: Optional[str] = None,
722
+ rule_id: Optional[str] = None,
723
+ target_rule_id: Optional[str] = None,
548
724
  root_dir: Optional[Path] = None
549
725
  ) -> Dict[str, Any]:
550
726
  """
551
- Build the pre-computed, token-budgeted context window as structured JSON cards.
552
- Guarantees strict token enforcement <= max_tokens.
727
+ Build the pre-computed or role-tailored context window as structured JSON cards.
728
+ Guarantees strict token enforcement <= max_tokens with proximity and role weighting.
553
729
  """
730
+ rule_id = target_rule_id or rule_id
554
731
  paths = ensure_memory_structure(root_dir)
555
732
  profile = get_project_profile(root_dir=root_dir)
556
733
 
@@ -561,77 +738,140 @@ def compute_context_window(
561
738
  except Exception:
562
739
  patterns = []
563
740
 
564
- # Sort patterns by priority: regressions (95) -> idioms (90) -> recurring fixes (85) -> FP (80) -> common vulns (70)
565
- type_priority = {
566
- "regression_watch": 95,
567
- "security_idiom": 90,
568
- "recurring_fix": 85,
569
- "false_positive_class": 80,
570
- "common_vulnerability": 70
741
+ # Persona-tailored base priority matrices
742
+ role_priority_matrices = {
743
+ "all": {
744
+ "regression_watch": 95,
745
+ "golden_fix_recipe": 90,
746
+ "security_idiom": 85,
747
+ "recurring_fix": 80,
748
+ "false_positive_class": 75,
749
+ "common_vulnerability": 70
750
+ },
751
+ "auditor": {
752
+ "false_positive_class": 100,
753
+ "common_vulnerability": 95,
754
+ "regression_watch": 90,
755
+ "recurring_fix": 80,
756
+ "security_idiom": 60,
757
+ "golden_fix_recipe": 40
758
+ },
759
+ "remediator": {
760
+ "golden_fix_recipe": 115,
761
+ "security_idiom": 100,
762
+ "recurring_fix": 90,
763
+ "regression_watch": 80,
764
+ "common_vulnerability": 60,
765
+ "false_positive_class": 50
766
+ },
767
+ "reviewer": {
768
+ "regression_watch": 115,
769
+ "false_positive_class": 95,
770
+ "golden_fix_recipe": 85,
771
+ "recurring_fix": 75,
772
+ "security_idiom": 65,
773
+ "common_vulnerability": 55
774
+ }
571
775
  }
572
776
 
777
+ type_priority = role_priority_matrices.get(target_role, role_priority_matrices["all"])
778
+
573
779
  # Generate Candidate Cards
574
780
  cards: List[Dict[str, Any]] = []
575
781
  card_idx = 1
576
782
 
577
- # 1. Project Profile Card (Priority: 100)
783
+ # 1. Project Profile Card (Always top priority)
578
784
  cards.append({
579
785
  "card_id": f"CARD-{card_idx:03d}",
580
786
  "card_type": "profile",
581
- "priority": 100,
787
+ "type": "profile",
788
+ "priority": 150,
789
+ "proximity_score": 0,
582
790
  "title": "Project Security Posture & DNA",
583
791
  "summary": f"Stack: {', '.join(profile.get('stack') or ['Generic'])}; Total Events: {profile.get('total_events', 0)}; Fix Rate: {profile.get('fix_rate_percentage')}%",
584
792
  "card_data": {
585
793
  "stack": profile.get("stack", []),
586
794
  "total_events": profile.get("total_events", 0),
587
795
  "fix_rate_percentage": profile.get("fix_rate_percentage"),
588
- "top_vulnerabilities": profile.get("top_vulnerabilities", [])
796
+ "top_vulnerabilities": profile.get("top_vulnerabilities", []),
797
+ "target_role": target_role
589
798
  }
590
799
  })
591
800
  card_idx += 1
592
801
 
593
- # Sort patterns by type priority then confidence then occurrences
594
- sorted_patterns = sorted(
595
- patterns,
596
- key=lambda p: (
597
- type_priority.get(p.get("pattern_type", ""), 50),
598
- p.get("confidence", 0),
599
- p.get("occurrences", 0)
802
+ # Score patterns: base priority + proximity score
803
+ scored_patterns = []
804
+ for pat in patterns:
805
+ ptype = pat.get("pattern_type", "pattern")
806
+ base_prio = type_priority.get(ptype, 50)
807
+ proximity = compute_proximity_score(pat, target_file=target_file, target_rule_id=rule_id)
808
+ effective_prio = base_prio + proximity
809
+ scored_patterns.append((effective_prio, proximity, pat))
810
+
811
+ # Sort patterns by effective priority -> confidence -> occurrences
812
+ scored_patterns.sort(
813
+ key=lambda x: (
814
+ x[0],
815
+ x[2].get("confidence", 0),
816
+ x[2].get("occurrences", 0)
600
817
  ),
601
818
  reverse=True
602
819
  )
603
820
 
604
- for pat in sorted_patterns:
821
+ for eff_prio, prox_score, pat in scored_patterns:
605
822
  ptype = pat.get("pattern_type", "pattern")
606
823
  card_type = "pattern"
824
+ title = f"[{pat.get('rule_id')}] {pat.get('pattern_type')}: {pat.get('description', '')[:60]}"
825
+ summary = pat.get("description", "")
826
+ card_data: Dict[str, Any] = {
827
+ "rule_id": pat.get("rule_id"),
828
+ "pattern_type": pat.get("pattern_type"),
829
+ "confidence": pat.get("confidence"),
830
+ "occurrences": pat.get("occurrences"),
831
+ "affected_files": pat.get("affected_files", [])[:5]
832
+ }
833
+
607
834
  if ptype == "regression_watch":
608
835
  card_type = "regression_watch"
609
836
  elif ptype == "false_positive_class":
610
- card_type = "false_positive"
837
+ card_type = "false_positive_suppression"
838
+ elif ptype == "common_vulnerability":
839
+ card_type = "common_vulnerability"
611
840
  elif ptype == "security_idiom":
612
841
  card_type = "fix_idiom"
842
+ card_data["fix_strategy"] = pat.get("fix_strategy")
843
+ elif ptype == "golden_fix_recipe":
844
+ card_type = "golden_recipe"
845
+ recipe_obj = pat.get("recipe_data", {})
846
+ title = f"[{pat.get('rule_id')}] Golden Recipe: {pat.get('description', '')[:50]}"
847
+ card_data["diff_snippet"] = recipe_obj.get("diff_snippet", "")
848
+ card_data["ponytail_metrics"] = recipe_obj.get("ponytail_metrics", {})
849
+ card_data["framework"] = recipe_obj.get("framework")
613
850
 
614
851
  cards.append({
615
852
  "card_id": f"CARD-{card_idx:03d}",
616
853
  "card_type": card_type,
617
- "priority": type_priority.get(ptype, 60),
618
- "title": f"[{pat.get('rule_id')}] {pat.get('pattern_type')}: {pat.get('description', '')[:60]}",
619
- "summary": pat.get("description", ""),
620
- "card_data": {
621
- "rule_id": pat.get("rule_id"),
622
- "pattern_type": pat.get("pattern_type"),
623
- "fix_strategy": pat.get("fix_strategy"),
624
- "confidence": pat.get("confidence"),
625
- "occurrences": pat.get("occurrences"),
626
- "affected_files": pat.get("affected_files", [])[:5]
627
- }
854
+ "type": card_type,
855
+ "priority": eff_prio,
856
+ "proximity_score": prox_score,
857
+ "title": title,
858
+ "summary": summary,
859
+ "card_data": card_data
628
860
  })
629
861
  card_idx += 1
630
862
 
631
863
  # Build Context Window Payload
632
864
  context = {
633
865
  "version": VERSION,
634
- "generated_at": datetime.datetime.utcnow().isoformat() + "Z",
866
+ "generated_at": _utc_now_iso(),
867
+ "role": target_role,
868
+ "target_role": target_role,
869
+ "target_query": {
870
+ "file": target_file,
871
+ "rule_id": rule_id
872
+ },
873
+ "target_file": target_file,
874
+ "rule_id": rule_id,
635
875
  "token_estimate": 0,
636
876
  "max_token_budget": max_tokens,
637
877
  "project_profile": {
@@ -645,27 +885,43 @@ def compute_context_window(
645
885
  }
646
886
 
647
887
  # Token budget enforcement: Drop lowest-priority cards until within limit
648
- # Always keep at least the profile card (cards[0])
649
888
  while len(context["cards"]) > 1 and estimate_tokens(context) > max_tokens:
650
889
  context["cards"].pop()
651
890
 
652
891
  context["token_estimate"] = estimate_tokens(context)
653
892
 
654
- # Persist context.json
655
- paths["context"].write_text(json.dumps(context, indent=2), encoding="utf-8")
893
+ # Persist default context.json if generating for role="all" without filters
894
+ if target_role == "all" and not target_file and not rule_id:
895
+ paths["context"].write_text(json.dumps(context, indent=2), encoding="utf-8")
896
+
656
897
  return context
657
898
 
658
899
 
659
- def get_context(root_dir: Optional[Path] = None) -> Dict[str, Any]:
660
- """Retrieve the current pre-computed context window."""
661
- paths = ensure_memory_structure(root_dir)
662
- if paths["context"].exists():
663
- try:
664
- return json.loads(paths["context"].read_text(encoding="utf-8"))
665
- except Exception:
666
- pass
900
+ def get_context(
901
+ target_role: str = "all",
902
+ target_file: Optional[str] = None,
903
+ rule_id: Optional[str] = None,
904
+ max_tokens: int = DEFAULT_TOKEN_BUDGET,
905
+ root_dir: Optional[Path] = None
906
+ ) -> Dict[str, Any]:
907
+ """Retrieve or compute persona-tailored and file-scoped context window."""
908
+ if target_role == "all" and not target_file and not rule_id:
909
+ paths = ensure_memory_structure(root_dir)
910
+ if paths["context"].exists():
911
+ try:
912
+ data = json.loads(paths["context"].read_text(encoding="utf-8"))
913
+ if data.get("version") == VERSION:
914
+ return data
915
+ except Exception:
916
+ pass
667
917
 
668
- return compute_context_window(root_dir=root_dir)
918
+ return compute_context_window(
919
+ max_tokens=max_tokens,
920
+ target_role=target_role,
921
+ target_file=target_file,
922
+ rule_id=rule_id,
923
+ root_dir=root_dir
924
+ )
669
925
 
670
926
 
671
927
  def record_false_positive(
@@ -688,45 +944,70 @@ def record_false_positive(
688
944
  return evt
689
945
 
690
946
 
691
- def export_memory(target_path: str, root_dir: Optional[Path] = None) -> Dict[str, Any]:
947
+ def export_memory(
948
+ target_path: Optional[Any] = None,
949
+ target_file: Optional[Any] = None,
950
+ sanitized: bool = False,
951
+ root_dir: Optional[Path] = None
952
+ ) -> Dict[str, Any]:
692
953
  """
693
954
  Export memory to an external file for team sharing.
694
- Per user choice: retain code hashes for high-fidelity matching, but sanitize absolute paths.
955
+ Supports sanitized mode for safe repository version control (strips absolute paths & secrets).
695
956
  """
696
957
  paths = ensure_memory_structure(root_dir)
697
958
  events = load_all_events(root_dir)
698
959
  profile = get_project_profile(root_dir=root_dir)
699
960
 
700
- patterns = []
961
+ patterns: List[Dict[str, Any]] = []
701
962
  if paths["patterns"].exists():
702
963
  try:
703
964
  patterns = json.loads(paths["patterns"].read_text(encoding="utf-8"))
704
965
  except Exception:
705
966
  pass
706
967
 
707
- decay_cfg = {}
968
+ decay_cfg: Dict[str, Any] = {}
708
969
  if paths["decay"].exists():
709
970
  try:
710
971
  decay_cfg = json.loads(paths["decay"].read_text(encoding="utf-8"))
711
972
  except Exception:
712
973
  pass
713
974
 
975
+ if sanitized:
976
+ sanitized_events = []
977
+ for e in events:
978
+ ce = dict(e)
979
+ if "file_path" in ce and ce["file_path"]:
980
+ ce["file_path"] = Path(ce["file_path"]).name
981
+ sanitized_events.append(ce)
982
+ events = sanitized_events
983
+
984
+ sanitized_patterns = []
985
+ for p in patterns:
986
+ cp = dict(p)
987
+ if "affected_files" in cp:
988
+ cp["affected_files"] = [Path(f).name for f in cp["affected_files"]]
989
+ sanitized_patterns.append(cp)
990
+ patterns = sanitized_patterns
991
+
714
992
  export_payload = {
715
993
  "format": "torusguard-memory-bundle",
716
- "schema_version": "1.0.0",
717
- "exported_at": datetime.datetime.utcnow().isoformat() + "Z",
994
+ "schema_version": VERSION,
995
+ "sanitized": sanitized,
996
+ "exported_at": _utc_now_iso(),
718
997
  "project_profile": profile,
719
998
  "decay_config": decay_cfg,
720
999
  "patterns": patterns,
721
1000
  "events": events
722
1001
  }
723
1002
 
724
- out_file = Path(target_path).resolve()
1003
+ dest = target_file or target_path or "torusguard-memory-export.json"
1004
+ out_file = Path(dest).resolve()
725
1005
  out_file.parent.mkdir(parents=True, exist_ok=True)
726
1006
  out_file.write_text(json.dumps(export_payload, indent=2), encoding="utf-8")
727
1007
 
728
1008
  return {
729
1009
  "target_path": str(out_file),
1010
+ "sanitized": sanitized,
730
1011
  "exported_events_count": len(events),
731
1012
  "exported_patterns_count": len(patterns)
732
1013
  }
@@ -767,35 +1048,34 @@ def import_memory(source_path: str, merge: bool = True, root_dir: Optional[Path]
767
1048
 
768
1049
 
769
1050
  def compact_events(older_than_days: int = 30, root_dir: Optional[Path] = None) -> int:
770
- """
771
- Compact loose event JSON files older than older_than_days into compacted_archive.json.
772
- Prevents filesystem inode saturation while preserving full history.
773
- """
1051
+ """Archive events older than older_than_days into compacted_archive.json."""
774
1052
  paths = ensure_memory_structure(root_dir)
775
- cutoff = datetime.datetime.utcnow() - datetime.timedelta(days=older_than_days)
1053
+ cutoff = _utc_now() - datetime.timedelta(days=older_than_days)
776
1054
 
777
1055
  archived: List[Dict[str, Any]] = []
1056
+ archived_ids = set()
1057
+
778
1058
  if paths["compacted"].exists():
779
1059
  try:
780
- with open(paths["compacted"], "r", encoding="utf-8") as f:
781
- data = json.load(f)
782
- if isinstance(data, list):
783
- archived = data
1060
+ archived = json.loads(paths["compacted"].read_text(encoding="utf-8"))
1061
+ for item in archived:
1062
+ eid = item.get("event_id")
1063
+ if eid:
1064
+ archived_ids.add(eid)
784
1065
  except Exception:
785
1066
  archived = []
786
1067
 
787
- archived_ids = {e.get("event_id") for e in archived if e.get("event_id")}
788
1068
  compacted_count = 0
789
-
790
- for item in sorted(paths["events"].glob("*.json")):
1069
+ for item in paths["events"].glob("*.json"):
791
1070
  if item.name == "compacted_archive.json":
792
1071
  continue
793
1072
  try:
794
- with open(item, "r", encoding="utf-8") as f:
795
- evt = json.load(f)
796
- ts_str = evt.get("timestamp", "").rstrip("Z")
797
- evt_dt = datetime.datetime.fromisoformat(ts_str) if ts_str else None
798
- if evt_dt and evt_dt < cutoff:
1073
+ evt = json.loads(item.read_text(encoding="utf-8"))
1074
+ ts_str = evt.get("timestamp")
1075
+ if not ts_str:
1076
+ continue
1077
+ ts = _parse_iso_utc(ts_str)
1078
+ if ts < cutoff:
799
1079
  eid = evt.get("event_id")
800
1080
  if eid and eid not in archived_ids:
801
1081
  archived.append(evt)
@@ -811,11 +1091,116 @@ def compact_events(older_than_days: int = 30, root_dir: Optional[Path] = None) -
811
1091
  return compacted_count
812
1092
 
813
1093
 
1094
+ def install_git_hook(root_dir: Optional[Path] = None) -> Dict[str, Any]:
1095
+ """Install a pre-commit git hook running TorusGuard diff_guard."""
1096
+ base = Path(root_dir or find_project_root()).resolve()
1097
+ git_dir = base / ".git"
1098
+ if not git_dir.exists():
1099
+ raise RuntimeError(f"Not a git repository: {base}")
1100
+
1101
+ hooks_dir = git_dir / "hooks"
1102
+ hooks_dir.mkdir(parents=True, exist_ok=True)
1103
+ pre_commit_hook = hooks_dir / "pre-commit"
1104
+
1105
+ hook_content = (
1106
+ "#!/bin/sh\n"
1107
+ "# TorusGuard Autonomous Security & Regression Guard\n"
1108
+ "python .torusguard/scripts/diff_guard.py --pre-commit\n"
1109
+ "EXIT_CODE=$?\n"
1110
+ "if [ $EXIT_CODE -ne 0 ]; then\n"
1111
+ " echo \"[BLOCKED] Commit rejected by TorusGuard diff security guardrails.\"\n"
1112
+ " exit $EXIT_CODE\n"
1113
+ "fi\n"
1114
+ "exit 0\n"
1115
+ )
1116
+
1117
+ pre_commit_hook.write_text(hook_content, encoding="utf-8")
1118
+ try:
1119
+ os.chmod(pre_commit_hook, 0o755)
1120
+ except Exception:
1121
+ pass
1122
+
1123
+ return {
1124
+ "status": "installed",
1125
+ "hook_path": str(pre_commit_hook)
1126
+ }
1127
+
1128
+
1129
+ def uninstall_git_hook(root_dir: Optional[Path] = None) -> Dict[str, Any]:
1130
+ """Uninstall the TorusGuard pre-commit git hook."""
1131
+ base = Path(root_dir or find_project_root()).resolve()
1132
+ pre_commit_hook = base / ".git" / "hooks" / "pre-commit"
1133
+ if pre_commit_hook.exists():
1134
+ content = pre_commit_hook.read_text(encoding="utf-8", errors="replace")
1135
+ if "TorusGuard" in content:
1136
+ pre_commit_hook.unlink()
1137
+ return {"status": "uninstalled", "hook_path": str(pre_commit_hook)}
1138
+ else:
1139
+ return {"status": "skipped", "reason": "Pre-commit hook does not belong to TorusGuard"}
1140
+ return {"status": "not_found"}
1141
+
1142
+
1143
+ def learn_from_git(commit_range: str = "HEAD~10..HEAD", root_dir: Optional[Path] = None) -> Dict[str, Any]:
1144
+ """Ingest developer security fixes from git commits into local memory."""
1145
+ import subprocess
1146
+ base = Path(root_dir or find_project_root()).resolve()
1147
+ learned_events = []
1148
+
1149
+ try:
1150
+ res = subprocess.run(
1151
+ ["git", "log", "-n", "10", "--pretty=format:%H|||%s", "--name-only"],
1152
+ cwd=str(base),
1153
+ capture_output=True,
1154
+ text=True,
1155
+ encoding="utf-8",
1156
+ errors="replace"
1157
+ )
1158
+ output = res.stdout
1159
+ except Exception as e:
1160
+ return {"commits_scanned": 0, "learned_events_count": 0, "events_recorded": 0, "error": str(e)}
1161
+
1162
+ current_hash = ""
1163
+ current_subject = ""
1164
+ for line in output.splitlines():
1165
+ line = line.strip()
1166
+ if not line:
1167
+ continue
1168
+ if "|||" in line:
1169
+ parts = line.split("|||", 1)
1170
+ current_hash = parts[0]
1171
+ current_subject = parts[1]
1172
+ else:
1173
+ file_mod = line
1174
+ sub_lower = current_subject.lower()
1175
+ if any(k in sub_lower for k in ["security", "fix", "vuln", "cve", "sanitize", "auth", "tenant"]):
1176
+ evt = record_event(
1177
+ "pattern_learned",
1178
+ {
1179
+ "source": "git_history",
1180
+ "commit_hash": current_hash[:8],
1181
+ "commit_message": current_subject,
1182
+ "file_path": file_mod
1183
+ },
1184
+ root_dir=base
1185
+ )
1186
+ learned_events.append(evt)
1187
+
1188
+ if learned_events:
1189
+ distill_patterns(root_dir=base)
1190
+
1191
+ return {
1192
+ "commits_scanned": 10,
1193
+ "learned_events_count": len(learned_events),
1194
+ "events_recorded": len(learned_events)
1195
+ }
1196
+
1197
+
814
1198
  # ─── Command Line Interface ──────────────────────────────────────────────────
815
1199
  def main():
816
1200
  parser = argparse.ArgumentParser(description="TorusGuard Security Memory Engine")
817
1201
  parser.add_argument("--action", required=True, choices=[
818
- "record", "distill", "context", "profile", "decay", "fp", "export", "import", "compact", "status"
1202
+ "record", "distill", "context", "profile", "decay", "fp", "export", "import", "compact", "status",
1203
+ "recipe", "hook-install", "hook-uninstall", "learn"
819
1204
  ], help="Action to perform")
820
1205
  parser.add_argument("--root", help="Project root directory override")
821
1206
  parser.add_argument("--type", help="Event type (audit_finding, fix_applied, etc.)")
@@ -827,8 +1212,13 @@ def main():
827
1212
  parser.add_argument("--strategy", help="Fix strategy description")
828
1213
  parser.add_argument("--result", choices=["fixed", "regressed", "partial", "not_tested"])
829
1214
  parser.add_argument("--reason", help="Suppression reason for false positive")
1215
+ parser.add_argument("--role", choices=["all", "auditor", "remediator", "reviewer"], default="all", help="Target agent role for context generation")
830
1216
  parser.add_argument("--target", help="Export target path")
831
1217
  parser.add_argument("--source", help="Import source path")
1218
+ parser.add_argument("--sanitized", action="store_true", help="Sanitize paths & secrets for export")
1219
+ parser.add_argument("--before", help="Before code snippet for golden recipe")
1220
+ parser.add_argument("--after", help="After code snippet for golden recipe")
1221
+ parser.add_argument("--diff", help="Diff snippet for golden recipe")
832
1222
  parser.add_argument("--ttl", type=int, default=DEFAULT_TTL_DAYS, help="Decay TTL in days")
833
1223
  parser.add_argument("--older-than", type=int, default=30, help="Compaction age in days")
834
1224
  parser.add_argument("--json", action="store_true", help="Output raw JSON")
@@ -862,7 +1252,12 @@ def main():
862
1252
  print(f"Distilled {len(pats)} active patterns.")
863
1253
 
864
1254
  elif args.action == "context":
865
- ctx = get_context(root_dir=root)
1255
+ ctx = get_context(
1256
+ target_role=args.role,
1257
+ target_file=args.file,
1258
+ rule_id=args.rule_id,
1259
+ root_dir=root
1260
+ )
866
1261
  print(json.dumps(ctx, indent=2))
867
1262
 
868
1263
  elif args.action == "profile":
@@ -880,11 +1275,38 @@ def main():
880
1275
  evt = record_false_positive(args.rule_id, file_path=args.file, reason=args.reason or "False positive", root_dir=root)
881
1276
  print(f"Suppressed false positive for {args.rule_id}")
882
1277
 
1278
+ elif args.action == "recipe":
1279
+ if not args.rule_id or not args.diff:
1280
+ print("Error: --rule-id and --diff are required for recipe action", file=sys.stderr)
1281
+ sys.exit(1)
1282
+ rec = record_golden_recipe(
1283
+ rule_id=args.rule_id,
1284
+ before_snippet=args.before or "",
1285
+ after_snippet=args.after or "",
1286
+ diff_snippet=args.diff,
1287
+ file_type=Path(args.file).suffix if args.file else ".py",
1288
+ description=args.strategy or "Verified AST remediation recipe",
1289
+ root_dir=root
1290
+ )
1291
+ print(json.dumps(rec, indent=2) if args.json else f"Recorded golden recipe: {rec['recipe_id']}")
1292
+
1293
+ elif args.action == "hook-install":
1294
+ res_h = install_git_hook(root_dir=root)
1295
+ print(json.dumps(res_h, indent=2) if args.json else f"Git pre-commit hook installed: {res_h['hook_path']}")
1296
+
1297
+ elif args.action == "hook-uninstall":
1298
+ res_u = uninstall_git_hook(root_dir=root)
1299
+ print(json.dumps(res_u, indent=2) if args.json else f"Git pre-commit hook {res_u['status']}")
1300
+
1301
+ elif args.action == "learn":
1302
+ res_l = learn_from_git(root_dir=root)
1303
+ print(json.dumps(res_l, indent=2) if args.json else f"Learned {res_l['learned_events_count']} events from git history.")
1304
+
883
1305
  elif args.action == "export":
884
1306
  if not args.target:
885
1307
  print("Error: --target is required for export action", file=sys.stderr)
886
1308
  sys.exit(1)
887
- res = export_memory(args.target, root_dir=root)
1309
+ res = export_memory(args.target, sanitized=args.sanitized, root_dir=root)
888
1310
  print(json.dumps(res, indent=2) if args.json else f"Exported {res['exported_events_count']} events to {res['target_path']}")
889
1311
 
890
1312
  elif args.action == "import":