thinkstack-core 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (205) hide show
  1. thinkstack_core/__init__.py +158 -0
  2. thinkstack_core/aggphi_textual.py +275 -0
  3. thinkstack_core/alerts/__init__.py +23 -0
  4. thinkstack_core/alerts/base.py +46 -0
  5. thinkstack_core/alerts/config.py +60 -0
  6. thinkstack_core/alerts/dispatcher.py +110 -0
  7. thinkstack_core/alerts/jira.py +96 -0
  8. thinkstack_core/alerts/linear.py +72 -0
  9. thinkstack_core/alerts/pagerduty.py +66 -0
  10. thinkstack_core/alerts/slack.py +81 -0
  11. thinkstack_core/alerts/teams.py +70 -0
  12. thinkstack_core/audit/__init__.py +43 -0
  13. thinkstack_core/audit/exporter.py +297 -0
  14. thinkstack_core/audit/privacy.py +101 -0
  15. thinkstack_core/audit/scrubber.py +149 -0
  16. thinkstack_core/audit/service.py +67 -0
  17. thinkstack_core/audit/signing.py +127 -0
  18. thinkstack_core/broadcast/__init__.py +4 -0
  19. thinkstack_core/broadcast/broadcaster.py +100 -0
  20. thinkstack_core/broadcast/watcher.py +71 -0
  21. thinkstack_core/capability.py +639 -0
  22. thinkstack_core/cloud/__init__.py +1 -0
  23. thinkstack_core/cloud/client_config.py +472 -0
  24. thinkstack_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
  25. thinkstack_core/cloud/client_configs/.claude-stdio.json +13 -0
  26. thinkstack_core/cloud/client_configs/.cursor-mcp.json +13 -0
  27. thinkstack_core/cloud/client_configs/.opencode-bridge.json +13 -0
  28. thinkstack_core/cloud/client_configs/.opencode.json +15 -0
  29. thinkstack_core/cloud/client_configs/.vscode-mcp.json +13 -0
  30. thinkstack_core/cloud/mcp_client.py +229 -0
  31. thinkstack_core/cloud/setup.py +144 -0
  32. thinkstack_core/cloud/sync.py +143 -0
  33. thinkstack_core/cloud/sync_bundle.py +639 -0
  34. thinkstack_core/cloud/sync_conflicts.py +183 -0
  35. thinkstack_core/cloud/sync_state.py +159 -0
  36. thinkstack_core/cloud/team_sync.py +337 -0
  37. thinkstack_core/cloud/thinkstack-mcp-bridge.js +357 -0
  38. thinkstack_core/codex/__init__.py +9 -0
  39. thinkstack_core/codex/__main__.py +97 -0
  40. thinkstack_core/codex/capture.py +208 -0
  41. thinkstack_core/codex/proxy.py +412 -0
  42. thinkstack_core/compat.py +103 -0
  43. thinkstack_core/concept_catalog.py +209 -0
  44. thinkstack_core/consolidation/__init__.py +3 -0
  45. thinkstack_core/consolidation/synthesizer.py +87 -0
  46. thinkstack_core/consolidation/workflow.py +175 -0
  47. thinkstack_core/daemon/__init__.py +27 -0
  48. thinkstack_core/daemon/supervisor.py +293 -0
  49. thinkstack_core/daemon/watcher.py +244 -0
  50. thinkstack_core/dashboard_api.py +2012 -0
  51. thinkstack_core/deltaf.py +97 -0
  52. thinkstack_core/disclosure.py +50 -0
  53. thinkstack_core/divergence/__init__.py +3 -0
  54. thinkstack_core/divergence/detector.py +166 -0
  55. thinkstack_core/gateway/__init__.py +32 -0
  56. thinkstack_core/gateway/key_manager.py +124 -0
  57. thinkstack_core/gateway/metrics_webhook.py +252 -0
  58. thinkstack_core/gateway/policy.py +262 -0
  59. thinkstack_core/gateway/server.py +727 -0
  60. thinkstack_core/gateway/sso.py +233 -0
  61. thinkstack_core/gcc.py +1246 -0
  62. thinkstack_core/github/__init__.py +35 -0
  63. thinkstack_core/github/app.py +240 -0
  64. thinkstack_core/github/comment_builder.py +113 -0
  65. thinkstack_core/github/pat.py +76 -0
  66. thinkstack_core/github/pr_parser.py +82 -0
  67. thinkstack_core/github/pr_reporter.py +555 -0
  68. thinkstack_core/gitlab/__init__.py +177 -0
  69. thinkstack_core/hitl/__init__.py +4 -0
  70. thinkstack_core/hitl/channels.py +129 -0
  71. thinkstack_core/hitl/orchestrator.py +95 -0
  72. thinkstack_core/hooks/__init__.py +17 -0
  73. thinkstack_core/hooks/claude_code.py +228 -0
  74. thinkstack_core/hooks/git_capture.py +341 -0
  75. thinkstack_core/hooks/git_commit.py +182 -0
  76. thinkstack_core/hooks/installer.py +850 -0
  77. thinkstack_core/hooks/pre_commit.py +157 -0
  78. thinkstack_core/hooks/runner.py +386 -0
  79. thinkstack_core/identity/__init__.py +4 -0
  80. thinkstack_core/identity/agent.py +86 -0
  81. thinkstack_core/identity/providers.py +85 -0
  82. thinkstack_core/invariants.py +182 -0
  83. thinkstack_core/mcp/__init__.py +10 -0
  84. thinkstack_core/mcp/auth.py +177 -0
  85. thinkstack_core/mcp/server.py +1215 -0
  86. thinkstack_core/metrics/__init__.py +35 -0
  87. thinkstack_core/metrics/aggregate.py +215 -0
  88. thinkstack_core/metrics/calibrate.py +198 -0
  89. thinkstack_core/metrics/calibration.py +125 -0
  90. thinkstack_core/metrics/credibility.py +288 -0
  91. thinkstack_core/metrics/delivery_time.py +70 -0
  92. thinkstack_core/metrics/dhs.py +126 -0
  93. thinkstack_core/metrics/mcs.py +96 -0
  94. thinkstack_core/metrics/roi.py +88 -0
  95. thinkstack_core/metrics/session_writer.py +81 -0
  96. thinkstack_core/metrics/shadow_ai.py +117 -0
  97. thinkstack_core/metrics/sprint_writer.py +243 -0
  98. thinkstack_core/observability/__init__.py +78 -0
  99. thinkstack_core/observability/datadog.py +157 -0
  100. thinkstack_core/observability/formatter.py +119 -0
  101. thinkstack_core/observability/report.py +264 -0
  102. thinkstack_core/observability/servicenow.py +147 -0
  103. thinkstack_core/observability/splunk.py +218 -0
  104. thinkstack_core/observability/webhook.py +227 -0
  105. thinkstack_core/parser/__init__.py +30 -0
  106. thinkstack_core/parser/blocks.py +216 -0
  107. thinkstack_core/parser/inference.py +159 -0
  108. thinkstack_core/parser/thinking.py +112 -0
  109. thinkstack_core/projects.py +169 -0
  110. thinkstack_core/prompt_artifact.py +76 -0
  111. thinkstack_core/proxy/__init__.py +9 -0
  112. thinkstack_core/proxy/routes/__init__.py +1 -0
  113. thinkstack_core/proxy/routes/anthropic.py +264 -0
  114. thinkstack_core/proxy/routes/azure_openai.py +336 -0
  115. thinkstack_core/proxy/routes/gemini.py +331 -0
  116. thinkstack_core/proxy/routes/groq.py +284 -0
  117. thinkstack_core/proxy/routes/ollama.py +279 -0
  118. thinkstack_core/proxy/routes/openai.py +287 -0
  119. thinkstack_core/proxy/server.py +356 -0
  120. thinkstack_core/query/__init__.py +15 -0
  121. thinkstack_core/query/grep.py +181 -0
  122. thinkstack_core/query/hybrid.py +86 -0
  123. thinkstack_core/query/semantic.py +157 -0
  124. thinkstack_core/rdp.py +105 -0
  125. thinkstack_core/reasoning/__init__.py +4 -0
  126. thinkstack_core/reasoning/entry.py +31 -0
  127. thinkstack_core/reasoning/store.py +122 -0
  128. thinkstack_core/reasoning_plus/__init__.py +70 -0
  129. thinkstack_core/reasoning_plus/augmenter.py +337 -0
  130. thinkstack_core/reasoning_plus/capture.py +51 -0
  131. thinkstack_core/reasoning_plus/config.py +313 -0
  132. thinkstack_core/reasoning_plus/context.py +262 -0
  133. thinkstack_core/reasoning_plus/learning/__init__.py +125 -0
  134. thinkstack_core/reasoning_plus/learning/analytics.py +141 -0
  135. thinkstack_core/reasoning_plus/learning/api.py +784 -0
  136. thinkstack_core/reasoning_plus/learning/chain.py +285 -0
  137. thinkstack_core/reasoning_plus/learning/composer.py +141 -0
  138. thinkstack_core/reasoning_plus/learning/conflicts.py +184 -0
  139. thinkstack_core/reasoning_plus/learning/context_collector.py +194 -0
  140. thinkstack_core/reasoning_plus/learning/cross_project.py +234 -0
  141. thinkstack_core/reasoning_plus/learning/deny_list.py +108 -0
  142. thinkstack_core/reasoning_plus/learning/embeddings.py +209 -0
  143. thinkstack_core/reasoning_plus/learning/evolution.py +119 -0
  144. thinkstack_core/reasoning_plus/learning/extractor.py +271 -0
  145. thinkstack_core/reasoning_plus/learning/filter_five_layer.py +95 -0
  146. thinkstack_core/reasoning_plus/learning/models.py +149 -0
  147. thinkstack_core/reasoning_plus/learning/org_store.py +156 -0
  148. thinkstack_core/reasoning_plus/learning/pii.py +142 -0
  149. thinkstack_core/reasoning_plus/learning/promotion.py +58 -0
  150. thinkstack_core/reasoning_plus/learning/provenance.py +126 -0
  151. thinkstack_core/reasoning_plus/learning/recorder.py +81 -0
  152. thinkstack_core/reasoning_plus/learning/relevance.py +122 -0
  153. thinkstack_core/reasoning_plus/learning/state.py +86 -0
  154. thinkstack_core/reasoning_plus/learning/store.py +178 -0
  155. thinkstack_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
  156. thinkstack_core/reasoning_plus/prompt.py +90 -0
  157. thinkstack_core/rep.py +134 -0
  158. thinkstack_core/rep_network/__init__.py +25 -0
  159. thinkstack_core/rep_network/merge.py +70 -0
  160. thinkstack_core/rep_network/node.py +137 -0
  161. thinkstack_core/rep_network/server.py +140 -0
  162. thinkstack_core/rep_network/sync.py +207 -0
  163. thinkstack_core/sensitivity.py +182 -0
  164. thinkstack_core/serve.py +258 -0
  165. thinkstack_core/session/__init__.py +39 -0
  166. thinkstack_core/session/disagreement.py +188 -0
  167. thinkstack_core/session/models.py +114 -0
  168. thinkstack_core/session/orchestrator.py +182 -0
  169. thinkstack_core/session/planner.py +169 -0
  170. thinkstack_core/session/simulator.py +132 -0
  171. thinkstack_core/signing.py +290 -0
  172. thinkstack_core/sis.py +197 -0
  173. thinkstack_core/skills/pr-reviewer/SKILL.md +204 -0
  174. thinkstack_core/skills/thinkstack-auto-sync/SKILL.md +175 -0
  175. thinkstack_core/skills/thinkstack-session-start/SKILL.md +136 -0
  176. thinkstack_core/storage.py +308 -0
  177. thinkstack_core/templates/__init__.py +6 -0
  178. thinkstack_core/templates/engine.py +122 -0
  179. thinkstack_core/templates/go.py +18 -0
  180. thinkstack_core/templates/infra.py +19 -0
  181. thinkstack_core/templates/library/__init__.py +18 -0
  182. thinkstack_core/templates/library/api_design.md +27 -0
  183. thinkstack_core/templates/library/bug_fix.md +27 -0
  184. thinkstack_core/templates/library/decision_record.md +27 -0
  185. thinkstack_core/templates/library/engine.py +228 -0
  186. thinkstack_core/templates/library/security_review.md +30 -0
  187. thinkstack_core/templates/python.py +19 -0
  188. thinkstack_core/templates/react.py +18 -0
  189. thinkstack_core/templates/typescript.py +18 -0
  190. thinkstack_core/theta.py +221 -0
  191. thinkstack_core/theta_synthesis.py +268 -0
  192. thinkstack_core/topics.py +320 -0
  193. thinkstack_core/variance.py +219 -0
  194. thinkstack_core/wrapper/__init__.py +52 -0
  195. thinkstack_core/wrapper/anthropic.py +487 -0
  196. thinkstack_core/wrapper/base.py +562 -0
  197. thinkstack_core/wrapper/bedrock.py +342 -0
  198. thinkstack_core/wrapper/gemini.py +422 -0
  199. thinkstack_core/wrapper/ollama.py +527 -0
  200. thinkstack_core/wrapper/openai.py +461 -0
  201. thinkstack_core-4.0.0.dist-info/METADATA +868 -0
  202. thinkstack_core-4.0.0.dist-info/RECORD +205 -0
  203. thinkstack_core-4.0.0.dist-info/WHEEL +5 -0
  204. thinkstack_core-4.0.0.dist-info/entry_points.txt +2 -0
  205. thinkstack_core-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,149 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.models
3
+ ===========================================
4
+ Data models for Reasoning Plus Learning (DRPL).
5
+
6
+ A ReasoningCall is a single recorded step (LLM, tool, memory, or action).
7
+ A Learning is a distilled, reusable insight derived from one or more calls.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import json
13
+ import uuid
14
+ from dataclasses import asdict, dataclass, field, fields
15
+ from datetime import datetime, timezone
16
+ from typing import Any
17
+
18
+
19
+ VALID_CALL_TYPES = ("llm", "tool", "memory", "action")
20
+ VALID_OUTCOMES = ("success", "failure", "partial", "unknown")
21
+ VALID_LEARNING_TYPES = ("insight", "correction", "pattern", "avoid", "confirm", "prescription", "canon")
22
+ VALID_VALIDITIES = ("active", "stale", "deprecated")
23
+ VALID_DISCLOSURE_LEVELS = ("PUBLIC", "PROTECTED", "PRIVATE")
24
+ VALID_SCOPES = ("branch", "project", "org")
25
+
26
+
27
+ def _now_iso() -> str:
28
+ return datetime.now(timezone.utc).isoformat()
29
+
30
+
31
+ def _make_id(prefix: str = "drpl") -> str:
32
+ return f"{prefix}_{uuid.uuid4().hex[:16]}"
33
+
34
+
35
+ @dataclass
36
+ class ReasoningCall:
37
+ """One recorded reasoning step."""
38
+
39
+ reasoning: str
40
+ call_type: str = "llm"
41
+ name: str = ""
42
+ input: str = ""
43
+ output: str = ""
44
+ outcome: str = "unknown"
45
+ state_hash: str = ""
46
+ concepts: list[str] = field(default_factory=list)
47
+ confidence: float = 1.0
48
+ sensitivity: str = "PUBLIC"
49
+ scope: str = "project"
50
+ session_id: str = ""
51
+ timestamp: str = field(default_factory=_now_iso)
52
+ id: str = field(default_factory=lambda: _make_id("call"))
53
+ meta: dict[str, Any] = field(default_factory=dict)
54
+
55
+ def __post_init__(self) -> None:
56
+ if self.call_type not in VALID_CALL_TYPES:
57
+ raise ValueError(f"Invalid call_type: {self.call_type}")
58
+ if self.outcome not in VALID_OUTCOMES:
59
+ raise ValueError(f"Invalid outcome: {self.outcome}")
60
+ if self.sensitivity not in VALID_DISCLOSURE_LEVELS:
61
+ raise ValueError(f"Invalid sensitivity: {self.sensitivity}")
62
+ if self.scope not in VALID_SCOPES:
63
+ raise ValueError(f"Invalid scope: {self.scope}")
64
+ self.confidence = max(0.0, min(1.0, float(self.confidence)))
65
+
66
+ def to_dict(self) -> dict[str, Any]:
67
+ return asdict(self)
68
+
69
+ @classmethod
70
+ def from_dict(cls, data: dict[str, Any]) -> "ReasoningCall":
71
+ valid = {f.name for f in fields(cls)}
72
+ return cls(**{k: v for k, v in data.items() if k in valid})
73
+
74
+
75
+ @dataclass
76
+ class Learning:
77
+ """A distilled, reusable learning from one or more reasoning calls."""
78
+
79
+ content: str
80
+ type: str = "insight"
81
+ trigger_concepts: list[str] = field(default_factory=list)
82
+ state_hash: str = ""
83
+ validity: str = "active"
84
+ confidence: float = 1.0
85
+ source_reasoning_ids: list[str] = field(default_factory=list)
86
+ sensitivity: str = "PUBLIC"
87
+ scope: str = "project"
88
+ supersedes: list[str] = field(default_factory=list)
89
+ superseded_by: str | None = None
90
+ evolution_chain: str | None = None
91
+ promoted_by: str | None = None
92
+ promoted_at: str | None = None
93
+ source_project_root: str | None = None
94
+ created_at: str = field(default_factory=_now_iso)
95
+ updated_at: str = field(default_factory=_now_iso)
96
+ id: str = field(default_factory=lambda: _make_id("learn"))
97
+ meta: dict[str, Any] = field(default_factory=dict)
98
+
99
+ def promote(
100
+ self,
101
+ to_scope: str = "org",
102
+ promoted_by: str | None = None,
103
+ boost_confidence: bool = True,
104
+ ) -> None:
105
+ """Promote this learning to a broader scope.
106
+
107
+ Sets scope to ``to_scope``, sets type to ``"canon"``, boosts confidence
108
+ to 1.0 (when ``boost_confidence`` is True), and records promotion
109
+ metadata.
110
+ """
111
+ if boost_confidence:
112
+ self.confidence = 1.0
113
+ self.scope = to_scope
114
+ self.type = "canon"
115
+ self.promoted_by = promoted_by
116
+ self.promoted_at = _now_iso()
117
+ self.updated_at = _now_iso()
118
+
119
+ def __post_init__(self) -> None:
120
+ if self.type not in VALID_LEARNING_TYPES:
121
+ raise ValueError(f"Invalid learning type: {self.type}")
122
+ if self.validity not in VALID_VALIDITIES:
123
+ raise ValueError(f"Invalid validity: {self.validity}")
124
+ if self.sensitivity not in VALID_DISCLOSURE_LEVELS:
125
+ raise ValueError(f"Invalid sensitivity: {self.sensitivity}")
126
+ if self.scope not in VALID_SCOPES:
127
+ raise ValueError(f"Invalid scope: {self.scope}")
128
+ self.confidence = max(0.0, min(1.0, float(self.confidence)))
129
+
130
+ def to_dict(self) -> dict[str, Any]:
131
+ return asdict(self)
132
+
133
+ @classmethod
134
+ def from_dict(cls, data: dict[str, Any]) -> "Learning":
135
+ valid = {f.name for f in fields(cls)}
136
+ return cls(**{k: v for k, v in data.items() if k in valid})
137
+
138
+ def short_form(self, max_chars: int = 200) -> str:
139
+ """Compact, prompt-ready representation."""
140
+ text = self.content.strip().replace("\n", " ")
141
+ if len(text) > max_chars:
142
+ text = text[: max_chars - 3].rstrip() + "..."
143
+ return text
144
+
145
+
146
+ def hash_state(*items: str) -> str:
147
+ """Stable hash for a set of state tokens."""
148
+ joined = "|".join(sorted(items))
149
+ return hashlib.sha256(joined.encode("utf-8")).hexdigest()[:16]
@@ -0,0 +1,156 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.org_store
3
+ =================================================
4
+ Org-level learning store aggregation.
5
+
6
+ Merges learnings from multiple governed projects with conflict-aware
7
+ aggregation: max confidence and most-restrictive validity. Used by the
8
+ five-layer filter and learning_search to produce an org-wide view of
9
+ reusable insights.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import logging
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ from thinkstack_core.reasoning_plus.learning.models import Learning
20
+
21
+ logger = logging.getLogger("thinkstack.reasoning_plus.learning")
22
+
23
+ DEFAULT_CROSS_PROJECT_THRESHOLD = 0.3
24
+
25
+
26
+ @dataclass
27
+ class OrgLearningAggregate:
28
+ """A single row in the aggregated org-learning view."""
29
+
30
+ content: str
31
+ type: str = "insight"
32
+ trigger_concepts: list[str] = field(default_factory=list)
33
+ validity: str = "active"
34
+ confidence: float = 1.0
35
+ scope: str = "org"
36
+ sensitivity: str = "PUBLIC"
37
+ source_project_roots: list[str] = field(default_factory=list)
38
+ source_learning_ids: list[str] = field(default_factory=list)
39
+ evolution_chain: str | None = None
40
+
41
+
42
+ def _merge_into_aggregate(
43
+ agg: OrgLearningAggregate,
44
+ learning: Learning,
45
+ project_root: str,
46
+ ) -> None:
47
+ """Merge one learning into an aggregate, taking the best per field."""
48
+ agg.confidence = max(agg.confidence, learning.confidence)
49
+ s_map = {"deprecated": 0, "stale": 1, "active": 2}
50
+ if s_map.get(learning.validity, 0) > s_map.get(agg.validity, 0):
51
+ agg.validity = learning.validity
52
+ agg.sensitivity = learning.sensitivity
53
+ if project_root not in agg.source_project_roots:
54
+ agg.source_project_roots.append(project_root)
55
+ if learning.id not in agg.source_learning_ids:
56
+ agg.source_learning_ids.append(learning.id)
57
+ if not agg.evolution_chain and learning.evolution_chain:
58
+ agg.evolution_chain = learning.evolution_chain
59
+
60
+
61
+ def aggregate_org_learnings(
62
+ projects_dir: Path | str,
63
+ *,
64
+ min_confidence: float = 0.3,
65
+ top_n: int = 5,
66
+ ) -> list[OrgLearningAggregate]:
67
+ """Aggregate PUBLIC-scope learnings from multiple governed projects.
68
+
69
+ Reads ``.GCC/reasoning_learnings/learnings/`` from each project directory,
70
+ groups learnings by overlapping trigger concepts, and merges them with
71
+ max-confidence / most-restrictive-validity semantics.
72
+
73
+ Returns up to ``top_n`` :class:`OrgLearningAggregate` sorted by confidence
74
+ descending.
75
+ """
76
+ out: dict[str, list[OrgLearningAggregate]] = {}
77
+
78
+ for project_root in _resolve_project_dirs(projects_dir):
79
+ try:
80
+ learnings = _load_active_learnings(project_root)
81
+ except Exception as exc:
82
+ logger.debug("thinkstack: skipping project %s — %s", project_root, exc)
83
+ continue
84
+ for learning in learnings:
85
+ concepts_key = _concept_key(learning.trigger_concepts)
86
+ if concepts_key not in out:
87
+ out[concepts_key] = []
88
+ existing = out[concepts_key]
89
+ for agg in existing:
90
+ cset = set(agg.trigger_concepts)
91
+ lset = set(learning.trigger_concepts)
92
+ overlap = cset & lset
93
+ if overlap and _content_similarity(agg.content, learning.content) > 0.6:
94
+ _merge_into_aggregate(agg, learning, str(project_root))
95
+ break
96
+ else:
97
+ agg = OrgLearningAggregate(
98
+ content=learning.content,
99
+ type=learning.type,
100
+ trigger_concepts=list(learning.trigger_concepts),
101
+ validity=learning.validity,
102
+ confidence=learning.confidence,
103
+ scope=learning.scope,
104
+ sensitivity=learning.sensitivity,
105
+ source_project_roots=[str(project_root)],
106
+ source_learning_ids=[learning.id],
107
+ evolution_chain=learning.evolution_chain,
108
+ )
109
+ out[concepts_key].append(agg)
110
+
111
+ results: list[OrgLearningAggregate] = []
112
+ for aggs in out.values():
113
+ for agg in aggs:
114
+ if agg.confidence >= min_confidence:
115
+ results.append(agg)
116
+ results.sort(key=lambda x: x.confidence, reverse=True)
117
+ return results[:top_n]
118
+
119
+
120
+ def _concept_key(concepts: list[str]) -> str:
121
+ return "|".join(sorted(c.lower() for c in concepts))
122
+
123
+
124
+ def _content_similarity(a: str, b: str) -> float:
125
+ a_words = set(a.lower().split())
126
+ b_words = set(b.lower().split())
127
+ if not a_words or not b_words:
128
+ return 0.0
129
+ intersection = a_words & b_words
130
+ return len(intersection) / max(len(a_words), len(b_words))
131
+
132
+
133
+ def _resolve_project_dirs(projects_dir: Path | str) -> list[Path]:
134
+ dirs: list[Path] = []
135
+ registry_path = Path(projects_dir)
136
+ if not registry_path.exists():
137
+ return []
138
+ try:
139
+ data = json.loads(registry_path.read_text(encoding="utf-8"))
140
+ except (json.JSONDecodeError, OSError):
141
+ return []
142
+ governed = data.get("governed_projects", [])
143
+ if isinstance(governed, list):
144
+ for entry in governed:
145
+ if isinstance(entry, dict) and "project_root" in entry:
146
+ p = Path(entry["project_root"])
147
+ if p.exists():
148
+ dirs.append(p)
149
+ return dirs
150
+
151
+
152
+ def _load_active_learnings(project_root: Path) -> list[Learning]:
153
+ from thinkstack_core.reasoning_plus.learning.store import LearningStore
154
+ gcc_dir = project_root / ".GCC"
155
+ store = LearningStore(gcc_dir)
156
+ return store.list(validity="active")
@@ -0,0 +1,142 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.pii
3
+ =========================================
4
+ PII detection, sanitization, and secret scanning for the learning pipeline.
5
+
6
+ Two layers of protection:
7
+ 1. **Pre-extraction sanitization** — strip PII from reasoning text before
8
+ it reaches the LLM extractor. Uses presidio-analyzer when available,
9
+ with a pure-regex fallback.
10
+ 2. **Post-extraction secret scanning** — detect credentials, tokens, and
11
+ API keys in extracted learning content. Learnings that match are
12
+ quarantined (not stored).
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import logging
17
+ import re
18
+ from typing import Any
19
+
20
+ from thinkstack_core.reasoning_plus.learning.models import Learning
21
+
22
+ logger = logging.getLogger("thinkstack.reasoning_plus.learning.pii")
23
+
24
+ # ---------------------------------------------------------------------------
25
+ # Pre-extraction PII sanitization
26
+ # ---------------------------------------------------------------------------
27
+
28
+
29
+ def _get_presidio_engine() -> Any:
30
+ """Return a presidio AnalyzerEngine, or None if presidio is unavailable."""
31
+ try:
32
+ from presidio_analyzer import AnalyzerEngine
33
+ return AnalyzerEngine()
34
+ except ImportError:
35
+ logger.debug("thinkstack: presidio not available, using regex PII fallback")
36
+ return None
37
+
38
+
39
+ _PRESIDIO_ENGINE: Any | None = None
40
+
41
+
42
+ def _ensure_engine() -> Any:
43
+ global _PRESIDIO_ENGINE
44
+ if _PRESIDIO_ENGINE is None:
45
+ _PRESIDIO_ENGINE = _get_presidio_engine()
46
+ return _PRESIDIO_ENGINE
47
+
48
+
49
+ # Fallback regex patterns for PII detection (used when presidio is unavailable)
50
+ _PII_REGEX_PATTERNS: list[tuple[str, str]] = [
51
+ ("CREDIT_CARD", r"\b(?:\d{4}[-\s]?){3}\d{4}\b"),
52
+ ("EMAIL", r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b"),
53
+ ("PHONE", r"\b\+?\d{1,3}[-.\s]?\(?\d{1,4}\)?[-.\s]?\d{1,4}[-.\s]?\d{1,9}\b"),
54
+ ("SSN", r"\b\d{3}-\d{2}-\d{4}\b"),
55
+ ("API_KEY_GENERIC", r"\b(?:sk-[A-Za-z0-9]{20,}|[A-Za-z0-9]{32,})\b"),
56
+ ]
57
+
58
+
59
+ def sanitize_for_extraction(text: str) -> str:
60
+ """Strip PII from *text*, replacing it with ``[REDACTED:<type>]``.
61
+
62
+ Presidio-analyzer runs first (when available) for NLP-based detection.
63
+ Regex patterns always run second as a catch-all for API keys, emails,
64
+ credit cards, phones, and SSNs. The regex pass never overwrites
65
+ presidio's output because it operates on the result of the presidio pass.
66
+ """
67
+ if not text:
68
+ return text
69
+
70
+ result = text
71
+
72
+ # Presidio — optional NLP-based detection for additional coverage.
73
+ engine = _ensure_engine()
74
+ if engine is not None:
75
+ try:
76
+ presidio_results = engine.analyze(text=result, language="en")
77
+ if presidio_results:
78
+ sanitized = list(result)
79
+ for r in sorted(presidio_results, key=lambda x: x.start, reverse=True):
80
+ entity_type = r.entity_type or "PII"
81
+ placeholder = f"[REDACTED:{entity_type}]"
82
+ sanitized[r.start : r.end] = list(placeholder)
83
+ result = "".join(sanitized)
84
+ except Exception as exc:
85
+ logger.debug("thinkstack: presidio analysis failed — %s", exc)
86
+
87
+ # Apply regex patterns — always runs, catches API keys and other patterns
88
+ # that presidio may have missed. Operates on presidio's output (or the
89
+ # original text if presidio is unavailable), so it never overwrites
90
+ # presidio redactions.
91
+ for entity_type, pattern in _PII_REGEX_PATTERNS:
92
+ try:
93
+ result = re.sub(pattern, f"[REDACTED:{entity_type}]", result)
94
+ except Exception:
95
+ continue
96
+
97
+ return result
98
+
99
+
100
+ # ---------------------------------------------------------------------------
101
+ # Post-extraction secret scanning
102
+ # ---------------------------------------------------------------------------
103
+
104
+ _SECRET_PATTERNS: list[tuple[str, str]] = [
105
+ ("OPENAI_KEY", r"sk-[A-Za-z0-9]{20,}"),
106
+ ("GITHUB_TOKEN", r"(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9_]{36,}"),
107
+ ("AWS_ACCESS_KEY", r"AKIA[0-9A-Z]{16}"),
108
+ ("BEARER_TOKEN", r"Bearer\s+[A-Za-z0-9\-._~+/]{20,}"),
109
+ ("GENERIC_SECRET", r"(?:secret|password|token|key)\s*[:=]\s*['\"]?[A-Za-z0-9_\-]{16,}"),
110
+ ]
111
+
112
+
113
+ def scan_for_secrets(text: str) -> list[str]:
114
+ """Scan *text* for credentials, tokens, and API keys.
115
+
116
+ Returns a list of secret type strings that were matched.
117
+ Empty list means no secrets detected.
118
+ """
119
+ if not text:
120
+ return []
121
+ found: list[str] = []
122
+ for name, pattern in _SECRET_PATTERNS:
123
+ try:
124
+ if re.search(pattern, text):
125
+ found.append(name)
126
+ except Exception:
127
+ continue
128
+ return found
129
+
130
+
131
+ # ---------------------------------------------------------------------------
132
+ # Sync filtering helper
133
+ # ---------------------------------------------------------------------------
134
+
135
+
136
+ def filter_learnings_for_sync(learnings: list[Learning]) -> list[Learning]:
137
+ """Filter learnings for sync: exclude PRIVATE-disclosure learnings.
138
+
139
+ Only PUBLIC and PROTECTED learnings are included in sync payloads.
140
+ PRIVATE learnings stay local.
141
+ """
142
+ return [l for l in learnings if l.sensitivity != "PRIVATE"]
@@ -0,0 +1,58 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.promotion
3
+ =================================================
4
+ Learning scope promotion workflow.
5
+
6
+ Promotion elevates a learning from the project scope to the org scope,
7
+ making it eligible for cross-project injection via the five-layer filter.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+ from pathlib import Path
13
+
14
+ from thinkstack_core.reasoning_plus.learning.models import Learning
15
+ from thinkstack_core.reasoning_plus.learning.store import LearningStore
16
+
17
+ logger = logging.getLogger("thinkstack.reasoning_plus.learning")
18
+
19
+
20
+ def promote_learning(
21
+ learning_id: str,
22
+ gcc_dir: Path | str,
23
+ *,
24
+ to_scope: str = "org",
25
+ promoted_by: str | None = None,
26
+ boost_confidence: bool = True,
27
+ ) -> Learning | None:
28
+ """Promote a learning to a broader scope and persist the change.
29
+
30
+ The learning is:
31
+ - reassigned to ``to_scope`` (default ``"org"``)
32
+ - its type becomes ``"canon"``
33
+ - its confidence is boosted to 1.0 (when ``boost_confidence``)
34
+ - promotion metadata (``promoted_by``, ``promoted_at``) is recorded
35
+ """
36
+ gcc_dir = Path(gcc_dir)
37
+ store = LearningStore(gcc_dir)
38
+ learning = store.get(learning_id)
39
+ if not learning:
40
+ logger.warning("thinkstack: promotion — learning %s not found", learning_id)
41
+ return None
42
+
43
+ if learning.sensitivity == "PRIVATE":
44
+ logger.warning(
45
+ "thinkstack: cannot promote PRIVATE learning %s — "
46
+ "PRIVATE learnings must remain project-local",
47
+ learning_id,
48
+ )
49
+ return None
50
+
51
+ learning.promote(to_scope=to_scope, promoted_by=promoted_by, boost_confidence=boost_confidence)
52
+ learning.sensitivity = "PUBLIC"
53
+ store.save(learning, deduplicate=False)
54
+ logger.info(
55
+ "thinkstack: promoted learning %s to scope=%s type=%s",
56
+ learning_id, to_scope, learning.type,
57
+ )
58
+ return learning
@@ -0,0 +1,126 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.provenance
3
+ ===============================================
4
+ Provenance tracking for injected learnings.
5
+
6
+ Records which Learning records were injected into each LLM/tool call so that
7
+ observed outcomes can be attributed back to the learnings that influenced the
8
+ call.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import hashlib
13
+ import json
14
+ import logging
15
+ from dataclasses import asdict, dataclass, field, fields
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ from thinkstack_core.reasoning_plus.learning.models import _make_id, _now_iso
20
+
21
+ logger = logging.getLogger("thinkstack.reasoning_plus.learning")
22
+
23
+
24
+ @dataclass
25
+ class ProvenanceRecord:
26
+ """Which learnings were injected into a single call, and what the outcome was."""
27
+
28
+ call_id: str
29
+ learning_ids: list[str] = field(default_factory=list)
30
+ session_id: str = ""
31
+ prompt_hash: str = ""
32
+ prompt_excerpt: str = ""
33
+ model_name: str = ""
34
+ outcome: str | None = None
35
+ injected_at: str = field(default_factory=_now_iso)
36
+ updated_at: str = field(default_factory=_now_iso)
37
+ id: str = field(default_factory=lambda: _make_id("prov"))
38
+ meta: dict[str, Any] = field(default_factory=dict)
39
+
40
+ def to_dict(self) -> dict[str, Any]:
41
+ return asdict(self)
42
+
43
+ @classmethod
44
+ def from_dict(cls, data: dict[str, Any]) -> "ProvenanceRecord":
45
+ valid = {f.name for f in fields(cls)}
46
+ return cls(**{k: v for k, v in data.items() if k in valid})
47
+
48
+
49
+ class ProvenanceStore:
50
+ """Persist ProvenanceRecord records under .GCC/reasoning_learnings/provenance/."""
51
+
52
+ def __init__(self, gcc_dir: Path | str) -> None:
53
+ self._gcc_dir = Path(gcc_dir)
54
+ self._provenance_dir = self._gcc_dir / "reasoning_learnings" / "provenance"
55
+
56
+ def save(self, record: ProvenanceRecord) -> Path:
57
+ self._provenance_dir.mkdir(parents=True, exist_ok=True)
58
+ path = self._provenance_dir / f"{record.id}.json"
59
+ _atomic_json_write(path, record.to_dict())
60
+ return path
61
+
62
+ def get(self, record_id: str) -> ProvenanceRecord | None:
63
+ path = self._provenance_dir / f"{record_id}.json"
64
+ if not path.exists():
65
+ return None
66
+ try:
67
+ data = json.loads(path.read_text(encoding="utf-8"))
68
+ return ProvenanceRecord.from_dict(data)
69
+ except Exception as exc:
70
+ logger.warning("thinkstack: failed to read provenance %s — %s", record_id, exc)
71
+ return None
72
+
73
+ def get_by_call(self, call_id: str) -> ProvenanceRecord | None:
74
+ for record in self.list():
75
+ if record.call_id == call_id:
76
+ return record
77
+ return None
78
+
79
+ def list(self, outcome: str | None = None, learning_id: str | None = None) -> list[ProvenanceRecord]:
80
+ out: list[ProvenanceRecord] = []
81
+ if not self._provenance_dir.exists():
82
+ return out
83
+ for path in sorted(self._provenance_dir.glob("*.json")):
84
+ try:
85
+ data = json.loads(path.read_text(encoding="utf-8"))
86
+ record = ProvenanceRecord.from_dict(data)
87
+ if outcome and record.outcome != outcome:
88
+ continue
89
+ if learning_id and learning_id not in record.learning_ids:
90
+ continue
91
+ out.append(record)
92
+ except Exception as exc:
93
+ logger.debug("thinkstack: skipping malformed provenance %s — %s", path, exc)
94
+ return out
95
+
96
+ def update_outcome(self, call_id: str, outcome: str) -> ProvenanceRecord | None:
97
+ record = self.get_by_call(call_id)
98
+ if not record:
99
+ return None
100
+ record.outcome = outcome
101
+ record.updated_at = _now_iso()
102
+ self.save(record)
103
+ return record
104
+
105
+ def count(self) -> int:
106
+ if not self._provenance_dir.exists():
107
+ return 0
108
+ return len(list(self._provenance_dir.glob("*.json")))
109
+
110
+
111
+ def _atomic_json_write(path: Path, data: dict[str, Any]) -> None:
112
+ tmp = path.with_suffix(path.suffix + ".tmp")
113
+ try:
114
+ with open(tmp, "w", encoding="utf-8") as f:
115
+ json.dump(data, f, indent=2)
116
+ tmp.replace(path)
117
+ finally:
118
+ if tmp.exists():
119
+ try:
120
+ tmp.unlink()
121
+ except Exception:
122
+ pass
123
+
124
+
125
+ def _hash_prompt(prompt_text: str) -> str:
126
+ return hashlib.sha256(prompt_text.encode("utf-8")).hexdigest()[:16]