thinkstack-core 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (205) hide show
  1. thinkstack_core/__init__.py +158 -0
  2. thinkstack_core/aggphi_textual.py +275 -0
  3. thinkstack_core/alerts/__init__.py +23 -0
  4. thinkstack_core/alerts/base.py +46 -0
  5. thinkstack_core/alerts/config.py +60 -0
  6. thinkstack_core/alerts/dispatcher.py +110 -0
  7. thinkstack_core/alerts/jira.py +96 -0
  8. thinkstack_core/alerts/linear.py +72 -0
  9. thinkstack_core/alerts/pagerduty.py +66 -0
  10. thinkstack_core/alerts/slack.py +81 -0
  11. thinkstack_core/alerts/teams.py +70 -0
  12. thinkstack_core/audit/__init__.py +43 -0
  13. thinkstack_core/audit/exporter.py +297 -0
  14. thinkstack_core/audit/privacy.py +101 -0
  15. thinkstack_core/audit/scrubber.py +149 -0
  16. thinkstack_core/audit/service.py +67 -0
  17. thinkstack_core/audit/signing.py +127 -0
  18. thinkstack_core/broadcast/__init__.py +4 -0
  19. thinkstack_core/broadcast/broadcaster.py +100 -0
  20. thinkstack_core/broadcast/watcher.py +71 -0
  21. thinkstack_core/capability.py +639 -0
  22. thinkstack_core/cloud/__init__.py +1 -0
  23. thinkstack_core/cloud/client_config.py +472 -0
  24. thinkstack_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
  25. thinkstack_core/cloud/client_configs/.claude-stdio.json +13 -0
  26. thinkstack_core/cloud/client_configs/.cursor-mcp.json +13 -0
  27. thinkstack_core/cloud/client_configs/.opencode-bridge.json +13 -0
  28. thinkstack_core/cloud/client_configs/.opencode.json +15 -0
  29. thinkstack_core/cloud/client_configs/.vscode-mcp.json +13 -0
  30. thinkstack_core/cloud/mcp_client.py +229 -0
  31. thinkstack_core/cloud/setup.py +144 -0
  32. thinkstack_core/cloud/sync.py +143 -0
  33. thinkstack_core/cloud/sync_bundle.py +639 -0
  34. thinkstack_core/cloud/sync_conflicts.py +183 -0
  35. thinkstack_core/cloud/sync_state.py +159 -0
  36. thinkstack_core/cloud/team_sync.py +337 -0
  37. thinkstack_core/cloud/thinkstack-mcp-bridge.js +357 -0
  38. thinkstack_core/codex/__init__.py +9 -0
  39. thinkstack_core/codex/__main__.py +97 -0
  40. thinkstack_core/codex/capture.py +208 -0
  41. thinkstack_core/codex/proxy.py +412 -0
  42. thinkstack_core/compat.py +103 -0
  43. thinkstack_core/concept_catalog.py +209 -0
  44. thinkstack_core/consolidation/__init__.py +3 -0
  45. thinkstack_core/consolidation/synthesizer.py +87 -0
  46. thinkstack_core/consolidation/workflow.py +175 -0
  47. thinkstack_core/daemon/__init__.py +27 -0
  48. thinkstack_core/daemon/supervisor.py +293 -0
  49. thinkstack_core/daemon/watcher.py +244 -0
  50. thinkstack_core/dashboard_api.py +2012 -0
  51. thinkstack_core/deltaf.py +97 -0
  52. thinkstack_core/disclosure.py +50 -0
  53. thinkstack_core/divergence/__init__.py +3 -0
  54. thinkstack_core/divergence/detector.py +166 -0
  55. thinkstack_core/gateway/__init__.py +32 -0
  56. thinkstack_core/gateway/key_manager.py +124 -0
  57. thinkstack_core/gateway/metrics_webhook.py +252 -0
  58. thinkstack_core/gateway/policy.py +262 -0
  59. thinkstack_core/gateway/server.py +727 -0
  60. thinkstack_core/gateway/sso.py +233 -0
  61. thinkstack_core/gcc.py +1246 -0
  62. thinkstack_core/github/__init__.py +35 -0
  63. thinkstack_core/github/app.py +240 -0
  64. thinkstack_core/github/comment_builder.py +113 -0
  65. thinkstack_core/github/pat.py +76 -0
  66. thinkstack_core/github/pr_parser.py +82 -0
  67. thinkstack_core/github/pr_reporter.py +555 -0
  68. thinkstack_core/gitlab/__init__.py +177 -0
  69. thinkstack_core/hitl/__init__.py +4 -0
  70. thinkstack_core/hitl/channels.py +129 -0
  71. thinkstack_core/hitl/orchestrator.py +95 -0
  72. thinkstack_core/hooks/__init__.py +17 -0
  73. thinkstack_core/hooks/claude_code.py +228 -0
  74. thinkstack_core/hooks/git_capture.py +341 -0
  75. thinkstack_core/hooks/git_commit.py +182 -0
  76. thinkstack_core/hooks/installer.py +850 -0
  77. thinkstack_core/hooks/pre_commit.py +157 -0
  78. thinkstack_core/hooks/runner.py +386 -0
  79. thinkstack_core/identity/__init__.py +4 -0
  80. thinkstack_core/identity/agent.py +86 -0
  81. thinkstack_core/identity/providers.py +85 -0
  82. thinkstack_core/invariants.py +182 -0
  83. thinkstack_core/mcp/__init__.py +10 -0
  84. thinkstack_core/mcp/auth.py +177 -0
  85. thinkstack_core/mcp/server.py +1215 -0
  86. thinkstack_core/metrics/__init__.py +35 -0
  87. thinkstack_core/metrics/aggregate.py +215 -0
  88. thinkstack_core/metrics/calibrate.py +198 -0
  89. thinkstack_core/metrics/calibration.py +125 -0
  90. thinkstack_core/metrics/credibility.py +288 -0
  91. thinkstack_core/metrics/delivery_time.py +70 -0
  92. thinkstack_core/metrics/dhs.py +126 -0
  93. thinkstack_core/metrics/mcs.py +96 -0
  94. thinkstack_core/metrics/roi.py +88 -0
  95. thinkstack_core/metrics/session_writer.py +81 -0
  96. thinkstack_core/metrics/shadow_ai.py +117 -0
  97. thinkstack_core/metrics/sprint_writer.py +243 -0
  98. thinkstack_core/observability/__init__.py +78 -0
  99. thinkstack_core/observability/datadog.py +157 -0
  100. thinkstack_core/observability/formatter.py +119 -0
  101. thinkstack_core/observability/report.py +264 -0
  102. thinkstack_core/observability/servicenow.py +147 -0
  103. thinkstack_core/observability/splunk.py +218 -0
  104. thinkstack_core/observability/webhook.py +227 -0
  105. thinkstack_core/parser/__init__.py +30 -0
  106. thinkstack_core/parser/blocks.py +216 -0
  107. thinkstack_core/parser/inference.py +159 -0
  108. thinkstack_core/parser/thinking.py +112 -0
  109. thinkstack_core/projects.py +169 -0
  110. thinkstack_core/prompt_artifact.py +76 -0
  111. thinkstack_core/proxy/__init__.py +9 -0
  112. thinkstack_core/proxy/routes/__init__.py +1 -0
  113. thinkstack_core/proxy/routes/anthropic.py +264 -0
  114. thinkstack_core/proxy/routes/azure_openai.py +336 -0
  115. thinkstack_core/proxy/routes/gemini.py +331 -0
  116. thinkstack_core/proxy/routes/groq.py +284 -0
  117. thinkstack_core/proxy/routes/ollama.py +279 -0
  118. thinkstack_core/proxy/routes/openai.py +287 -0
  119. thinkstack_core/proxy/server.py +356 -0
  120. thinkstack_core/query/__init__.py +15 -0
  121. thinkstack_core/query/grep.py +181 -0
  122. thinkstack_core/query/hybrid.py +86 -0
  123. thinkstack_core/query/semantic.py +157 -0
  124. thinkstack_core/rdp.py +105 -0
  125. thinkstack_core/reasoning/__init__.py +4 -0
  126. thinkstack_core/reasoning/entry.py +31 -0
  127. thinkstack_core/reasoning/store.py +122 -0
  128. thinkstack_core/reasoning_plus/__init__.py +70 -0
  129. thinkstack_core/reasoning_plus/augmenter.py +337 -0
  130. thinkstack_core/reasoning_plus/capture.py +51 -0
  131. thinkstack_core/reasoning_plus/config.py +313 -0
  132. thinkstack_core/reasoning_plus/context.py +262 -0
  133. thinkstack_core/reasoning_plus/learning/__init__.py +125 -0
  134. thinkstack_core/reasoning_plus/learning/analytics.py +141 -0
  135. thinkstack_core/reasoning_plus/learning/api.py +784 -0
  136. thinkstack_core/reasoning_plus/learning/chain.py +285 -0
  137. thinkstack_core/reasoning_plus/learning/composer.py +141 -0
  138. thinkstack_core/reasoning_plus/learning/conflicts.py +184 -0
  139. thinkstack_core/reasoning_plus/learning/context_collector.py +194 -0
  140. thinkstack_core/reasoning_plus/learning/cross_project.py +234 -0
  141. thinkstack_core/reasoning_plus/learning/deny_list.py +108 -0
  142. thinkstack_core/reasoning_plus/learning/embeddings.py +209 -0
  143. thinkstack_core/reasoning_plus/learning/evolution.py +119 -0
  144. thinkstack_core/reasoning_plus/learning/extractor.py +271 -0
  145. thinkstack_core/reasoning_plus/learning/filter_five_layer.py +95 -0
  146. thinkstack_core/reasoning_plus/learning/models.py +149 -0
  147. thinkstack_core/reasoning_plus/learning/org_store.py +156 -0
  148. thinkstack_core/reasoning_plus/learning/pii.py +142 -0
  149. thinkstack_core/reasoning_plus/learning/promotion.py +58 -0
  150. thinkstack_core/reasoning_plus/learning/provenance.py +126 -0
  151. thinkstack_core/reasoning_plus/learning/recorder.py +81 -0
  152. thinkstack_core/reasoning_plus/learning/relevance.py +122 -0
  153. thinkstack_core/reasoning_plus/learning/state.py +86 -0
  154. thinkstack_core/reasoning_plus/learning/store.py +178 -0
  155. thinkstack_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
  156. thinkstack_core/reasoning_plus/prompt.py +90 -0
  157. thinkstack_core/rep.py +134 -0
  158. thinkstack_core/rep_network/__init__.py +25 -0
  159. thinkstack_core/rep_network/merge.py +70 -0
  160. thinkstack_core/rep_network/node.py +137 -0
  161. thinkstack_core/rep_network/server.py +140 -0
  162. thinkstack_core/rep_network/sync.py +207 -0
  163. thinkstack_core/sensitivity.py +182 -0
  164. thinkstack_core/serve.py +258 -0
  165. thinkstack_core/session/__init__.py +39 -0
  166. thinkstack_core/session/disagreement.py +188 -0
  167. thinkstack_core/session/models.py +114 -0
  168. thinkstack_core/session/orchestrator.py +182 -0
  169. thinkstack_core/session/planner.py +169 -0
  170. thinkstack_core/session/simulator.py +132 -0
  171. thinkstack_core/signing.py +290 -0
  172. thinkstack_core/sis.py +197 -0
  173. thinkstack_core/skills/pr-reviewer/SKILL.md +204 -0
  174. thinkstack_core/skills/thinkstack-auto-sync/SKILL.md +175 -0
  175. thinkstack_core/skills/thinkstack-session-start/SKILL.md +136 -0
  176. thinkstack_core/storage.py +308 -0
  177. thinkstack_core/templates/__init__.py +6 -0
  178. thinkstack_core/templates/engine.py +122 -0
  179. thinkstack_core/templates/go.py +18 -0
  180. thinkstack_core/templates/infra.py +19 -0
  181. thinkstack_core/templates/library/__init__.py +18 -0
  182. thinkstack_core/templates/library/api_design.md +27 -0
  183. thinkstack_core/templates/library/bug_fix.md +27 -0
  184. thinkstack_core/templates/library/decision_record.md +27 -0
  185. thinkstack_core/templates/library/engine.py +228 -0
  186. thinkstack_core/templates/library/security_review.md +30 -0
  187. thinkstack_core/templates/python.py +19 -0
  188. thinkstack_core/templates/react.py +18 -0
  189. thinkstack_core/templates/typescript.py +18 -0
  190. thinkstack_core/theta.py +221 -0
  191. thinkstack_core/theta_synthesis.py +268 -0
  192. thinkstack_core/topics.py +320 -0
  193. thinkstack_core/variance.py +219 -0
  194. thinkstack_core/wrapper/__init__.py +52 -0
  195. thinkstack_core/wrapper/anthropic.py +487 -0
  196. thinkstack_core/wrapper/base.py +562 -0
  197. thinkstack_core/wrapper/bedrock.py +342 -0
  198. thinkstack_core/wrapper/gemini.py +422 -0
  199. thinkstack_core/wrapper/ollama.py +527 -0
  200. thinkstack_core/wrapper/openai.py +461 -0
  201. thinkstack_core-4.0.0.dist-info/METADATA +868 -0
  202. thinkstack_core-4.0.0.dist-info/RECORD +205 -0
  203. thinkstack_core-4.0.0.dist-info/WHEEL +5 -0
  204. thinkstack_core-4.0.0.dist-info/entry_points.txt +2 -0
  205. thinkstack_core-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,784 @@
1
+ """
2
+ thinkstack_core.reasoning_plus.learning.api
3
+ =========================================
4
+ High-level facade for Reasoning Plus Learning (DRPL).
5
+
6
+ This is the main entry point for callers. It orchestrates:
7
+ - recording calls
8
+ - extracting learnings
9
+ - retrieving relevant learnings
10
+ - composing prompt blocks
11
+ - applying outcome feedback
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ from pathlib import Path
17
+ from typing import Any, Callable
18
+
19
+ import logging
20
+
21
+ from thinkstack_core.reasoning_plus.config import ReasoningPlusConfig
22
+ from thinkstack_core.reasoning_plus.learning.composer import PromptComposer
23
+ from thinkstack_core.reasoning_plus.learning.embeddings import (
24
+ EmbeddingBackend,
25
+ KeywordEmbeddingBackend,
26
+ get_embedding_backend,
27
+ )
28
+ from thinkstack_core.reasoning_plus.learning.context_collector import (
29
+ ExtractionContext,
30
+ collect_extraction_context,
31
+ )
32
+ from thinkstack_core.reasoning_plus.learning.conflicts import (
33
+ Conflict,
34
+ ConflictDetector,
35
+ ConflictResolution,
36
+ resolve_conflict,
37
+ )
38
+ from thinkstack_core.reasoning_plus.learning.evolution import (
39
+ apply_confidence_decay,
40
+ goal_changed_since,
41
+ )
42
+ from thinkstack_core.reasoning_plus.learning.extractor import (
43
+ LearningExtractor,
44
+ make_extractor,
45
+ )
46
+ from thinkstack_core.reasoning_plus.learning.pii import (
47
+ filter_learnings_for_sync,
48
+ sanitize_for_extraction,
49
+ scan_for_secrets,
50
+ )
51
+ from thinkstack_core.reasoning_plus.learning.models import Learning, ReasoningCall, _make_id
52
+ from thinkstack_core.reasoning_plus.learning.provenance import ProvenanceRecord, ProvenanceStore, _hash_prompt
53
+ from thinkstack_core.reasoning_plus.learning.recorder import CallRecorder
54
+ from thinkstack_core.reasoning_plus.learning.relevance import RelevanceEngine
55
+ from thinkstack_core.reasoning_plus.learning.state import workspace_state_hash
56
+ from thinkstack_core.reasoning_plus.learning.store import LearningStore, _atomic_json_write
57
+ from thinkstack_core.reasoning_plus.learning import cross_project
58
+
59
+ logger = logging.getLogger("thinkstack.reasoning_plus.learning")
60
+
61
+
62
+ LLMClient = Callable[[str], str]
63
+
64
+
65
+ class ReasoningPlusLearning:
66
+ """
67
+ Facade for learning from reasoning and feeding it back into future calls.
68
+
69
+ Usage:
70
+ drpl = ReasoningPlusLearning(gcc_dir, repo_path)
71
+ call_id = drpl.record_call(
72
+ call_type="tool",
73
+ name="read_file",
74
+ reasoning="Reading utils.py because the error mentions parse_date.",
75
+ outcome="failure",
76
+ input="utils.py",
77
+ output="...",
78
+ state_hash=drpl.workspace_hash(),
79
+ session_id="session-1",
80
+ )
81
+ drpl.extract_learnings(call_id)
82
+ learnings = drpl.get_relevant_learnings("Fix the parse_date bug")
83
+ prompt = drpl.compose_user_prompt("Fix the parse_date bug", learnings)
84
+ """
85
+
86
+ def __init__(
87
+ self,
88
+ gcc_dir: Path | str,
89
+ repo_path: Path | str | None = None,
90
+ config: ReasoningPlusConfig | None = None,
91
+ llm_client: Callable[[str], str] | None = None,
92
+ ) -> None:
93
+ self._gcc_dir = Path(gcc_dir)
94
+ self._repo_path = Path(repo_path) if repo_path else self._gcc_dir.parent
95
+ self._config = config or ReasoningPlusConfig()
96
+ self._llm_client = llm_client
97
+
98
+ self._recorder = CallRecorder(self._gcc_dir)
99
+ self._store = LearningStore(self._gcc_dir)
100
+ self._provenance = ProvenanceStore(self._gcc_dir)
101
+ self._extractor = self._make_extractor()
102
+ self._relevance = RelevanceEngine(
103
+ backend=self._make_backend(),
104
+ cache_ttl=self._config.learning_relevance_cache_ttl,
105
+ )
106
+ self._composer = PromptComposer(max_lines=self._config.learning_max_lines)
107
+
108
+ # Batch extraction queue: call_id -> ExtractionContext
109
+ self._pending_extractions: dict[str, ExtractionContext] = {}
110
+ self._batch_threshold: int = getattr(self._config, "learning_batch_threshold", 5)
111
+
112
+ # Conflict detector reuses the embedding backend for similarity scoring.
113
+ self._conflict_detector = ConflictDetector(backend=self._relevance._backend)
114
+
115
+ def _make_extractor(self) -> LearningExtractor:
116
+ """Build the configured learning extractor, falling back to rule-based on errors."""
117
+ try:
118
+ return make_extractor(
119
+ strategy=self._config.learning_extraction_strategy,
120
+ llm_client=self._llm_client,
121
+ )
122
+ except Exception as exc:
123
+ logger.warning(
124
+ "thinkstack: failed to load learning extractor strategy '%s' — %s; falling back to rule-based",
125
+ self._config.learning_extraction_strategy,
126
+ exc,
127
+ )
128
+ return LearningExtractor()
129
+
130
+ def _make_backend(self) -> EmbeddingBackend:
131
+ """Build the configured embedding backend, falling back to keyword on errors."""
132
+ try:
133
+ return get_embedding_backend(
134
+ self._config.learning_embedding_backend,
135
+ self._config.learning_embedding_model,
136
+ max_keywords=self._config.max_keywords,
137
+ )
138
+ except Exception as exc:
139
+ logger.warning(
140
+ "thinkstack: failed to load embedding backend '%s' — %s; falling back to keyword",
141
+ self._config.learning_embedding_backend,
142
+ exc,
143
+ )
144
+ return KeywordEmbeddingBackend(max_keywords=self._config.max_keywords)
145
+
146
+ # ------------------------------------------------------------------
147
+ # Recording
148
+ # ------------------------------------------------------------------
149
+
150
+ def record_call(self, *, call_type: str, reasoning: str, session_id: str = "", name: str = "", input: str = "", output: str = "", outcome: str = "unknown", state_hash: str = "", concepts: list[str] | None = None, confidence: float = 1.0, sensitivity: str = "PUBLIC", scope: str = "project", meta: dict[str, Any] | None = None, call_id: str | None = None) -> str:
151
+ """Record a single call and return its ID."""
152
+ call = ReasoningCall(
153
+ id=call_id or _make_id("call"),
154
+ call_type=call_type,
155
+ name=name,
156
+ reasoning=reasoning,
157
+ input=input,
158
+ output=output,
159
+ outcome=outcome,
160
+ state_hash=state_hash or self.workspace_hash(),
161
+ concepts=concepts or [],
162
+ confidence=confidence,
163
+ sensitivity=sensitivity,
164
+ scope=scope,
165
+ session_id=session_id,
166
+ meta=meta or {},
167
+ )
168
+ self._recorder.record(call)
169
+ return call.id
170
+
171
+ def record_and_extract(self, *, call_type: str, reasoning: str, session_id: str = "", outcome: str = "unknown", **kwargs: Any) -> tuple[str, list[str]]:
172
+ """Record a call and immediately extract learnings from it."""
173
+ call_id = self.record_call(
174
+ call_type=call_type,
175
+ reasoning=reasoning,
176
+ session_id=session_id,
177
+ outcome=outcome,
178
+ **kwargs,
179
+ )
180
+ learning_ids = self.extract_learnings(call_id)
181
+ return call_id, learning_ids
182
+
183
+ # ------------------------------------------------------------------
184
+ # Extraction
185
+ # ------------------------------------------------------------------
186
+
187
+ def _apply_disclosure_gate(self, call: ReasoningCall) -> bool:
188
+ """Return True if the call passes the configured disclosure policy gate."""
189
+ policy = getattr(self._config, "disclosure_policy", "standard")
190
+ if policy == "strict" and call.sensitivity != "PUBLIC":
191
+ return False
192
+ if policy == "standard" and call.sensitivity == "PRIVATE":
193
+ return False
194
+ return True
195
+
196
+ def _extract_and_scan(
197
+ self,
198
+ call: ReasoningCall,
199
+ context: ExtractionContext,
200
+ ) -> list[str]:
201
+ """Extract learnings from a call, sanitize PII, scan for secrets, store.
202
+
203
+ Shared by both synchronous (extract_learnings) and batch (extract_batch)
204
+ paths so PII guardrails are never bypassed.
205
+ """
206
+ if not self._apply_disclosure_gate(call):
207
+ return []
208
+
209
+ call = ReasoningCall.from_dict(call.to_dict())
210
+ call.reasoning = sanitize_for_extraction(call.reasoning)
211
+
212
+ learnings = self._extractor.extract(call, context=context)
213
+
214
+ ids: list[str] = []
215
+ for learning in learnings:
216
+ secrets = scan_for_secrets(learning.content)
217
+ if secrets:
218
+ logger.warning(
219
+ "thinkstack: quarantined learning %s — matched secret types: %s",
220
+ learning.id, ",".join(secrets),
221
+ )
222
+ self._log_quarantine(learning, secrets)
223
+ continue
224
+ self._store.save(learning)
225
+ ids.append(learning.id)
226
+ return ids
227
+
228
+ def extract_learnings(self, call_id: str) -> list[str]:
229
+ """Extract and store learnings from a recorded call immediately.
230
+
231
+ Applies disclosure policy gating, PII sanitization on reasoning text,
232
+ and secret scanning on extracted learnings. For batch extraction with
233
+ LLM gating, use ``queue_extraction()`` instead.
234
+ """
235
+ call = self._recorder.get(call_id)
236
+ if not call:
237
+ return []
238
+
239
+ if not self._apply_disclosure_gate(call):
240
+ return []
241
+
242
+ context = collect_extraction_context(call, gcc_dir=self._gcc_dir)
243
+ return self._extract_and_scan(call, context)
244
+
245
+ def queue_extraction(self, call_id: str) -> list[str]:
246
+ """Queue a call for batch extraction.
247
+
248
+ When the queue reaches ``learning_batch_threshold``, a batch extraction
249
+ is triggered automatically. Returns learning IDs if batch was triggered,
250
+ or empty list if still queued.
251
+ """
252
+ call = self._recorder.get(call_id)
253
+ if not call:
254
+ return []
255
+
256
+ # Disclosure policy gating (S20).
257
+ policy = getattr(self._config, "disclosure_policy", "standard")
258
+ if policy == "strict" and call.sensitivity != "PUBLIC":
259
+ return []
260
+ if policy == "standard" and call.sensitivity == "PRIVATE":
261
+ return []
262
+
263
+ context = collect_extraction_context(call, gcc_dir=self._gcc_dir)
264
+ if not context.is_significant():
265
+ return []
266
+
267
+ self._pending_extractions[call_id] = context
268
+
269
+ if len(self._pending_extractions) >= self._batch_threshold:
270
+ return self.extract_batch()
271
+
272
+ return []
273
+
274
+ def extract_batch(self) -> list[str]:
275
+ """Extract all queued calls in a single batch.
276
+
277
+ Reduces per-call LLM extraction cost by combining multiple calls into
278
+ one extraction pass. PII guardrails are applied via the shared
279
+ ``_extract_and_scan`` helper. Can be called explicitly at session end.
280
+ """
281
+ if not self._pending_extractions:
282
+ return []
283
+
284
+ call_ids = list(self._pending_extractions.keys())
285
+ contexts = list(self._pending_extractions.values())
286
+ self._pending_extractions.clear()
287
+
288
+ learning_ids: list[str] = []
289
+ for call_id, context in zip(call_ids, contexts):
290
+ call = self._recorder.get(call_id)
291
+ if not call:
292
+ continue
293
+ try:
294
+ learning_ids.extend(self._extract_and_scan(call, context))
295
+ except Exception as exc:
296
+ logger.warning("thinkstack: batch extraction failed for call %s — %s", call_id, exc)
297
+
298
+ return learning_ids
299
+
300
+ def flush_extractions(self) -> list[str]:
301
+ """Flush any pending extractions now (called at session end)."""
302
+ if not self._pending_extractions:
303
+ return []
304
+ return self.extract_batch()
305
+
306
+ @property
307
+ def pending_extraction_count(self) -> int:
308
+ """Number of calls queued for batch extraction."""
309
+ return len(self._pending_extractions)
310
+
311
+ # ------------------------------------------------------------------
312
+ # Retrieval
313
+ # ------------------------------------------------------------------
314
+
315
+ def get_relevant_learnings(
316
+ self,
317
+ context: str,
318
+ top_n: int | None = None,
319
+ state_hash: str | None = None,
320
+ ) -> list[Learning]:
321
+ """Return the most relevant active learnings for the given context."""
322
+ if not self._config.learning_enabled:
323
+ return []
324
+ top_n = top_n or self._config.learning_top_n
325
+ all_active = self._store.list(validity="active")
326
+ return self._relevance.rank(
327
+ context,
328
+ all_active,
329
+ current_state_hash=state_hash or self.workspace_hash(),
330
+ top_n=top_n,
331
+ )
332
+
333
+ def cross_project_suggest(self, query: str, top_n: int = 3) -> list[Learning]:
334
+ """Return relevant learnings from sibling governed projects.
335
+
336
+ Looks up the machine-level registry of governed projects, finds projects
337
+ whose name or tags overlap with the query, loads their active learnings,
338
+ and returns the top-N most relevant ones.
339
+ """
340
+ if not self._config.learning_enabled:
341
+ return []
342
+ return cross_project.cross_project_suggest(
343
+ query=query,
344
+ current_project_root=self._repo_path,
345
+ top_n=top_n,
346
+ )
347
+
348
+ # ------------------------------------------------------------------
349
+ # Cross-project & org learning (S21)
350
+ # ------------------------------------------------------------------
351
+
352
+ def get_cross_project_learnings(
353
+ self,
354
+ context: str,
355
+ top_n: int | None = None,
356
+ apply_filter: bool = True,
357
+ ) -> list[Learning]:
358
+ """Return org-scope learnings from sibling projects, five-layer filtered."""
359
+ if not self._config.learning_cross_project_enabled:
360
+ return []
361
+
362
+ top_n = top_n if top_n is not None else self._config.learning_cross_project_top_n
363
+ candidates = cross_project.cross_project_suggest(
364
+ query=context,
365
+ current_project_root=self._repo_path,
366
+ top_n=top_n * 2,
367
+ )
368
+ if not apply_filter:
369
+ return candidates[:top_n]
370
+
371
+ from thinkstack_core.reasoning_plus.learning.filter_five_layer import five_layer_filter
372
+ from thinkstack_core.reasoning_plus.learning.deny_list import load_deny_list
373
+
374
+ deny_list = load_deny_list(self._repo_path)
375
+ from thinkstack_core.reasoning_plus.context import extract_keywords
376
+ ctx_emb = self._relevance._backend.encode(context)
377
+ ctx_kw = set(extract_keywords(context, max_keywords=30))
378
+ state = self.workspace_hash()
379
+ filtered: list[Learning] = []
380
+ for learning in candidates:
381
+ relevance = self._relevance._score(learning, ctx_emb, state, ctx_kw)
382
+ allowed, _reason = five_layer_filter(
383
+ learning,
384
+ target_project_root=self._repo_path,
385
+ prompt_context=context,
386
+ relevance_score=relevance,
387
+ deny_list=deny_list,
388
+ min_relevance=self._config.learning_relevance_threshold,
389
+ )
390
+ if allowed:
391
+ filtered.append(learning)
392
+ if len(filtered) >= top_n:
393
+ break
394
+ return filtered
395
+
396
+ def get_combined_learnings(
397
+ self,
398
+ context: str,
399
+ top_n: int | None = None,
400
+ state_hash: str | None = None,
401
+ ) -> list[Learning]:
402
+ """Return project-local + org learning recommendations, co-ranked.
403
+
404
+ Org learnings are pulled first, then merged with local learnings,
405
+ and the combined set is ranked by the RelevanceEngine.
406
+ """
407
+ local = self.get_relevant_learnings(
408
+ context=context,
409
+ top_n=top_n,
410
+ state_hash=state_hash,
411
+ )
412
+ if not self._config.learning_cross_project_enabled:
413
+ return local
414
+
415
+ cross = self.get_cross_project_learnings(
416
+ context=context,
417
+ top_n=min(self._config.learning_max_org_learnings_per_call, top_n or 5),
418
+ )
419
+ if not cross:
420
+ return local
421
+
422
+ existing_ids = {l.id for l in local}
423
+ combined = list(local)
424
+ for learning in cross:
425
+ if learning.id not in existing_ids:
426
+ combined.append(learning)
427
+
428
+ top = top_n or self._config.learning_top_n
429
+ return self._relevance.rank(
430
+ context, combined, current_state_hash=state_hash or self.workspace_hash(), top_n=top,
431
+ )
432
+
433
+ def promote_learning(
434
+ self,
435
+ learning_id: str,
436
+ to_scope: str = "org",
437
+ promoted_by: str | None = None,
438
+ ) -> Learning | None:
439
+ """Promote a learning to org scope, making it available cross-project."""
440
+ from thinkstack_core.reasoning_plus.learning.promotion import \
441
+ promote_learning as _promote
442
+ return _promote(
443
+ learning_id=learning_id,
444
+ gcc_dir=self._gcc_dir,
445
+ to_scope=to_scope,
446
+ promoted_by=promoted_by or "cli",
447
+ )
448
+
449
+ def learning_search(
450
+ self,
451
+ query: str,
452
+ search_type: str | None = None,
453
+ scope: str | None = None,
454
+ top_n: int = 10,
455
+ ) -> list[Learning]:
456
+ """Search learnings by concept, type, scope, or full-text.
457
+
458
+ Searches both the project-local store and the org aggregate.
459
+ """
460
+ results: list[Learning] = []
461
+ all_local = self._store.list(validity="active")
462
+ for learning in all_local:
463
+ if self._match_learning(learning, query, search_type, scope):
464
+ results.append(learning)
465
+
466
+ if self._config.learning_cross_project_enabled:
467
+ import os
468
+ from thinkstack_core.reasoning_plus.learning.cross_project import (
469
+ DEFAULT_REGISTRY_DIR,
470
+ DEFAULT_REGISTRY_FILE,
471
+ REGISTRY_PATH_ENV,
472
+ )
473
+ from thinkstack_core.reasoning_plus.learning.org_store import \
474
+ aggregate_org_learnings
475
+ env_path = os.environ.get(REGISTRY_PATH_ENV, "").strip()
476
+ if env_path:
477
+ registry_path = Path(env_path)
478
+ else:
479
+ registry_path = DEFAULT_REGISTRY_DIR / DEFAULT_REGISTRY_FILE
480
+ orgs = aggregate_org_learnings(registry_path)
481
+ for org in orgs:
482
+ dummy = Learning(
483
+ content=org.content,
484
+ type=org.type,
485
+ trigger_concepts=org.trigger_concepts,
486
+ validity=org.validity,
487
+ confidence=org.confidence,
488
+ scope=org.scope,
489
+ sensitivity=org.sensitivity,
490
+ )
491
+ if self._match_learning(dummy, query, search_type, scope):
492
+ results.append(dummy)
493
+
494
+ return sorted(results, key=lambda l: l.confidence, reverse=True)[:top_n]
495
+
496
+ @staticmethod
497
+ def _match_learning(
498
+ learning: Learning,
499
+ query: str,
500
+ search_type: str | None,
501
+ scope: str | None,
502
+ ) -> bool:
503
+ q = query.lower()
504
+ if search_type and learning.type != search_type:
505
+ return False
506
+ if scope and learning.scope != scope:
507
+ return False
508
+ if q in learning.content.lower():
509
+ return True
510
+ for concept in learning.trigger_concepts:
511
+ if q in concept.lower():
512
+ return True
513
+ return False
514
+
515
+ # ------------------------------------------------------------------
516
+ # Composition
517
+ # ------------------------------------------------------------------
518
+
519
+ def compose_user_prompt(self, user_prompt: str, learnings: list[Learning]) -> str:
520
+ """Append a compact <learnings> block to the user prompt.
521
+
522
+ The block is capped at ``learning_token_budget`` tokens and filtered
523
+ by ``learning_relevance_threshold`` confidence, both from config.
524
+ """
525
+ block = self._composer.compose(
526
+ learnings,
527
+ token_budget=self._config.learning_token_budget,
528
+ relevance_threshold=self._config.learning_relevance_threshold,
529
+ )
530
+ if not block:
531
+ return user_prompt
532
+ return f"{user_prompt}\n\n{block}"
533
+
534
+ def compose_system_prompt(self, system_prompt: str, learnings: list[Learning]) -> str:
535
+ """Append a learning appendix to the system prompt (fallback)."""
536
+ block = self._composer.compose_system_prompt(learnings)
537
+ if not block:
538
+ return system_prompt
539
+ return f"{system_prompt}\n\n{block}"
540
+
541
+ # ------------------------------------------------------------------
542
+ # Feedback
543
+ # ------------------------------------------------------------------
544
+
545
+ def apply_feedback(self, learning_id: str, outcome: str) -> Learning | None:
546
+ """Adjust confidence or validity based on observed outcome."""
547
+ learning = self._store.get(learning_id)
548
+ if not learning:
549
+ return None
550
+
551
+ if outcome == "success":
552
+ learning.confidence = min(1.0, learning.confidence + 0.1)
553
+ elif outcome == "failure":
554
+ learning.confidence = max(0.0, learning.confidence - 0.2)
555
+ if learning.confidence < 0.3:
556
+ learning.validity = "deprecated"
557
+ elif outcome == "stale":
558
+ learning.validity = "stale"
559
+
560
+ learning.updated_at = _now_iso()
561
+ self._store.save(learning)
562
+ return learning
563
+
564
+ # ------------------------------------------------------------------
565
+ # Conflict resolution & evolution
566
+ # ------------------------------------------------------------------
567
+
568
+ def detect_and_resolve_conflicts(
569
+ self,
570
+ incoming: Learning,
571
+ existing: list[Learning] | None = None,
572
+ ) -> list[ConflictResolution]:
573
+ """Detect conflicts between ``incoming`` and existing learnings, then resolve them.
574
+
575
+ For each conflict the resolution matrix decides:
576
+ - ``NEWER_WINS``: older learning deprecated (``superseded_by`` set),
577
+ newer learning records it in ``supersedes``.
578
+ - ``OLDER_WINS``: newer learning marked stale; its content appended
579
+ as a note to the older learning.
580
+ - ``FLAG_FOR_HUMAN``: conflict persisted to
581
+ ``.GCC/reasoning_learnings/conflicts/`` for manual resolution.
582
+
583
+ Returns the list of :class:`ConflictResolution` objects.
584
+ """
585
+ if existing is None:
586
+ existing = self._store.list(validity="active")
587
+ # Exclude the incoming learning itself from the candidate set.
588
+ existing = [l for l in existing if l.id != incoming.id]
589
+
590
+ goal_changed = goal_changed_since(incoming.id, self._gcc_dir)
591
+ conflicts = self._conflict_detector.detect_conflicts(incoming, existing)
592
+ resolutions: list[ConflictResolution] = []
593
+
594
+ for conflict in conflicts:
595
+ conflict.goal_changed = goal_changed
596
+ res = resolve_conflict(conflict)
597
+ self._apply_resolution(res)
598
+ if not res.auto_resolved:
599
+ self._persist_conflict(conflict, res)
600
+ resolutions.append(res)
601
+
602
+ return resolutions
603
+
604
+ def _apply_resolution(self, res: ConflictResolution) -> None:
605
+ """Mutate the losing/winning learnings on disk per the resolution.
606
+
607
+ All in-memory mutations are applied before any disk write so that each
608
+ learning is saved exactly once with its final state.
609
+ """
610
+ if res.action == "NEWER_WINS":
611
+ older = self._store.get(res.loser)
612
+ newer = self._store.get(res.winner)
613
+ # Apply all mutations in memory first.
614
+ if older and newer:
615
+ if older.id not in newer.supersedes:
616
+ newer.supersedes = list(newer.supersedes) + [older.id]
617
+ if not newer.evolution_chain:
618
+ newer.evolution_chain = f"chain-{newer.id[:12]}"
619
+ older.evolution_chain = newer.evolution_chain
620
+ if older:
621
+ older.validity = "deprecated"
622
+ older.superseded_by = res.winner
623
+ older.updated_at = _now_iso()
624
+ self._store.save(older, deduplicate=False)
625
+ if newer and older:
626
+ newer.updated_at = _now_iso()
627
+ self._store.save(newer, deduplicate=False)
628
+ elif res.action == "OLDER_WINS":
629
+ newer = self._store.get(res.loser)
630
+ older = self._store.get(res.winner)
631
+ # Apply all mutations in memory first.
632
+ if older and newer:
633
+ note = newer.short_form(max_chars=160)
634
+ existing_notes = older.meta.get("notes", "")
635
+ if note not in existing_notes:
636
+ older.meta["notes"] = f"{existing_notes} | superseded attempt: {note}".strip(" |")
637
+ older.updated_at = _now_iso()
638
+ if newer:
639
+ newer.validity = "stale"
640
+ newer.updated_at = _now_iso()
641
+ self._store.save(newer, deduplicate=False)
642
+ if older and newer:
643
+ self._store.save(older, deduplicate=False)
644
+ # FLAG_FOR_HUMAN: no learning mutation; persistence handled by caller.
645
+
646
+ def _persist_conflict(self, conflict: Conflict, res: ConflictResolution) -> None:
647
+ """Write a human-reviewable conflict record to .GCC/."""
648
+ conflicts_dir = self._gcc_dir / "reasoning_learnings" / "conflicts"
649
+ conflicts_dir.mkdir(parents=True, exist_ok=True)
650
+ record = {
651
+ "learning_a": conflict.learning_a.id,
652
+ "learning_b": conflict.learning_b.id,
653
+ "concept_overlap": conflict.concept_overlap,
654
+ "embedding_similarity": conflict.embedding_similarity,
655
+ "confidence_a": conflict.confidence_a,
656
+ "confidence_b": conflict.confidence_b,
657
+ "goal_changed": conflict.goal_changed,
658
+ "action": res.action,
659
+ "reason": res.reason,
660
+ "resolved": False,
661
+ "timestamp": _now_iso(),
662
+ }
663
+ path = conflicts_dir / f"conflict_{conflict.learning_a.id}_{conflict.learning_b.id}.json"
664
+ try:
665
+ _atomic_json_write(path, record)
666
+ except Exception as exc:
667
+ logger.warning("thinkstack: failed to persist conflict record — %s", exc)
668
+
669
+ def list_conflicts(self) -> list[dict[str, Any]]:
670
+ """Return all persisted unresolved conflict records."""
671
+ conflicts_dir = self._gcc_dir / "reasoning_learnings" / "conflicts"
672
+ if not conflicts_dir.exists():
673
+ return []
674
+ out: list[dict[str, Any]] = []
675
+ for path in sorted(conflicts_dir.glob("*.json")):
676
+ try:
677
+ record = json.loads(path.read_text(encoding="utf-8"))
678
+ if record.get("resolved"):
679
+ continue
680
+ out.append(record)
681
+ except Exception as exc:
682
+ logger.debug("thinkstack: skipping malformed conflict %s — %s", path, exc)
683
+ return out
684
+
685
+ def _log_quarantine(self, learning: Learning, secrets: list[str]) -> None:
686
+ """Log a quarantined learning event to .GCC/events.log.jsonl.
687
+
688
+ The content preview is redacted before logging so secrets that
689
+ triggered the quarantine are never written to the audit log.
690
+ """
691
+ events_path = self._gcc_dir / "events.log.jsonl"
692
+ try:
693
+ record = {
694
+ "type": "LEARNING_QUARANTINED",
695
+ "learning_id": learning.id,
696
+ "secrets": secrets,
697
+ "content_preview": sanitize_for_extraction(learning.content[:120]),
698
+ "timestamp": _now_iso(),
699
+ }
700
+ with open(events_path, "a", encoding="utf-8") as f:
701
+ f.write(json.dumps(record) + "\n")
702
+ except Exception as exc:
703
+ logger.warning("thinkstack: failed to log quarantine event — %s", exc)
704
+
705
+ def apply_confidence_decay_sweep(self) -> list[str]:
706
+ """Apply confidence decay to all active learnings.
707
+
708
+ Returns the list of learning IDs that were modified (decayed or staled).
709
+ """
710
+ changed: list[str] = []
711
+ for learning in self._store.list(validity="active"):
712
+ before_conf = learning.confidence
713
+ before_val = learning.validity
714
+ apply_confidence_decay(learning)
715
+ if learning.confidence != before_conf or learning.validity != before_val:
716
+ self._store.save(learning, deduplicate=False)
717
+ changed.append(learning.id)
718
+ return changed
719
+
720
+ # ------------------------------------------------------------------
721
+ # Provenance
722
+ # ------------------------------------------------------------------
723
+
724
+ def record_injection(
725
+ self,
726
+ learning_ids: list[str],
727
+ *,
728
+ call_id: str,
729
+ session_id: str = "",
730
+ prompt_text: str = "",
731
+ model_name: str = "",
732
+ ) -> ProvenanceRecord:
733
+ """Record that a set of learnings was injected into a call."""
734
+ record = ProvenanceRecord(
735
+ call_id=call_id,
736
+ learning_ids=learning_ids,
737
+ session_id=session_id,
738
+ prompt_hash=_hash_prompt(prompt_text),
739
+ prompt_excerpt=prompt_text[:200],
740
+ model_name=model_name,
741
+ )
742
+ self._provenance.save(record)
743
+ return record
744
+
745
+ def record_injection_outcome(self, call_id: str, outcome: str) -> None:
746
+ """Update provenance for a call and apply feedback to injected learnings."""
747
+ record = self._provenance.update_outcome(call_id, outcome)
748
+ if not record:
749
+ return
750
+ for learning_id in record.learning_ids:
751
+ try:
752
+ self.apply_feedback(learning_id, outcome)
753
+ except Exception as exc:
754
+ logger.warning("thinkstack: feedback failed for learning %s — %s", learning_id, exc)
755
+
756
+ def get_provenance(self, call_id: str) -> ProvenanceRecord | None:
757
+ """Return the provenance record for a call, if any."""
758
+ return self._provenance.get_by_call(call_id)
759
+
760
+ def list_provenance(self, learning_id: str | None = None) -> list[ProvenanceRecord]:
761
+ """Return provenance records, optionally filtered by learning."""
762
+ return self._provenance.list(learning_id=learning_id)
763
+
764
+ def invalidate_stale(self, changed_paths: list[str]) -> list[str]:
765
+ """Mark learnings affected by the given changed paths as stale."""
766
+ from thinkstack_core.reasoning_plus.learning.state import affected_by_state_change
767
+ invalidated: list[str] = []
768
+ for learning in self._store.list(validity="active"):
769
+ if affected_by_state_change(self._repo_path, learning.trigger_concepts, changed_paths):
770
+ self._store.invalidate(learning.id, reason="stale")
771
+ invalidated.append(learning.id)
772
+ return invalidated
773
+
774
+ # ------------------------------------------------------------------
775
+ # State
776
+ # ------------------------------------------------------------------
777
+
778
+ def workspace_hash(self, file_paths: list[str] | None = None) -> str:
779
+ return workspace_state_hash(self._repo_path, file_paths)
780
+
781
+
782
+ def _now_iso() -> str:
783
+ from datetime import datetime, timezone
784
+ return datetime.now(timezone.utc).isoformat()