devtorch-core 3.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. devtorch_core/__init__.py +158 -0
  2. devtorch_core/aggphi_textual.py +275 -0
  3. devtorch_core/alerts/__init__.py +23 -0
  4. devtorch_core/alerts/base.py +46 -0
  5. devtorch_core/alerts/config.py +60 -0
  6. devtorch_core/alerts/dispatcher.py +110 -0
  7. devtorch_core/alerts/jira.py +96 -0
  8. devtorch_core/alerts/linear.py +72 -0
  9. devtorch_core/alerts/pagerduty.py +66 -0
  10. devtorch_core/alerts/slack.py +81 -0
  11. devtorch_core/alerts/teams.py +70 -0
  12. devtorch_core/audit/__init__.py +43 -0
  13. devtorch_core/audit/exporter.py +297 -0
  14. devtorch_core/audit/privacy.py +101 -0
  15. devtorch_core/audit/scrubber.py +149 -0
  16. devtorch_core/audit/service.py +67 -0
  17. devtorch_core/audit/signing.py +127 -0
  18. devtorch_core/broadcast/__init__.py +4 -0
  19. devtorch_core/broadcast/broadcaster.py +100 -0
  20. devtorch_core/broadcast/watcher.py +71 -0
  21. devtorch_core/capability.py +639 -0
  22. devtorch_core/cloud/__init__.py +1 -0
  23. devtorch_core/cloud/client_config.py +472 -0
  24. devtorch_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
  25. devtorch_core/cloud/client_configs/.claude-stdio.json +13 -0
  26. devtorch_core/cloud/client_configs/.cursor-mcp.json +13 -0
  27. devtorch_core/cloud/client_configs/.opencode-bridge.json +13 -0
  28. devtorch_core/cloud/client_configs/.opencode.json +15 -0
  29. devtorch_core/cloud/client_configs/.vscode-mcp.json +13 -0
  30. devtorch_core/cloud/devtorch-mcp-bridge.js +357 -0
  31. devtorch_core/cloud/mcp_client.py +229 -0
  32. devtorch_core/cloud/setup.py +144 -0
  33. devtorch_core/cloud/sync.py +143 -0
  34. devtorch_core/cloud/sync_bundle.py +603 -0
  35. devtorch_core/cloud/sync_conflicts.py +159 -0
  36. devtorch_core/cloud/sync_state.py +159 -0
  37. devtorch_core/cloud/team_sync.py +283 -0
  38. devtorch_core/codex/__init__.py +9 -0
  39. devtorch_core/codex/__main__.py +97 -0
  40. devtorch_core/codex/capture.py +208 -0
  41. devtorch_core/codex/proxy.py +412 -0
  42. devtorch_core/concept_catalog.py +209 -0
  43. devtorch_core/consolidation/__init__.py +3 -0
  44. devtorch_core/consolidation/synthesizer.py +87 -0
  45. devtorch_core/consolidation/workflow.py +175 -0
  46. devtorch_core/daemon/__init__.py +27 -0
  47. devtorch_core/daemon/supervisor.py +293 -0
  48. devtorch_core/daemon/watcher.py +244 -0
  49. devtorch_core/dashboard_api.py +2012 -0
  50. devtorch_core/deltaf.py +97 -0
  51. devtorch_core/disclosure.py +50 -0
  52. devtorch_core/divergence/__init__.py +3 -0
  53. devtorch_core/divergence/detector.py +166 -0
  54. devtorch_core/gateway/__init__.py +32 -0
  55. devtorch_core/gateway/key_manager.py +124 -0
  56. devtorch_core/gateway/metrics_webhook.py +252 -0
  57. devtorch_core/gateway/policy.py +262 -0
  58. devtorch_core/gateway/server.py +727 -0
  59. devtorch_core/gateway/sso.py +233 -0
  60. devtorch_core/gcc.py +1246 -0
  61. devtorch_core/github/__init__.py +35 -0
  62. devtorch_core/github/app.py +240 -0
  63. devtorch_core/github/comment_builder.py +113 -0
  64. devtorch_core/github/pat.py +76 -0
  65. devtorch_core/github/pr_parser.py +82 -0
  66. devtorch_core/github/pr_reporter.py +555 -0
  67. devtorch_core/gitlab/__init__.py +177 -0
  68. devtorch_core/hitl/__init__.py +4 -0
  69. devtorch_core/hitl/channels.py +129 -0
  70. devtorch_core/hitl/orchestrator.py +95 -0
  71. devtorch_core/hooks/__init__.py +17 -0
  72. devtorch_core/hooks/claude_code.py +228 -0
  73. devtorch_core/hooks/git_capture.py +341 -0
  74. devtorch_core/hooks/git_commit.py +182 -0
  75. devtorch_core/hooks/installer.py +733 -0
  76. devtorch_core/hooks/pre_commit.py +157 -0
  77. devtorch_core/hooks/runner.py +344 -0
  78. devtorch_core/identity/__init__.py +4 -0
  79. devtorch_core/identity/agent.py +86 -0
  80. devtorch_core/identity/providers.py +85 -0
  81. devtorch_core/invariants.py +182 -0
  82. devtorch_core/mcp/__init__.py +10 -0
  83. devtorch_core/mcp/auth.py +177 -0
  84. devtorch_core/mcp/server.py +1049 -0
  85. devtorch_core/metrics/__init__.py +35 -0
  86. devtorch_core/metrics/aggregate.py +215 -0
  87. devtorch_core/metrics/calibrate.py +198 -0
  88. devtorch_core/metrics/calibration.py +125 -0
  89. devtorch_core/metrics/credibility.py +288 -0
  90. devtorch_core/metrics/delivery_time.py +70 -0
  91. devtorch_core/metrics/dhs.py +126 -0
  92. devtorch_core/metrics/mcs.py +96 -0
  93. devtorch_core/metrics/roi.py +88 -0
  94. devtorch_core/metrics/session_writer.py +81 -0
  95. devtorch_core/metrics/shadow_ai.py +117 -0
  96. devtorch_core/metrics/sprint_writer.py +243 -0
  97. devtorch_core/observability/__init__.py +78 -0
  98. devtorch_core/observability/datadog.py +157 -0
  99. devtorch_core/observability/formatter.py +119 -0
  100. devtorch_core/observability/report.py +264 -0
  101. devtorch_core/observability/servicenow.py +147 -0
  102. devtorch_core/observability/splunk.py +218 -0
  103. devtorch_core/observability/webhook.py +227 -0
  104. devtorch_core/parser/__init__.py +30 -0
  105. devtorch_core/parser/blocks.py +216 -0
  106. devtorch_core/parser/inference.py +159 -0
  107. devtorch_core/parser/thinking.py +112 -0
  108. devtorch_core/projects.py +169 -0
  109. devtorch_core/prompt_artifact.py +76 -0
  110. devtorch_core/proxy/__init__.py +9 -0
  111. devtorch_core/proxy/routes/__init__.py +1 -0
  112. devtorch_core/proxy/routes/anthropic.py +264 -0
  113. devtorch_core/proxy/routes/azure_openai.py +336 -0
  114. devtorch_core/proxy/routes/gemini.py +331 -0
  115. devtorch_core/proxy/routes/groq.py +284 -0
  116. devtorch_core/proxy/routes/ollama.py +279 -0
  117. devtorch_core/proxy/routes/openai.py +287 -0
  118. devtorch_core/proxy/server.py +356 -0
  119. devtorch_core/query/__init__.py +15 -0
  120. devtorch_core/query/grep.py +181 -0
  121. devtorch_core/query/hybrid.py +86 -0
  122. devtorch_core/query/semantic.py +157 -0
  123. devtorch_core/rdp.py +105 -0
  124. devtorch_core/reasoning/__init__.py +4 -0
  125. devtorch_core/reasoning/entry.py +31 -0
  126. devtorch_core/reasoning/store.py +122 -0
  127. devtorch_core/reasoning_plus/__init__.py +70 -0
  128. devtorch_core/reasoning_plus/augmenter.py +326 -0
  129. devtorch_core/reasoning_plus/capture.py +51 -0
  130. devtorch_core/reasoning_plus/config.py +256 -0
  131. devtorch_core/reasoning_plus/context.py +262 -0
  132. devtorch_core/reasoning_plus/learning/__init__.py +72 -0
  133. devtorch_core/reasoning_plus/learning/analytics.py +141 -0
  134. devtorch_core/reasoning_plus/learning/api.py +313 -0
  135. devtorch_core/reasoning_plus/learning/chain.py +285 -0
  136. devtorch_core/reasoning_plus/learning/composer.py +74 -0
  137. devtorch_core/reasoning_plus/learning/cross_project.py +234 -0
  138. devtorch_core/reasoning_plus/learning/embeddings.py +209 -0
  139. devtorch_core/reasoning_plus/learning/extractor.py +207 -0
  140. devtorch_core/reasoning_plus/learning/models.py +116 -0
  141. devtorch_core/reasoning_plus/learning/provenance.py +126 -0
  142. devtorch_core/reasoning_plus/learning/recorder.py +81 -0
  143. devtorch_core/reasoning_plus/learning/relevance.py +122 -0
  144. devtorch_core/reasoning_plus/learning/state.py +86 -0
  145. devtorch_core/reasoning_plus/learning/store.py +160 -0
  146. devtorch_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
  147. devtorch_core/reasoning_plus/prompt.py +90 -0
  148. devtorch_core/rep.py +134 -0
  149. devtorch_core/rep_network/__init__.py +25 -0
  150. devtorch_core/rep_network/merge.py +70 -0
  151. devtorch_core/rep_network/node.py +137 -0
  152. devtorch_core/rep_network/server.py +140 -0
  153. devtorch_core/rep_network/sync.py +207 -0
  154. devtorch_core/sensitivity.py +182 -0
  155. devtorch_core/serve.py +258 -0
  156. devtorch_core/session/__init__.py +39 -0
  157. devtorch_core/session/disagreement.py +188 -0
  158. devtorch_core/session/models.py +114 -0
  159. devtorch_core/session/orchestrator.py +182 -0
  160. devtorch_core/session/planner.py +169 -0
  161. devtorch_core/session/simulator.py +132 -0
  162. devtorch_core/signing.py +290 -0
  163. devtorch_core/sis.py +197 -0
  164. devtorch_core/storage.py +308 -0
  165. devtorch_core/templates/__init__.py +6 -0
  166. devtorch_core/templates/engine.py +122 -0
  167. devtorch_core/templates/go.py +18 -0
  168. devtorch_core/templates/infra.py +19 -0
  169. devtorch_core/templates/library/__init__.py +18 -0
  170. devtorch_core/templates/library/api_design.md +27 -0
  171. devtorch_core/templates/library/bug_fix.md +27 -0
  172. devtorch_core/templates/library/decision_record.md +27 -0
  173. devtorch_core/templates/library/engine.py +228 -0
  174. devtorch_core/templates/library/security_review.md +30 -0
  175. devtorch_core/templates/python.py +19 -0
  176. devtorch_core/templates/react.py +18 -0
  177. devtorch_core/templates/typescript.py +18 -0
  178. devtorch_core/theta.py +221 -0
  179. devtorch_core/theta_synthesis.py +268 -0
  180. devtorch_core/topics.py +320 -0
  181. devtorch_core/variance.py +219 -0
  182. devtorch_core/wrapper/__init__.py +52 -0
  183. devtorch_core/wrapper/anthropic.py +487 -0
  184. devtorch_core/wrapper/base.py +562 -0
  185. devtorch_core/wrapper/bedrock.py +342 -0
  186. devtorch_core/wrapper/gemini.py +422 -0
  187. devtorch_core/wrapper/ollama.py +527 -0
  188. devtorch_core/wrapper/openai.py +461 -0
  189. devtorch_core-3.0.1.dist-info/METADATA +867 -0
  190. devtorch_core-3.0.1.dist-info/RECORD +193 -0
  191. devtorch_core-3.0.1.dist-info/WHEEL +5 -0
  192. devtorch_core-3.0.1.dist-info/entry_points.txt +2 -0
  193. devtorch_core-3.0.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,268 @@
1
+ """
2
+ Textual Mode AggPhi — LLM-powered Θ synthesis.
3
+
4
+ Generates a human-readable summary of the top Θ concepts and writes
5
+ it to theta.json as `textual_summary`. This is gated by A1 (capability
6
+ check) and can be disabled via DEVTORCH_DISABLE env var.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import dataclasses
12
+ import os
13
+ from typing import Optional
14
+
15
+
16
+ # ---------------------------------------------------------------------------
17
+ # Configuration
18
+ # ---------------------------------------------------------------------------
19
+
20
+ @dataclasses.dataclass
21
+ class ThetaSynthesisConfig:
22
+ """Configuration for the Textual Mode AggPhi synthesis step."""
23
+
24
+ max_concepts: int = 10
25
+ max_summary_chars: int = 500
26
+ model: str = "claude-haiku-4-5-20251001"
27
+ temperature: float = 0.3
28
+
29
+
30
+ # ---------------------------------------------------------------------------
31
+ # Result
32
+ # ---------------------------------------------------------------------------
33
+
34
+ @dataclasses.dataclass
35
+ class ThetaSynthesisResult:
36
+ """Result returned by synthesise_theta and run_synthesis_if_eligible."""
37
+
38
+ summary: str
39
+ concepts_used: list
40
+ model: str
41
+ tokens_used: int
42
+ skipped: bool = False
43
+ skip_reason: str = ""
44
+
45
+
46
+ # ---------------------------------------------------------------------------
47
+ # Prompt builder
48
+ # ---------------------------------------------------------------------------
49
+
50
+ def build_synthesis_prompt(theta_vector: dict, max_concepts: int = 10) -> str:
51
+ """
52
+ Build an LLM prompt that asks for a 2–3 sentence synthesis of the top Θ
53
+ concepts.
54
+
55
+ Args:
56
+ theta_vector: Mapping of concept → {mean_confidence, ...}
57
+ max_concepts: Maximum number of top concepts to include.
58
+
59
+ Returns:
60
+ A prompt string ready to send to an LLM.
61
+ """
62
+ # Extract and sort concepts by mean_confidence descending
63
+ def _conf(entry: object) -> float:
64
+ if isinstance(entry, dict):
65
+ return float(entry.get("mean_confidence", 0.0))
66
+ return 0.0
67
+
68
+ sorted_concepts = sorted(
69
+ theta_vector.items(),
70
+ key=lambda kv: _conf(kv[1]),
71
+ reverse=True,
72
+ )[:max_concepts]
73
+
74
+ if not sorted_concepts:
75
+ return (
76
+ "The coordination vector Θ is currently empty. "
77
+ "Please write 2-3 sentences summarising that no concepts have been "
78
+ "learned yet and that the agent team should proceed with caution."
79
+ )
80
+
81
+ concept_lines = "\n".join(
82
+ f" - {concept}: mean_confidence={_conf(data):.3f}"
83
+ for concept, data in sorted_concepts
84
+ )
85
+
86
+ prompt = (
87
+ "You are a governance analyst reviewing an AI agent team's coordination "
88
+ "vector Θ. The following concepts have been learned by the team, ranked "
89
+ "by confidence:\n\n"
90
+ f"{concept_lines}\n\n"
91
+ "Please write 2-3 concise sentences that:\n"
92
+ "1. Summarise what the agent team has learned so far.\n"
93
+ "2. Highlight any sensitive or high-confidence concepts to be careful about.\n"
94
+ "3. Note any concerns the team should keep in mind going forward.\n\n"
95
+ "Be direct and precise. Do not repeat the concept names verbatim — "
96
+ "synthesise the meaning."
97
+ )
98
+ return prompt
99
+
100
+
101
+ # ---------------------------------------------------------------------------
102
+ # Core synthesis function
103
+ # ---------------------------------------------------------------------------
104
+
105
+ def synthesise_theta(
106
+ gcc_repo,
107
+ config: Optional[ThetaSynthesisConfig] = None,
108
+ ) -> ThetaSynthesisResult:
109
+ """
110
+ Run LLM-powered Θ synthesis against the given GCCRepository.
111
+
112
+ Never raises — all errors produce a skipped result with a reason string.
113
+
114
+ Args:
115
+ gcc_repo: A GCCRepository instance (or compatible duck-type).
116
+ config: Optional ThetaSynthesisConfig; defaults are used if None.
117
+
118
+ Returns:
119
+ ThetaSynthesisResult with skipped=False on success, skipped=True otherwise.
120
+ """
121
+ if config is None:
122
+ config = ThetaSynthesisConfig()
123
+
124
+ # 1. Check DEVTORCH_DISABLE env var
125
+ if os.environ.get("DEVTORCH_DISABLE"):
126
+ return ThetaSynthesisResult(
127
+ summary="",
128
+ concepts_used=[],
129
+ model=config.model,
130
+ tokens_used=0,
131
+ skipped=True,
132
+ skip_reason="DEVTORCH_DISABLE is set",
133
+ )
134
+
135
+ # 2. Load current theta vector
136
+ try:
137
+ theta = gcc_repo.get_theta()
138
+ except Exception as exc:
139
+ return ThetaSynthesisResult(
140
+ summary="",
141
+ concepts_used=[],
142
+ model=config.model,
143
+ tokens_used=0,
144
+ skipped=True,
145
+ skip_reason=f"get_theta failed: {exc}",
146
+ )
147
+
148
+ coordination_vector: dict = theta.get("coordination_vector", {}) if isinstance(theta, dict) else {}
149
+
150
+ if not coordination_vector:
151
+ return ThetaSynthesisResult(
152
+ summary="",
153
+ concepts_used=[],
154
+ model=config.model,
155
+ tokens_used=0,
156
+ skipped=True,
157
+ skip_reason="empty theta",
158
+ )
159
+
160
+ # 3. Try to import anthropic (lazy / gated)
161
+ try:
162
+ import anthropic # noqa: PLC0415
163
+ except ImportError:
164
+ return ThetaSynthesisResult(
165
+ summary="",
166
+ concepts_used=[],
167
+ model=config.model,
168
+ tokens_used=0,
169
+ skipped=True,
170
+ skip_reason="anthropic not installed",
171
+ )
172
+
173
+ # 4. Build prompt and select top concepts
174
+ def _conf(entry: object) -> float:
175
+ if isinstance(entry, dict):
176
+ return float(entry.get("mean_confidence", 0.0))
177
+ return 0.0
178
+
179
+ top_concepts = sorted(
180
+ coordination_vector.items(),
181
+ key=lambda kv: _conf(kv[1]),
182
+ reverse=True,
183
+ )[: config.max_concepts]
184
+
185
+ concepts_used = [c for c, _ in top_concepts]
186
+ prompt = build_synthesis_prompt(coordination_vector, max_concepts=config.max_concepts)
187
+
188
+ # 5. Call the Anthropic API
189
+ try:
190
+ client = anthropic.Anthropic()
191
+ response = client.messages.create(
192
+ model=config.model,
193
+ max_tokens=256,
194
+ temperature=config.temperature,
195
+ messages=[{"role": "user", "content": prompt}],
196
+ )
197
+ raw_summary: str = response.content[0].text
198
+ tokens_used: int = response.usage.input_tokens + response.usage.output_tokens
199
+
200
+ # Truncate to max_summary_chars
201
+ summary = raw_summary[: config.max_summary_chars]
202
+
203
+ # 6. Write textual_summary back to the repo
204
+ gcc_repo.update_theta_field("textual_summary", summary)
205
+
206
+ return ThetaSynthesisResult(
207
+ summary=summary,
208
+ concepts_used=concepts_used,
209
+ model=config.model,
210
+ tokens_used=tokens_used,
211
+ skipped=False,
212
+ skip_reason="",
213
+ )
214
+
215
+ except Exception as exc:
216
+ return ThetaSynthesisResult(
217
+ summary="",
218
+ concepts_used=[],
219
+ model=config.model,
220
+ tokens_used=0,
221
+ skipped=True,
222
+ skip_reason=str(exc),
223
+ )
224
+
225
+
226
+ # ---------------------------------------------------------------------------
227
+ # Eligibility gate
228
+ # ---------------------------------------------------------------------------
229
+
230
+ def run_synthesis_if_eligible(
231
+ gcc_repo,
232
+ omega_capability=None,
233
+ ) -> ThetaSynthesisResult:
234
+ """
235
+ Run synthesis only when the A1 capability gate passes.
236
+
237
+ Args:
238
+ gcc_repo: A GCCRepository instance.
239
+ omega_capability: An OmegaCapability instance (or None to skip gate).
240
+ If provided, its `textual_mode_allowed` attribute or
241
+ method is checked before running synthesis.
242
+
243
+ Returns:
244
+ ThetaSynthesisResult — skipped if the gate fails.
245
+ """
246
+ # Import for side-effects / availability check (not used directly here)
247
+ from devtorch_core.capability import check_round_cap # noqa: F401, PLC0415
248
+
249
+ if omega_capability is not None:
250
+ tma = omega_capability.textual_mode_allowed
251
+ # textual_mode_allowed may be a method (returns (bool, reasons)) or a
252
+ # plain boolean attribute (e.g. on a MagicMock / CapabilityValidationResult)
253
+ if callable(tma):
254
+ allowed, _ = tma()
255
+ else:
256
+ allowed = bool(tma)
257
+
258
+ if not allowed:
259
+ return ThetaSynthesisResult(
260
+ summary="",
261
+ concepts_used=[],
262
+ model=ThetaSynthesisConfig().model,
263
+ tokens_used=0,
264
+ skipped=True,
265
+ skip_reason="A1 gate: textual mode not allowed",
266
+ )
267
+
268
+ return synthesise_theta(gcc_repo)
@@ -0,0 +1,320 @@
1
+ """
2
+ devtorch_core.topics
3
+ =====================
4
+ Topic taxonomy, persistence, and auto-derivation.
5
+
6
+ Topics are higher-level groupings of concepts. The default taxonomy is
7
+ hard-coded and merged at read time (not persisted). Custom topics are
8
+ stored in ``.GCC/topics.json``. Auto-derived topics are computed from
9
+ concept co-occurrence in commits and learnings using graph-based
10
+ connected components clustering.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import logging
16
+ from collections import defaultdict
17
+ from datetime import datetime, timezone
18
+ from itertools import combinations
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ logger = logging.getLogger("devtorch.topics")
23
+
24
+ DEFAULT_TAXONOMY: dict[str, list[str]] = {
25
+ "authentication": ["auth", "jwt", "oauth", "session", "secrets"],
26
+ "database": ["schema", "migration", "indexing", "partitioning"],
27
+ "api": ["api", "rest", "graphql", "websocket", "endpoints"],
28
+ "deployment": ["deploy", "docker", "kubernetes", "ci-cd", "config"],
29
+ "testing": ["testing", "unit-test", "integration-test", "e2e"],
30
+ "performance": ["performance", "caching", "optimization", "profiling"],
31
+ "security": ["security", "xss", "csrf", "sql-injection", "secrets", "pii"],
32
+ "governance": ["audit", "compliance", "disclosure", "sensitivity"],
33
+ }
34
+
35
+ _MIN_COMPONENT_SIZE = 3
36
+ _JACCARD_THRESHOLD = 0.3
37
+
38
+
39
+ class TopicStore:
40
+ """Persist and query topic→concepts mappings."""
41
+
42
+ def __init__(self, gcc_dir: Path | str) -> None:
43
+ self._gcc_dir = Path(gcc_dir)
44
+ self._path = self._gcc_dir / "topics.json"
45
+
46
+ def list_topics(self, with_data_only: bool = False) -> list[dict[str, Any]]:
47
+ """Return all topics with metadata.
48
+
49
+ Each entry: ``{"name", "source", "concepts", "has_data"}``
50
+
51
+ If *with_data_only* is True, only topics where at least one concept
52
+ appears in theta, sensitivity, learnings, or commits are returned.
53
+ """
54
+ custom = self._read_custom()
55
+ derived = self._read_derived()
56
+
57
+ all_topics: list[dict[str, Any]] = []
58
+
59
+ for name, concepts in DEFAULT_TAXONOMY.items():
60
+ has_data = self._topic_has_data(concepts)
61
+ all_topics.append({
62
+ "name": name,
63
+ "source": "default",
64
+ "concepts": concepts,
65
+ "has_data": has_data,
66
+ })
67
+
68
+ for name, concepts in custom.items():
69
+ has_data = self._topic_has_data(concepts)
70
+ all_topics.append({
71
+ "name": name,
72
+ "source": "custom",
73
+ "concepts": concepts,
74
+ "has_data": has_data,
75
+ })
76
+
77
+ for name, concepts in derived.items():
78
+ has_data = self._topic_has_data(concepts)
79
+ all_topics.append({
80
+ "name": name,
81
+ "source": "derived",
82
+ "concepts": concepts,
83
+ "has_data": has_data,
84
+ })
85
+
86
+ if with_data_only:
87
+ all_topics = [t for t in all_topics if t["has_data"]]
88
+
89
+ return sorted(all_topics, key=lambda t: t["name"])
90
+
91
+ def get_concepts(self, topic: str) -> list[str]:
92
+ """Return concepts for a topic from any source (custom, default, derived)."""
93
+ custom = self._read_custom()
94
+ if topic in custom:
95
+ return custom[topic]
96
+
97
+ derived = self._read_derived()
98
+ if topic in derived:
99
+ return derived[topic]
100
+
101
+ if topic in DEFAULT_TAXONOMY:
102
+ return DEFAULT_TAXONOMY[topic]
103
+
104
+ return []
105
+
106
+ def add_topic(self, name: str, concepts: list[str]) -> None:
107
+ """Add or update a custom topic mapping."""
108
+ data = self._read_full()
109
+ data.setdefault("custom_topics", {})[name] = sorted(set(concepts))
110
+ data["updated_at"] = _now_iso()
111
+ self._write_full(data)
112
+
113
+ def remove_topic(self, name: str) -> bool:
114
+ """Remove a custom topic. Returns True if removed."""
115
+ data = self._read_full()
116
+ custom = data.get("custom_topics", {})
117
+ if name not in custom:
118
+ return False
119
+ del custom[name]
120
+ data["updated_at"] = _now_iso()
121
+ self._write_full(data)
122
+ return True
123
+
124
+ def derive_topics(self, force: bool = False) -> dict[str, list[str]]:
125
+ """Derive topics from concept co-occurrence and cache results.
126
+
127
+ Re-derivation is skipped unless *force* is True or the commit/learning
128
+ count signature has changed since last derivation.
129
+ """
130
+ data = self._read_full()
131
+ current_sig = self._compute_signature()
132
+
133
+ if not force and data.get("derived_signature") == current_sig:
134
+ return data.get("derived_topics", {})
135
+
136
+ derived = self._compute_derived_topics()
137
+ data["derived_topics"] = derived
138
+ data["derived_at"] = _now_iso()
139
+ data["derived_signature"] = current_sig
140
+ data["updated_at"] = _now_iso()
141
+ self._write_full(data)
142
+ return derived
143
+
144
+ def _compute_signature(self) -> str:
145
+ """Compute a signature based on commit and learning counts."""
146
+ commits_dir = self._gcc_dir / "commits"
147
+ learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
148
+ n_commits = len(list(commits_dir.glob("*.json"))) if commits_dir.exists() else 0
149
+ n_learnings = len(list(learnings_dir.glob("*.json"))) if learnings_dir.exists() else 0
150
+ return f"commits:{n_commits},learnings:{n_learnings}"
151
+
152
+ def _compute_derived_topics(self) -> dict[str, list[str]]:
153
+ """Compute auto-derived topics using graph-based connected components.
154
+
155
+ Algorithm:
156
+ 1. Build a co-occurrence graph from commits and learnings.
157
+ 2. Weight edges by Jaccard similarity of co-occurrence.
158
+ 3. Remove edges with Jaccard < threshold.
159
+ 4. Find connected components.
160
+ 5. Discard components with fewer than MIN_COMPONENT_SIZE concepts.
161
+ 6. Label each component by its highest-degree concept.
162
+ """
163
+ co_occurrence: dict[frozenset[str], int] = defaultdict(int)
164
+ occurrence_count: dict[str, int] = defaultdict(int)
165
+
166
+ concept_sets = self._load_concept_sets()
167
+ for concept_list in concept_sets:
168
+ concepts = [c.lower() for c in concept_list]
169
+ for c in concepts:
170
+ occurrence_count[c] += 1
171
+ for pair in combinations(sorted(set(concepts)), 2):
172
+ co_occurrence[frozenset(pair)] += 1
173
+
174
+ if not co_occurrence:
175
+ return {}
176
+
177
+ adj: dict[str, set[str]] = defaultdict(set)
178
+ for pair, co_count in co_occurrence.items():
179
+ c1, c2 = sorted(pair)
180
+ union = occurrence_count[c1] + occurrence_count[c2] - co_count
181
+ jaccard = co_count / union if union > 0 else 0
182
+ if jaccard >= _JACCARD_THRESHOLD:
183
+ adj[c1].add(c2)
184
+ adj[c2].add(c1)
185
+
186
+ components = self._connected_components(adj)
187
+ derived: dict[str, list[str]] = {}
188
+ for comp in components:
189
+ if len(comp) < _MIN_COMPONENT_SIZE:
190
+ continue
191
+ label = max(comp, key=lambda c: len(adj.get(c, set())))
192
+ derived[label] = sorted(comp)
193
+
194
+ return derived
195
+
196
+ def _load_concept_sets(self) -> list[list[str]]:
197
+ """Load concept sets from commits and learnings."""
198
+ concept_sets: list[list[str]] = []
199
+
200
+ commits_dir = self._gcc_dir / "commits"
201
+ if commits_dir.exists():
202
+ for path in commits_dir.glob("*.json"):
203
+ try:
204
+ data = json.loads(path.read_text(encoding="utf-8"))
205
+ concepts = data.get("concepts", [])
206
+ if concepts:
207
+ concept_sets.append(concepts)
208
+ except (json.JSONDecodeError, OSError):
209
+ continue
210
+
211
+ learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
212
+ if learnings_dir.exists():
213
+ for path in learnings_dir.glob("*.json"):
214
+ try:
215
+ data = json.loads(path.read_text(encoding="utf-8"))
216
+ concepts = data.get("trigger_concepts", [])
217
+ if concepts:
218
+ concept_sets.append(concepts)
219
+ except (json.JSONDecodeError, OSError):
220
+ continue
221
+
222
+ return concept_sets
223
+
224
+ @staticmethod
225
+ def _connected_components(adj: dict[str, set[str]]) -> list[list[str]]:
226
+ """Find connected components in an undirected graph."""
227
+ visited: set[str] = set()
228
+ components: list[list[str]] = []
229
+ for node in adj:
230
+ if node in visited:
231
+ continue
232
+ stack = [node]
233
+ comp: list[str] = []
234
+ while stack:
235
+ curr = stack.pop()
236
+ if curr in visited:
237
+ continue
238
+ visited.add(curr)
239
+ comp.append(curr)
240
+ stack.extend(adj.get(curr, set()) - visited)
241
+ components.append(comp)
242
+ return components
243
+
244
+ def _topic_has_data(self, concepts: list[str]) -> bool:
245
+ """Check if at least one concept appears in theta, sensitivity, learnings, or commits."""
246
+ known = self._get_known_concepts()
247
+ return any(c.lower() in known for c in concepts)
248
+
249
+ def _get_known_concepts(self) -> set[str]:
250
+ """Get the set of all concepts that appear in any data source."""
251
+ known: set[str] = set()
252
+
253
+ theta_path = self._gcc_dir / "theta.json"
254
+ if theta_path.exists():
255
+ try:
256
+ theta = json.loads(theta_path.read_text(encoding="utf-8"))
257
+ known.update(k.lower() for k in theta.get("coordination_vector", {}).keys())
258
+ except (json.JSONDecodeError, OSError):
259
+ pass
260
+
261
+ events_path = self._gcc_dir / "sensitivities" / "events.jsonl"
262
+ if events_path.exists():
263
+ try:
264
+ for line in events_path.read_text(encoding="utf-8").splitlines():
265
+ if not line.strip():
266
+ continue
267
+ ev = json.loads(line)
268
+ payload = ev.get("payload", ev)
269
+ concept = payload.get("target_concept", "")
270
+ if concept:
271
+ known.add(concept.lower())
272
+ except (json.JSONDecodeError, OSError):
273
+ pass
274
+
275
+ learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
276
+ if learnings_dir.exists():
277
+ for path in learnings_dir.glob("*.json"):
278
+ try:
279
+ data = json.loads(path.read_text(encoding="utf-8"))
280
+ for c in data.get("trigger_concepts", []):
281
+ known.add(c.lower())
282
+ except (json.JSONDecodeError, OSError):
283
+ continue
284
+
285
+ commits_dir = self._gcc_dir / "commits"
286
+ if commits_dir.exists():
287
+ for path in commits_dir.glob("*.json"):
288
+ try:
289
+ data = json.loads(path.read_text(encoding="utf-8"))
290
+ for c in data.get("concepts", []):
291
+ known.add(c.lower())
292
+ except (json.JSONDecodeError, OSError):
293
+ continue
294
+
295
+ return known
296
+
297
+ def _read_custom(self) -> dict[str, list[str]]:
298
+ return self._read_full().get("custom_topics", {})
299
+
300
+ def _read_derived(self) -> dict[str, list[str]]:
301
+ return self._read_full().get("derived_topics", {})
302
+
303
+ def _read_full(self) -> dict[str, Any]:
304
+ if not self._path.exists():
305
+ return {"version": "1", "custom_topics": {}, "derived_topics": {}}
306
+ try:
307
+ return json.loads(self._path.read_text(encoding="utf-8"))
308
+ except (json.JSONDecodeError, OSError):
309
+ return {"version": "1", "custom_topics": {}, "derived_topics": {}}
310
+
311
+ def _write_full(self, data: dict[str, Any]) -> None:
312
+ self._gcc_dir.mkdir(parents=True, exist_ok=True)
313
+ self._path.write_text(
314
+ json.dumps(data, indent=2, sort_keys=True) + "\n",
315
+ encoding="utf-8",
316
+ )
317
+
318
+
319
+ def _now_iso() -> str:
320
+ return datetime.now(timezone.utc).isoformat()