devtorch-core 3.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- devtorch_core/__init__.py +158 -0
- devtorch_core/aggphi_textual.py +275 -0
- devtorch_core/alerts/__init__.py +23 -0
- devtorch_core/alerts/base.py +46 -0
- devtorch_core/alerts/config.py +60 -0
- devtorch_core/alerts/dispatcher.py +110 -0
- devtorch_core/alerts/jira.py +96 -0
- devtorch_core/alerts/linear.py +72 -0
- devtorch_core/alerts/pagerduty.py +66 -0
- devtorch_core/alerts/slack.py +81 -0
- devtorch_core/alerts/teams.py +70 -0
- devtorch_core/audit/__init__.py +43 -0
- devtorch_core/audit/exporter.py +297 -0
- devtorch_core/audit/privacy.py +101 -0
- devtorch_core/audit/scrubber.py +149 -0
- devtorch_core/audit/service.py +67 -0
- devtorch_core/audit/signing.py +127 -0
- devtorch_core/broadcast/__init__.py +4 -0
- devtorch_core/broadcast/broadcaster.py +100 -0
- devtorch_core/broadcast/watcher.py +71 -0
- devtorch_core/capability.py +639 -0
- devtorch_core/cloud/__init__.py +1 -0
- devtorch_core/cloud/client_config.py +472 -0
- devtorch_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
- devtorch_core/cloud/client_configs/.claude-stdio.json +13 -0
- devtorch_core/cloud/client_configs/.cursor-mcp.json +13 -0
- devtorch_core/cloud/client_configs/.opencode-bridge.json +13 -0
- devtorch_core/cloud/client_configs/.opencode.json +15 -0
- devtorch_core/cloud/client_configs/.vscode-mcp.json +13 -0
- devtorch_core/cloud/devtorch-mcp-bridge.js +357 -0
- devtorch_core/cloud/mcp_client.py +229 -0
- devtorch_core/cloud/setup.py +144 -0
- devtorch_core/cloud/sync.py +143 -0
- devtorch_core/cloud/sync_bundle.py +603 -0
- devtorch_core/cloud/sync_conflicts.py +159 -0
- devtorch_core/cloud/sync_state.py +159 -0
- devtorch_core/cloud/team_sync.py +283 -0
- devtorch_core/codex/__init__.py +9 -0
- devtorch_core/codex/__main__.py +97 -0
- devtorch_core/codex/capture.py +208 -0
- devtorch_core/codex/proxy.py +412 -0
- devtorch_core/concept_catalog.py +209 -0
- devtorch_core/consolidation/__init__.py +3 -0
- devtorch_core/consolidation/synthesizer.py +87 -0
- devtorch_core/consolidation/workflow.py +175 -0
- devtorch_core/daemon/__init__.py +27 -0
- devtorch_core/daemon/supervisor.py +293 -0
- devtorch_core/daemon/watcher.py +244 -0
- devtorch_core/dashboard_api.py +2012 -0
- devtorch_core/deltaf.py +97 -0
- devtorch_core/disclosure.py +50 -0
- devtorch_core/divergence/__init__.py +3 -0
- devtorch_core/divergence/detector.py +166 -0
- devtorch_core/gateway/__init__.py +32 -0
- devtorch_core/gateway/key_manager.py +124 -0
- devtorch_core/gateway/metrics_webhook.py +252 -0
- devtorch_core/gateway/policy.py +262 -0
- devtorch_core/gateway/server.py +727 -0
- devtorch_core/gateway/sso.py +233 -0
- devtorch_core/gcc.py +1246 -0
- devtorch_core/github/__init__.py +35 -0
- devtorch_core/github/app.py +240 -0
- devtorch_core/github/comment_builder.py +113 -0
- devtorch_core/github/pat.py +76 -0
- devtorch_core/github/pr_parser.py +82 -0
- devtorch_core/github/pr_reporter.py +555 -0
- devtorch_core/gitlab/__init__.py +177 -0
- devtorch_core/hitl/__init__.py +4 -0
- devtorch_core/hitl/channels.py +129 -0
- devtorch_core/hitl/orchestrator.py +95 -0
- devtorch_core/hooks/__init__.py +17 -0
- devtorch_core/hooks/claude_code.py +228 -0
- devtorch_core/hooks/git_capture.py +341 -0
- devtorch_core/hooks/git_commit.py +182 -0
- devtorch_core/hooks/installer.py +733 -0
- devtorch_core/hooks/pre_commit.py +157 -0
- devtorch_core/hooks/runner.py +344 -0
- devtorch_core/identity/__init__.py +4 -0
- devtorch_core/identity/agent.py +86 -0
- devtorch_core/identity/providers.py +85 -0
- devtorch_core/invariants.py +182 -0
- devtorch_core/mcp/__init__.py +10 -0
- devtorch_core/mcp/auth.py +177 -0
- devtorch_core/mcp/server.py +1049 -0
- devtorch_core/metrics/__init__.py +35 -0
- devtorch_core/metrics/aggregate.py +215 -0
- devtorch_core/metrics/calibrate.py +198 -0
- devtorch_core/metrics/calibration.py +125 -0
- devtorch_core/metrics/credibility.py +288 -0
- devtorch_core/metrics/delivery_time.py +70 -0
- devtorch_core/metrics/dhs.py +126 -0
- devtorch_core/metrics/mcs.py +96 -0
- devtorch_core/metrics/roi.py +88 -0
- devtorch_core/metrics/session_writer.py +81 -0
- devtorch_core/metrics/shadow_ai.py +117 -0
- devtorch_core/metrics/sprint_writer.py +243 -0
- devtorch_core/observability/__init__.py +78 -0
- devtorch_core/observability/datadog.py +157 -0
- devtorch_core/observability/formatter.py +119 -0
- devtorch_core/observability/report.py +264 -0
- devtorch_core/observability/servicenow.py +147 -0
- devtorch_core/observability/splunk.py +218 -0
- devtorch_core/observability/webhook.py +227 -0
- devtorch_core/parser/__init__.py +30 -0
- devtorch_core/parser/blocks.py +216 -0
- devtorch_core/parser/inference.py +159 -0
- devtorch_core/parser/thinking.py +112 -0
- devtorch_core/projects.py +169 -0
- devtorch_core/prompt_artifact.py +76 -0
- devtorch_core/proxy/__init__.py +9 -0
- devtorch_core/proxy/routes/__init__.py +1 -0
- devtorch_core/proxy/routes/anthropic.py +264 -0
- devtorch_core/proxy/routes/azure_openai.py +336 -0
- devtorch_core/proxy/routes/gemini.py +331 -0
- devtorch_core/proxy/routes/groq.py +284 -0
- devtorch_core/proxy/routes/ollama.py +279 -0
- devtorch_core/proxy/routes/openai.py +287 -0
- devtorch_core/proxy/server.py +356 -0
- devtorch_core/query/__init__.py +15 -0
- devtorch_core/query/grep.py +181 -0
- devtorch_core/query/hybrid.py +86 -0
- devtorch_core/query/semantic.py +157 -0
- devtorch_core/rdp.py +105 -0
- devtorch_core/reasoning/__init__.py +4 -0
- devtorch_core/reasoning/entry.py +31 -0
- devtorch_core/reasoning/store.py +122 -0
- devtorch_core/reasoning_plus/__init__.py +70 -0
- devtorch_core/reasoning_plus/augmenter.py +326 -0
- devtorch_core/reasoning_plus/capture.py +51 -0
- devtorch_core/reasoning_plus/config.py +256 -0
- devtorch_core/reasoning_plus/context.py +262 -0
- devtorch_core/reasoning_plus/learning/__init__.py +72 -0
- devtorch_core/reasoning_plus/learning/analytics.py +141 -0
- devtorch_core/reasoning_plus/learning/api.py +313 -0
- devtorch_core/reasoning_plus/learning/chain.py +285 -0
- devtorch_core/reasoning_plus/learning/composer.py +74 -0
- devtorch_core/reasoning_plus/learning/cross_project.py +234 -0
- devtorch_core/reasoning_plus/learning/embeddings.py +209 -0
- devtorch_core/reasoning_plus/learning/extractor.py +207 -0
- devtorch_core/reasoning_plus/learning/models.py +116 -0
- devtorch_core/reasoning_plus/learning/provenance.py +126 -0
- devtorch_core/reasoning_plus/learning/recorder.py +81 -0
- devtorch_core/reasoning_plus/learning/relevance.py +122 -0
- devtorch_core/reasoning_plus/learning/state.py +86 -0
- devtorch_core/reasoning_plus/learning/store.py +160 -0
- devtorch_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
- devtorch_core/reasoning_plus/prompt.py +90 -0
- devtorch_core/rep.py +134 -0
- devtorch_core/rep_network/__init__.py +25 -0
- devtorch_core/rep_network/merge.py +70 -0
- devtorch_core/rep_network/node.py +137 -0
- devtorch_core/rep_network/server.py +140 -0
- devtorch_core/rep_network/sync.py +207 -0
- devtorch_core/sensitivity.py +182 -0
- devtorch_core/serve.py +258 -0
- devtorch_core/session/__init__.py +39 -0
- devtorch_core/session/disagreement.py +188 -0
- devtorch_core/session/models.py +114 -0
- devtorch_core/session/orchestrator.py +182 -0
- devtorch_core/session/planner.py +169 -0
- devtorch_core/session/simulator.py +132 -0
- devtorch_core/signing.py +290 -0
- devtorch_core/sis.py +197 -0
- devtorch_core/storage.py +308 -0
- devtorch_core/templates/__init__.py +6 -0
- devtorch_core/templates/engine.py +122 -0
- devtorch_core/templates/go.py +18 -0
- devtorch_core/templates/infra.py +19 -0
- devtorch_core/templates/library/__init__.py +18 -0
- devtorch_core/templates/library/api_design.md +27 -0
- devtorch_core/templates/library/bug_fix.md +27 -0
- devtorch_core/templates/library/decision_record.md +27 -0
- devtorch_core/templates/library/engine.py +228 -0
- devtorch_core/templates/library/security_review.md +30 -0
- devtorch_core/templates/python.py +19 -0
- devtorch_core/templates/react.py +18 -0
- devtorch_core/templates/typescript.py +18 -0
- devtorch_core/theta.py +221 -0
- devtorch_core/theta_synthesis.py +268 -0
- devtorch_core/topics.py +320 -0
- devtorch_core/variance.py +219 -0
- devtorch_core/wrapper/__init__.py +52 -0
- devtorch_core/wrapper/anthropic.py +487 -0
- devtorch_core/wrapper/base.py +562 -0
- devtorch_core/wrapper/bedrock.py +342 -0
- devtorch_core/wrapper/gemini.py +422 -0
- devtorch_core/wrapper/ollama.py +527 -0
- devtorch_core/wrapper/openai.py +461 -0
- devtorch_core-3.0.1.dist-info/METADATA +867 -0
- devtorch_core-3.0.1.dist-info/RECORD +193 -0
- devtorch_core-3.0.1.dist-info/WHEEL +5 -0
- devtorch_core-3.0.1.dist-info/entry_points.txt +2 -0
- devtorch_core-3.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Textual Mode AggPhi — LLM-powered Θ synthesis.
|
|
3
|
+
|
|
4
|
+
Generates a human-readable summary of the top Θ concepts and writes
|
|
5
|
+
it to theta.json as `textual_summary`. This is gated by A1 (capability
|
|
6
|
+
check) and can be disabled via DEVTORCH_DISABLE env var.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import dataclasses
|
|
12
|
+
import os
|
|
13
|
+
from typing import Optional
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
# Configuration
|
|
18
|
+
# ---------------------------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
@dataclasses.dataclass
|
|
21
|
+
class ThetaSynthesisConfig:
|
|
22
|
+
"""Configuration for the Textual Mode AggPhi synthesis step."""
|
|
23
|
+
|
|
24
|
+
max_concepts: int = 10
|
|
25
|
+
max_summary_chars: int = 500
|
|
26
|
+
model: str = "claude-haiku-4-5-20251001"
|
|
27
|
+
temperature: float = 0.3
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# ---------------------------------------------------------------------------
|
|
31
|
+
# Result
|
|
32
|
+
# ---------------------------------------------------------------------------
|
|
33
|
+
|
|
34
|
+
@dataclasses.dataclass
|
|
35
|
+
class ThetaSynthesisResult:
|
|
36
|
+
"""Result returned by synthesise_theta and run_synthesis_if_eligible."""
|
|
37
|
+
|
|
38
|
+
summary: str
|
|
39
|
+
concepts_used: list
|
|
40
|
+
model: str
|
|
41
|
+
tokens_used: int
|
|
42
|
+
skipped: bool = False
|
|
43
|
+
skip_reason: str = ""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# ---------------------------------------------------------------------------
|
|
47
|
+
# Prompt builder
|
|
48
|
+
# ---------------------------------------------------------------------------
|
|
49
|
+
|
|
50
|
+
def build_synthesis_prompt(theta_vector: dict, max_concepts: int = 10) -> str:
|
|
51
|
+
"""
|
|
52
|
+
Build an LLM prompt that asks for a 2–3 sentence synthesis of the top Θ
|
|
53
|
+
concepts.
|
|
54
|
+
|
|
55
|
+
Args:
|
|
56
|
+
theta_vector: Mapping of concept → {mean_confidence, ...}
|
|
57
|
+
max_concepts: Maximum number of top concepts to include.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
A prompt string ready to send to an LLM.
|
|
61
|
+
"""
|
|
62
|
+
# Extract and sort concepts by mean_confidence descending
|
|
63
|
+
def _conf(entry: object) -> float:
|
|
64
|
+
if isinstance(entry, dict):
|
|
65
|
+
return float(entry.get("mean_confidence", 0.0))
|
|
66
|
+
return 0.0
|
|
67
|
+
|
|
68
|
+
sorted_concepts = sorted(
|
|
69
|
+
theta_vector.items(),
|
|
70
|
+
key=lambda kv: _conf(kv[1]),
|
|
71
|
+
reverse=True,
|
|
72
|
+
)[:max_concepts]
|
|
73
|
+
|
|
74
|
+
if not sorted_concepts:
|
|
75
|
+
return (
|
|
76
|
+
"The coordination vector Θ is currently empty. "
|
|
77
|
+
"Please write 2-3 sentences summarising that no concepts have been "
|
|
78
|
+
"learned yet and that the agent team should proceed with caution."
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
concept_lines = "\n".join(
|
|
82
|
+
f" - {concept}: mean_confidence={_conf(data):.3f}"
|
|
83
|
+
for concept, data in sorted_concepts
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
prompt = (
|
|
87
|
+
"You are a governance analyst reviewing an AI agent team's coordination "
|
|
88
|
+
"vector Θ. The following concepts have been learned by the team, ranked "
|
|
89
|
+
"by confidence:\n\n"
|
|
90
|
+
f"{concept_lines}\n\n"
|
|
91
|
+
"Please write 2-3 concise sentences that:\n"
|
|
92
|
+
"1. Summarise what the agent team has learned so far.\n"
|
|
93
|
+
"2. Highlight any sensitive or high-confidence concepts to be careful about.\n"
|
|
94
|
+
"3. Note any concerns the team should keep in mind going forward.\n\n"
|
|
95
|
+
"Be direct and precise. Do not repeat the concept names verbatim — "
|
|
96
|
+
"synthesise the meaning."
|
|
97
|
+
)
|
|
98
|
+
return prompt
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ---------------------------------------------------------------------------
|
|
102
|
+
# Core synthesis function
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
|
|
105
|
+
def synthesise_theta(
|
|
106
|
+
gcc_repo,
|
|
107
|
+
config: Optional[ThetaSynthesisConfig] = None,
|
|
108
|
+
) -> ThetaSynthesisResult:
|
|
109
|
+
"""
|
|
110
|
+
Run LLM-powered Θ synthesis against the given GCCRepository.
|
|
111
|
+
|
|
112
|
+
Never raises — all errors produce a skipped result with a reason string.
|
|
113
|
+
|
|
114
|
+
Args:
|
|
115
|
+
gcc_repo: A GCCRepository instance (or compatible duck-type).
|
|
116
|
+
config: Optional ThetaSynthesisConfig; defaults are used if None.
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
ThetaSynthesisResult with skipped=False on success, skipped=True otherwise.
|
|
120
|
+
"""
|
|
121
|
+
if config is None:
|
|
122
|
+
config = ThetaSynthesisConfig()
|
|
123
|
+
|
|
124
|
+
# 1. Check DEVTORCH_DISABLE env var
|
|
125
|
+
if os.environ.get("DEVTORCH_DISABLE"):
|
|
126
|
+
return ThetaSynthesisResult(
|
|
127
|
+
summary="",
|
|
128
|
+
concepts_used=[],
|
|
129
|
+
model=config.model,
|
|
130
|
+
tokens_used=0,
|
|
131
|
+
skipped=True,
|
|
132
|
+
skip_reason="DEVTORCH_DISABLE is set",
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
# 2. Load current theta vector
|
|
136
|
+
try:
|
|
137
|
+
theta = gcc_repo.get_theta()
|
|
138
|
+
except Exception as exc:
|
|
139
|
+
return ThetaSynthesisResult(
|
|
140
|
+
summary="",
|
|
141
|
+
concepts_used=[],
|
|
142
|
+
model=config.model,
|
|
143
|
+
tokens_used=0,
|
|
144
|
+
skipped=True,
|
|
145
|
+
skip_reason=f"get_theta failed: {exc}",
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
coordination_vector: dict = theta.get("coordination_vector", {}) if isinstance(theta, dict) else {}
|
|
149
|
+
|
|
150
|
+
if not coordination_vector:
|
|
151
|
+
return ThetaSynthesisResult(
|
|
152
|
+
summary="",
|
|
153
|
+
concepts_used=[],
|
|
154
|
+
model=config.model,
|
|
155
|
+
tokens_used=0,
|
|
156
|
+
skipped=True,
|
|
157
|
+
skip_reason="empty theta",
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
# 3. Try to import anthropic (lazy / gated)
|
|
161
|
+
try:
|
|
162
|
+
import anthropic # noqa: PLC0415
|
|
163
|
+
except ImportError:
|
|
164
|
+
return ThetaSynthesisResult(
|
|
165
|
+
summary="",
|
|
166
|
+
concepts_used=[],
|
|
167
|
+
model=config.model,
|
|
168
|
+
tokens_used=0,
|
|
169
|
+
skipped=True,
|
|
170
|
+
skip_reason="anthropic not installed",
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
# 4. Build prompt and select top concepts
|
|
174
|
+
def _conf(entry: object) -> float:
|
|
175
|
+
if isinstance(entry, dict):
|
|
176
|
+
return float(entry.get("mean_confidence", 0.0))
|
|
177
|
+
return 0.0
|
|
178
|
+
|
|
179
|
+
top_concepts = sorted(
|
|
180
|
+
coordination_vector.items(),
|
|
181
|
+
key=lambda kv: _conf(kv[1]),
|
|
182
|
+
reverse=True,
|
|
183
|
+
)[: config.max_concepts]
|
|
184
|
+
|
|
185
|
+
concepts_used = [c for c, _ in top_concepts]
|
|
186
|
+
prompt = build_synthesis_prompt(coordination_vector, max_concepts=config.max_concepts)
|
|
187
|
+
|
|
188
|
+
# 5. Call the Anthropic API
|
|
189
|
+
try:
|
|
190
|
+
client = anthropic.Anthropic()
|
|
191
|
+
response = client.messages.create(
|
|
192
|
+
model=config.model,
|
|
193
|
+
max_tokens=256,
|
|
194
|
+
temperature=config.temperature,
|
|
195
|
+
messages=[{"role": "user", "content": prompt}],
|
|
196
|
+
)
|
|
197
|
+
raw_summary: str = response.content[0].text
|
|
198
|
+
tokens_used: int = response.usage.input_tokens + response.usage.output_tokens
|
|
199
|
+
|
|
200
|
+
# Truncate to max_summary_chars
|
|
201
|
+
summary = raw_summary[: config.max_summary_chars]
|
|
202
|
+
|
|
203
|
+
# 6. Write textual_summary back to the repo
|
|
204
|
+
gcc_repo.update_theta_field("textual_summary", summary)
|
|
205
|
+
|
|
206
|
+
return ThetaSynthesisResult(
|
|
207
|
+
summary=summary,
|
|
208
|
+
concepts_used=concepts_used,
|
|
209
|
+
model=config.model,
|
|
210
|
+
tokens_used=tokens_used,
|
|
211
|
+
skipped=False,
|
|
212
|
+
skip_reason="",
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
except Exception as exc:
|
|
216
|
+
return ThetaSynthesisResult(
|
|
217
|
+
summary="",
|
|
218
|
+
concepts_used=[],
|
|
219
|
+
model=config.model,
|
|
220
|
+
tokens_used=0,
|
|
221
|
+
skipped=True,
|
|
222
|
+
skip_reason=str(exc),
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
# ---------------------------------------------------------------------------
|
|
227
|
+
# Eligibility gate
|
|
228
|
+
# ---------------------------------------------------------------------------
|
|
229
|
+
|
|
230
|
+
def run_synthesis_if_eligible(
|
|
231
|
+
gcc_repo,
|
|
232
|
+
omega_capability=None,
|
|
233
|
+
) -> ThetaSynthesisResult:
|
|
234
|
+
"""
|
|
235
|
+
Run synthesis only when the A1 capability gate passes.
|
|
236
|
+
|
|
237
|
+
Args:
|
|
238
|
+
gcc_repo: A GCCRepository instance.
|
|
239
|
+
omega_capability: An OmegaCapability instance (or None to skip gate).
|
|
240
|
+
If provided, its `textual_mode_allowed` attribute or
|
|
241
|
+
method is checked before running synthesis.
|
|
242
|
+
|
|
243
|
+
Returns:
|
|
244
|
+
ThetaSynthesisResult — skipped if the gate fails.
|
|
245
|
+
"""
|
|
246
|
+
# Import for side-effects / availability check (not used directly here)
|
|
247
|
+
from devtorch_core.capability import check_round_cap # noqa: F401, PLC0415
|
|
248
|
+
|
|
249
|
+
if omega_capability is not None:
|
|
250
|
+
tma = omega_capability.textual_mode_allowed
|
|
251
|
+
# textual_mode_allowed may be a method (returns (bool, reasons)) or a
|
|
252
|
+
# plain boolean attribute (e.g. on a MagicMock / CapabilityValidationResult)
|
|
253
|
+
if callable(tma):
|
|
254
|
+
allowed, _ = tma()
|
|
255
|
+
else:
|
|
256
|
+
allowed = bool(tma)
|
|
257
|
+
|
|
258
|
+
if not allowed:
|
|
259
|
+
return ThetaSynthesisResult(
|
|
260
|
+
summary="",
|
|
261
|
+
concepts_used=[],
|
|
262
|
+
model=ThetaSynthesisConfig().model,
|
|
263
|
+
tokens_used=0,
|
|
264
|
+
skipped=True,
|
|
265
|
+
skip_reason="A1 gate: textual mode not allowed",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
return synthesise_theta(gcc_repo)
|
devtorch_core/topics.py
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
"""
|
|
2
|
+
devtorch_core.topics
|
|
3
|
+
=====================
|
|
4
|
+
Topic taxonomy, persistence, and auto-derivation.
|
|
5
|
+
|
|
6
|
+
Topics are higher-level groupings of concepts. The default taxonomy is
|
|
7
|
+
hard-coded and merged at read time (not persisted). Custom topics are
|
|
8
|
+
stored in ``.GCC/topics.json``. Auto-derived topics are computed from
|
|
9
|
+
concept co-occurrence in commits and learnings using graph-based
|
|
10
|
+
connected components clustering.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
from collections import defaultdict
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from itertools import combinations
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("devtorch.topics")
|
|
23
|
+
|
|
24
|
+
DEFAULT_TAXONOMY: dict[str, list[str]] = {
|
|
25
|
+
"authentication": ["auth", "jwt", "oauth", "session", "secrets"],
|
|
26
|
+
"database": ["schema", "migration", "indexing", "partitioning"],
|
|
27
|
+
"api": ["api", "rest", "graphql", "websocket", "endpoints"],
|
|
28
|
+
"deployment": ["deploy", "docker", "kubernetes", "ci-cd", "config"],
|
|
29
|
+
"testing": ["testing", "unit-test", "integration-test", "e2e"],
|
|
30
|
+
"performance": ["performance", "caching", "optimization", "profiling"],
|
|
31
|
+
"security": ["security", "xss", "csrf", "sql-injection", "secrets", "pii"],
|
|
32
|
+
"governance": ["audit", "compliance", "disclosure", "sensitivity"],
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
_MIN_COMPONENT_SIZE = 3
|
|
36
|
+
_JACCARD_THRESHOLD = 0.3
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class TopicStore:
|
|
40
|
+
"""Persist and query topic→concepts mappings."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, gcc_dir: Path | str) -> None:
|
|
43
|
+
self._gcc_dir = Path(gcc_dir)
|
|
44
|
+
self._path = self._gcc_dir / "topics.json"
|
|
45
|
+
|
|
46
|
+
def list_topics(self, with_data_only: bool = False) -> list[dict[str, Any]]:
|
|
47
|
+
"""Return all topics with metadata.
|
|
48
|
+
|
|
49
|
+
Each entry: ``{"name", "source", "concepts", "has_data"}``
|
|
50
|
+
|
|
51
|
+
If *with_data_only* is True, only topics where at least one concept
|
|
52
|
+
appears in theta, sensitivity, learnings, or commits are returned.
|
|
53
|
+
"""
|
|
54
|
+
custom = self._read_custom()
|
|
55
|
+
derived = self._read_derived()
|
|
56
|
+
|
|
57
|
+
all_topics: list[dict[str, Any]] = []
|
|
58
|
+
|
|
59
|
+
for name, concepts in DEFAULT_TAXONOMY.items():
|
|
60
|
+
has_data = self._topic_has_data(concepts)
|
|
61
|
+
all_topics.append({
|
|
62
|
+
"name": name,
|
|
63
|
+
"source": "default",
|
|
64
|
+
"concepts": concepts,
|
|
65
|
+
"has_data": has_data,
|
|
66
|
+
})
|
|
67
|
+
|
|
68
|
+
for name, concepts in custom.items():
|
|
69
|
+
has_data = self._topic_has_data(concepts)
|
|
70
|
+
all_topics.append({
|
|
71
|
+
"name": name,
|
|
72
|
+
"source": "custom",
|
|
73
|
+
"concepts": concepts,
|
|
74
|
+
"has_data": has_data,
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
for name, concepts in derived.items():
|
|
78
|
+
has_data = self._topic_has_data(concepts)
|
|
79
|
+
all_topics.append({
|
|
80
|
+
"name": name,
|
|
81
|
+
"source": "derived",
|
|
82
|
+
"concepts": concepts,
|
|
83
|
+
"has_data": has_data,
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
if with_data_only:
|
|
87
|
+
all_topics = [t for t in all_topics if t["has_data"]]
|
|
88
|
+
|
|
89
|
+
return sorted(all_topics, key=lambda t: t["name"])
|
|
90
|
+
|
|
91
|
+
def get_concepts(self, topic: str) -> list[str]:
|
|
92
|
+
"""Return concepts for a topic from any source (custom, default, derived)."""
|
|
93
|
+
custom = self._read_custom()
|
|
94
|
+
if topic in custom:
|
|
95
|
+
return custom[topic]
|
|
96
|
+
|
|
97
|
+
derived = self._read_derived()
|
|
98
|
+
if topic in derived:
|
|
99
|
+
return derived[topic]
|
|
100
|
+
|
|
101
|
+
if topic in DEFAULT_TAXONOMY:
|
|
102
|
+
return DEFAULT_TAXONOMY[topic]
|
|
103
|
+
|
|
104
|
+
return []
|
|
105
|
+
|
|
106
|
+
def add_topic(self, name: str, concepts: list[str]) -> None:
|
|
107
|
+
"""Add or update a custom topic mapping."""
|
|
108
|
+
data = self._read_full()
|
|
109
|
+
data.setdefault("custom_topics", {})[name] = sorted(set(concepts))
|
|
110
|
+
data["updated_at"] = _now_iso()
|
|
111
|
+
self._write_full(data)
|
|
112
|
+
|
|
113
|
+
def remove_topic(self, name: str) -> bool:
|
|
114
|
+
"""Remove a custom topic. Returns True if removed."""
|
|
115
|
+
data = self._read_full()
|
|
116
|
+
custom = data.get("custom_topics", {})
|
|
117
|
+
if name not in custom:
|
|
118
|
+
return False
|
|
119
|
+
del custom[name]
|
|
120
|
+
data["updated_at"] = _now_iso()
|
|
121
|
+
self._write_full(data)
|
|
122
|
+
return True
|
|
123
|
+
|
|
124
|
+
def derive_topics(self, force: bool = False) -> dict[str, list[str]]:
|
|
125
|
+
"""Derive topics from concept co-occurrence and cache results.
|
|
126
|
+
|
|
127
|
+
Re-derivation is skipped unless *force* is True or the commit/learning
|
|
128
|
+
count signature has changed since last derivation.
|
|
129
|
+
"""
|
|
130
|
+
data = self._read_full()
|
|
131
|
+
current_sig = self._compute_signature()
|
|
132
|
+
|
|
133
|
+
if not force and data.get("derived_signature") == current_sig:
|
|
134
|
+
return data.get("derived_topics", {})
|
|
135
|
+
|
|
136
|
+
derived = self._compute_derived_topics()
|
|
137
|
+
data["derived_topics"] = derived
|
|
138
|
+
data["derived_at"] = _now_iso()
|
|
139
|
+
data["derived_signature"] = current_sig
|
|
140
|
+
data["updated_at"] = _now_iso()
|
|
141
|
+
self._write_full(data)
|
|
142
|
+
return derived
|
|
143
|
+
|
|
144
|
+
def _compute_signature(self) -> str:
|
|
145
|
+
"""Compute a signature based on commit and learning counts."""
|
|
146
|
+
commits_dir = self._gcc_dir / "commits"
|
|
147
|
+
learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
|
|
148
|
+
n_commits = len(list(commits_dir.glob("*.json"))) if commits_dir.exists() else 0
|
|
149
|
+
n_learnings = len(list(learnings_dir.glob("*.json"))) if learnings_dir.exists() else 0
|
|
150
|
+
return f"commits:{n_commits},learnings:{n_learnings}"
|
|
151
|
+
|
|
152
|
+
def _compute_derived_topics(self) -> dict[str, list[str]]:
|
|
153
|
+
"""Compute auto-derived topics using graph-based connected components.
|
|
154
|
+
|
|
155
|
+
Algorithm:
|
|
156
|
+
1. Build a co-occurrence graph from commits and learnings.
|
|
157
|
+
2. Weight edges by Jaccard similarity of co-occurrence.
|
|
158
|
+
3. Remove edges with Jaccard < threshold.
|
|
159
|
+
4. Find connected components.
|
|
160
|
+
5. Discard components with fewer than MIN_COMPONENT_SIZE concepts.
|
|
161
|
+
6. Label each component by its highest-degree concept.
|
|
162
|
+
"""
|
|
163
|
+
co_occurrence: dict[frozenset[str], int] = defaultdict(int)
|
|
164
|
+
occurrence_count: dict[str, int] = defaultdict(int)
|
|
165
|
+
|
|
166
|
+
concept_sets = self._load_concept_sets()
|
|
167
|
+
for concept_list in concept_sets:
|
|
168
|
+
concepts = [c.lower() for c in concept_list]
|
|
169
|
+
for c in concepts:
|
|
170
|
+
occurrence_count[c] += 1
|
|
171
|
+
for pair in combinations(sorted(set(concepts)), 2):
|
|
172
|
+
co_occurrence[frozenset(pair)] += 1
|
|
173
|
+
|
|
174
|
+
if not co_occurrence:
|
|
175
|
+
return {}
|
|
176
|
+
|
|
177
|
+
adj: dict[str, set[str]] = defaultdict(set)
|
|
178
|
+
for pair, co_count in co_occurrence.items():
|
|
179
|
+
c1, c2 = sorted(pair)
|
|
180
|
+
union = occurrence_count[c1] + occurrence_count[c2] - co_count
|
|
181
|
+
jaccard = co_count / union if union > 0 else 0
|
|
182
|
+
if jaccard >= _JACCARD_THRESHOLD:
|
|
183
|
+
adj[c1].add(c2)
|
|
184
|
+
adj[c2].add(c1)
|
|
185
|
+
|
|
186
|
+
components = self._connected_components(adj)
|
|
187
|
+
derived: dict[str, list[str]] = {}
|
|
188
|
+
for comp in components:
|
|
189
|
+
if len(comp) < _MIN_COMPONENT_SIZE:
|
|
190
|
+
continue
|
|
191
|
+
label = max(comp, key=lambda c: len(adj.get(c, set())))
|
|
192
|
+
derived[label] = sorted(comp)
|
|
193
|
+
|
|
194
|
+
return derived
|
|
195
|
+
|
|
196
|
+
def _load_concept_sets(self) -> list[list[str]]:
|
|
197
|
+
"""Load concept sets from commits and learnings."""
|
|
198
|
+
concept_sets: list[list[str]] = []
|
|
199
|
+
|
|
200
|
+
commits_dir = self._gcc_dir / "commits"
|
|
201
|
+
if commits_dir.exists():
|
|
202
|
+
for path in commits_dir.glob("*.json"):
|
|
203
|
+
try:
|
|
204
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
205
|
+
concepts = data.get("concepts", [])
|
|
206
|
+
if concepts:
|
|
207
|
+
concept_sets.append(concepts)
|
|
208
|
+
except (json.JSONDecodeError, OSError):
|
|
209
|
+
continue
|
|
210
|
+
|
|
211
|
+
learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
|
|
212
|
+
if learnings_dir.exists():
|
|
213
|
+
for path in learnings_dir.glob("*.json"):
|
|
214
|
+
try:
|
|
215
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
216
|
+
concepts = data.get("trigger_concepts", [])
|
|
217
|
+
if concepts:
|
|
218
|
+
concept_sets.append(concepts)
|
|
219
|
+
except (json.JSONDecodeError, OSError):
|
|
220
|
+
continue
|
|
221
|
+
|
|
222
|
+
return concept_sets
|
|
223
|
+
|
|
224
|
+
@staticmethod
|
|
225
|
+
def _connected_components(adj: dict[str, set[str]]) -> list[list[str]]:
|
|
226
|
+
"""Find connected components in an undirected graph."""
|
|
227
|
+
visited: set[str] = set()
|
|
228
|
+
components: list[list[str]] = []
|
|
229
|
+
for node in adj:
|
|
230
|
+
if node in visited:
|
|
231
|
+
continue
|
|
232
|
+
stack = [node]
|
|
233
|
+
comp: list[str] = []
|
|
234
|
+
while stack:
|
|
235
|
+
curr = stack.pop()
|
|
236
|
+
if curr in visited:
|
|
237
|
+
continue
|
|
238
|
+
visited.add(curr)
|
|
239
|
+
comp.append(curr)
|
|
240
|
+
stack.extend(adj.get(curr, set()) - visited)
|
|
241
|
+
components.append(comp)
|
|
242
|
+
return components
|
|
243
|
+
|
|
244
|
+
def _topic_has_data(self, concepts: list[str]) -> bool:
|
|
245
|
+
"""Check if at least one concept appears in theta, sensitivity, learnings, or commits."""
|
|
246
|
+
known = self._get_known_concepts()
|
|
247
|
+
return any(c.lower() in known for c in concepts)
|
|
248
|
+
|
|
249
|
+
def _get_known_concepts(self) -> set[str]:
|
|
250
|
+
"""Get the set of all concepts that appear in any data source."""
|
|
251
|
+
known: set[str] = set()
|
|
252
|
+
|
|
253
|
+
theta_path = self._gcc_dir / "theta.json"
|
|
254
|
+
if theta_path.exists():
|
|
255
|
+
try:
|
|
256
|
+
theta = json.loads(theta_path.read_text(encoding="utf-8"))
|
|
257
|
+
known.update(k.lower() for k in theta.get("coordination_vector", {}).keys())
|
|
258
|
+
except (json.JSONDecodeError, OSError):
|
|
259
|
+
pass
|
|
260
|
+
|
|
261
|
+
events_path = self._gcc_dir / "sensitivities" / "events.jsonl"
|
|
262
|
+
if events_path.exists():
|
|
263
|
+
try:
|
|
264
|
+
for line in events_path.read_text(encoding="utf-8").splitlines():
|
|
265
|
+
if not line.strip():
|
|
266
|
+
continue
|
|
267
|
+
ev = json.loads(line)
|
|
268
|
+
payload = ev.get("payload", ev)
|
|
269
|
+
concept = payload.get("target_concept", "")
|
|
270
|
+
if concept:
|
|
271
|
+
known.add(concept.lower())
|
|
272
|
+
except (json.JSONDecodeError, OSError):
|
|
273
|
+
pass
|
|
274
|
+
|
|
275
|
+
learnings_dir = self._gcc_dir / "reasoning_learnings" / "learnings"
|
|
276
|
+
if learnings_dir.exists():
|
|
277
|
+
for path in learnings_dir.glob("*.json"):
|
|
278
|
+
try:
|
|
279
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
280
|
+
for c in data.get("trigger_concepts", []):
|
|
281
|
+
known.add(c.lower())
|
|
282
|
+
except (json.JSONDecodeError, OSError):
|
|
283
|
+
continue
|
|
284
|
+
|
|
285
|
+
commits_dir = self._gcc_dir / "commits"
|
|
286
|
+
if commits_dir.exists():
|
|
287
|
+
for path in commits_dir.glob("*.json"):
|
|
288
|
+
try:
|
|
289
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
290
|
+
for c in data.get("concepts", []):
|
|
291
|
+
known.add(c.lower())
|
|
292
|
+
except (json.JSONDecodeError, OSError):
|
|
293
|
+
continue
|
|
294
|
+
|
|
295
|
+
return known
|
|
296
|
+
|
|
297
|
+
def _read_custom(self) -> dict[str, list[str]]:
|
|
298
|
+
return self._read_full().get("custom_topics", {})
|
|
299
|
+
|
|
300
|
+
def _read_derived(self) -> dict[str, list[str]]:
|
|
301
|
+
return self._read_full().get("derived_topics", {})
|
|
302
|
+
|
|
303
|
+
def _read_full(self) -> dict[str, Any]:
|
|
304
|
+
if not self._path.exists():
|
|
305
|
+
return {"version": "1", "custom_topics": {}, "derived_topics": {}}
|
|
306
|
+
try:
|
|
307
|
+
return json.loads(self._path.read_text(encoding="utf-8"))
|
|
308
|
+
except (json.JSONDecodeError, OSError):
|
|
309
|
+
return {"version": "1", "custom_topics": {}, "derived_topics": {}}
|
|
310
|
+
|
|
311
|
+
def _write_full(self, data: dict[str, Any]) -> None:
|
|
312
|
+
self._gcc_dir.mkdir(parents=True, exist_ok=True)
|
|
313
|
+
self._path.write_text(
|
|
314
|
+
json.dumps(data, indent=2, sort_keys=True) + "\n",
|
|
315
|
+
encoding="utf-8",
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _now_iso() -> str:
|
|
320
|
+
return datetime.now(timezone.utc).isoformat()
|