thinkstack-core 4.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thinkstack_core/__init__.py +158 -0
- thinkstack_core/aggphi_textual.py +275 -0
- thinkstack_core/alerts/__init__.py +23 -0
- thinkstack_core/alerts/base.py +46 -0
- thinkstack_core/alerts/config.py +60 -0
- thinkstack_core/alerts/dispatcher.py +110 -0
- thinkstack_core/alerts/jira.py +96 -0
- thinkstack_core/alerts/linear.py +72 -0
- thinkstack_core/alerts/pagerduty.py +66 -0
- thinkstack_core/alerts/slack.py +81 -0
- thinkstack_core/alerts/teams.py +70 -0
- thinkstack_core/audit/__init__.py +43 -0
- thinkstack_core/audit/exporter.py +297 -0
- thinkstack_core/audit/privacy.py +101 -0
- thinkstack_core/audit/scrubber.py +149 -0
- thinkstack_core/audit/service.py +67 -0
- thinkstack_core/audit/signing.py +127 -0
- thinkstack_core/broadcast/__init__.py +4 -0
- thinkstack_core/broadcast/broadcaster.py +100 -0
- thinkstack_core/broadcast/watcher.py +71 -0
- thinkstack_core/capability.py +639 -0
- thinkstack_core/cloud/__init__.py +1 -0
- thinkstack_core/cloud/client_config.py +472 -0
- thinkstack_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
- thinkstack_core/cloud/client_configs/.claude-stdio.json +13 -0
- thinkstack_core/cloud/client_configs/.cursor-mcp.json +13 -0
- thinkstack_core/cloud/client_configs/.opencode-bridge.json +13 -0
- thinkstack_core/cloud/client_configs/.opencode.json +15 -0
- thinkstack_core/cloud/client_configs/.vscode-mcp.json +13 -0
- thinkstack_core/cloud/mcp_client.py +229 -0
- thinkstack_core/cloud/setup.py +144 -0
- thinkstack_core/cloud/sync.py +143 -0
- thinkstack_core/cloud/sync_bundle.py +639 -0
- thinkstack_core/cloud/sync_conflicts.py +183 -0
- thinkstack_core/cloud/sync_state.py +159 -0
- thinkstack_core/cloud/team_sync.py +337 -0
- thinkstack_core/cloud/thinkstack-mcp-bridge.js +357 -0
- thinkstack_core/codex/__init__.py +9 -0
- thinkstack_core/codex/__main__.py +97 -0
- thinkstack_core/codex/capture.py +208 -0
- thinkstack_core/codex/proxy.py +412 -0
- thinkstack_core/compat.py +103 -0
- thinkstack_core/concept_catalog.py +209 -0
- thinkstack_core/consolidation/__init__.py +3 -0
- thinkstack_core/consolidation/synthesizer.py +87 -0
- thinkstack_core/consolidation/workflow.py +175 -0
- thinkstack_core/daemon/__init__.py +27 -0
- thinkstack_core/daemon/supervisor.py +293 -0
- thinkstack_core/daemon/watcher.py +244 -0
- thinkstack_core/dashboard_api.py +2012 -0
- thinkstack_core/deltaf.py +97 -0
- thinkstack_core/disclosure.py +50 -0
- thinkstack_core/divergence/__init__.py +3 -0
- thinkstack_core/divergence/detector.py +166 -0
- thinkstack_core/gateway/__init__.py +32 -0
- thinkstack_core/gateway/key_manager.py +124 -0
- thinkstack_core/gateway/metrics_webhook.py +252 -0
- thinkstack_core/gateway/policy.py +262 -0
- thinkstack_core/gateway/server.py +727 -0
- thinkstack_core/gateway/sso.py +233 -0
- thinkstack_core/gcc.py +1246 -0
- thinkstack_core/github/__init__.py +35 -0
- thinkstack_core/github/app.py +240 -0
- thinkstack_core/github/comment_builder.py +113 -0
- thinkstack_core/github/pat.py +76 -0
- thinkstack_core/github/pr_parser.py +82 -0
- thinkstack_core/github/pr_reporter.py +555 -0
- thinkstack_core/gitlab/__init__.py +177 -0
- thinkstack_core/hitl/__init__.py +4 -0
- thinkstack_core/hitl/channels.py +129 -0
- thinkstack_core/hitl/orchestrator.py +95 -0
- thinkstack_core/hooks/__init__.py +17 -0
- thinkstack_core/hooks/claude_code.py +228 -0
- thinkstack_core/hooks/git_capture.py +341 -0
- thinkstack_core/hooks/git_commit.py +182 -0
- thinkstack_core/hooks/installer.py +850 -0
- thinkstack_core/hooks/pre_commit.py +157 -0
- thinkstack_core/hooks/runner.py +386 -0
- thinkstack_core/identity/__init__.py +4 -0
- thinkstack_core/identity/agent.py +86 -0
- thinkstack_core/identity/providers.py +85 -0
- thinkstack_core/invariants.py +182 -0
- thinkstack_core/mcp/__init__.py +10 -0
- thinkstack_core/mcp/auth.py +177 -0
- thinkstack_core/mcp/server.py +1215 -0
- thinkstack_core/metrics/__init__.py +35 -0
- thinkstack_core/metrics/aggregate.py +215 -0
- thinkstack_core/metrics/calibrate.py +198 -0
- thinkstack_core/metrics/calibration.py +125 -0
- thinkstack_core/metrics/credibility.py +288 -0
- thinkstack_core/metrics/delivery_time.py +70 -0
- thinkstack_core/metrics/dhs.py +126 -0
- thinkstack_core/metrics/mcs.py +96 -0
- thinkstack_core/metrics/roi.py +88 -0
- thinkstack_core/metrics/session_writer.py +81 -0
- thinkstack_core/metrics/shadow_ai.py +117 -0
- thinkstack_core/metrics/sprint_writer.py +243 -0
- thinkstack_core/observability/__init__.py +78 -0
- thinkstack_core/observability/datadog.py +157 -0
- thinkstack_core/observability/formatter.py +119 -0
- thinkstack_core/observability/report.py +264 -0
- thinkstack_core/observability/servicenow.py +147 -0
- thinkstack_core/observability/splunk.py +218 -0
- thinkstack_core/observability/webhook.py +227 -0
- thinkstack_core/parser/__init__.py +30 -0
- thinkstack_core/parser/blocks.py +216 -0
- thinkstack_core/parser/inference.py +159 -0
- thinkstack_core/parser/thinking.py +112 -0
- thinkstack_core/projects.py +169 -0
- thinkstack_core/prompt_artifact.py +76 -0
- thinkstack_core/proxy/__init__.py +9 -0
- thinkstack_core/proxy/routes/__init__.py +1 -0
- thinkstack_core/proxy/routes/anthropic.py +264 -0
- thinkstack_core/proxy/routes/azure_openai.py +336 -0
- thinkstack_core/proxy/routes/gemini.py +331 -0
- thinkstack_core/proxy/routes/groq.py +284 -0
- thinkstack_core/proxy/routes/ollama.py +279 -0
- thinkstack_core/proxy/routes/openai.py +287 -0
- thinkstack_core/proxy/server.py +356 -0
- thinkstack_core/query/__init__.py +15 -0
- thinkstack_core/query/grep.py +181 -0
- thinkstack_core/query/hybrid.py +86 -0
- thinkstack_core/query/semantic.py +157 -0
- thinkstack_core/rdp.py +105 -0
- thinkstack_core/reasoning/__init__.py +4 -0
- thinkstack_core/reasoning/entry.py +31 -0
- thinkstack_core/reasoning/store.py +122 -0
- thinkstack_core/reasoning_plus/__init__.py +70 -0
- thinkstack_core/reasoning_plus/augmenter.py +337 -0
- thinkstack_core/reasoning_plus/capture.py +51 -0
- thinkstack_core/reasoning_plus/config.py +313 -0
- thinkstack_core/reasoning_plus/context.py +262 -0
- thinkstack_core/reasoning_plus/learning/__init__.py +125 -0
- thinkstack_core/reasoning_plus/learning/analytics.py +141 -0
- thinkstack_core/reasoning_plus/learning/api.py +784 -0
- thinkstack_core/reasoning_plus/learning/chain.py +285 -0
- thinkstack_core/reasoning_plus/learning/composer.py +141 -0
- thinkstack_core/reasoning_plus/learning/conflicts.py +184 -0
- thinkstack_core/reasoning_plus/learning/context_collector.py +194 -0
- thinkstack_core/reasoning_plus/learning/cross_project.py +234 -0
- thinkstack_core/reasoning_plus/learning/deny_list.py +108 -0
- thinkstack_core/reasoning_plus/learning/embeddings.py +209 -0
- thinkstack_core/reasoning_plus/learning/evolution.py +119 -0
- thinkstack_core/reasoning_plus/learning/extractor.py +271 -0
- thinkstack_core/reasoning_plus/learning/filter_five_layer.py +95 -0
- thinkstack_core/reasoning_plus/learning/models.py +149 -0
- thinkstack_core/reasoning_plus/learning/org_store.py +156 -0
- thinkstack_core/reasoning_plus/learning/pii.py +142 -0
- thinkstack_core/reasoning_plus/learning/promotion.py +58 -0
- thinkstack_core/reasoning_plus/learning/provenance.py +126 -0
- thinkstack_core/reasoning_plus/learning/recorder.py +81 -0
- thinkstack_core/reasoning_plus/learning/relevance.py +122 -0
- thinkstack_core/reasoning_plus/learning/state.py +86 -0
- thinkstack_core/reasoning_plus/learning/store.py +178 -0
- thinkstack_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
- thinkstack_core/reasoning_plus/prompt.py +90 -0
- thinkstack_core/rep.py +134 -0
- thinkstack_core/rep_network/__init__.py +25 -0
- thinkstack_core/rep_network/merge.py +70 -0
- thinkstack_core/rep_network/node.py +137 -0
- thinkstack_core/rep_network/server.py +140 -0
- thinkstack_core/rep_network/sync.py +207 -0
- thinkstack_core/sensitivity.py +182 -0
- thinkstack_core/serve.py +258 -0
- thinkstack_core/session/__init__.py +39 -0
- thinkstack_core/session/disagreement.py +188 -0
- thinkstack_core/session/models.py +114 -0
- thinkstack_core/session/orchestrator.py +182 -0
- thinkstack_core/session/planner.py +169 -0
- thinkstack_core/session/simulator.py +132 -0
- thinkstack_core/signing.py +290 -0
- thinkstack_core/sis.py +197 -0
- thinkstack_core/skills/pr-reviewer/SKILL.md +204 -0
- thinkstack_core/skills/thinkstack-auto-sync/SKILL.md +175 -0
- thinkstack_core/skills/thinkstack-session-start/SKILL.md +136 -0
- thinkstack_core/storage.py +308 -0
- thinkstack_core/templates/__init__.py +6 -0
- thinkstack_core/templates/engine.py +122 -0
- thinkstack_core/templates/go.py +18 -0
- thinkstack_core/templates/infra.py +19 -0
- thinkstack_core/templates/library/__init__.py +18 -0
- thinkstack_core/templates/library/api_design.md +27 -0
- thinkstack_core/templates/library/bug_fix.md +27 -0
- thinkstack_core/templates/library/decision_record.md +27 -0
- thinkstack_core/templates/library/engine.py +228 -0
- thinkstack_core/templates/library/security_review.md +30 -0
- thinkstack_core/templates/python.py +19 -0
- thinkstack_core/templates/react.py +18 -0
- thinkstack_core/templates/typescript.py +18 -0
- thinkstack_core/theta.py +221 -0
- thinkstack_core/theta_synthesis.py +268 -0
- thinkstack_core/topics.py +320 -0
- thinkstack_core/variance.py +219 -0
- thinkstack_core/wrapper/__init__.py +52 -0
- thinkstack_core/wrapper/anthropic.py +487 -0
- thinkstack_core/wrapper/base.py +562 -0
- thinkstack_core/wrapper/bedrock.py +342 -0
- thinkstack_core/wrapper/gemini.py +422 -0
- thinkstack_core/wrapper/ollama.py +527 -0
- thinkstack_core/wrapper/openai.py +461 -0
- thinkstack_core-4.0.0.dist-info/METADATA +868 -0
- thinkstack_core-4.0.0.dist-info/RECORD +205 -0
- thinkstack_core-4.0.0.dist-info/WHEEL +5 -0
- thinkstack_core-4.0.0.dist-info/entry_points.txt +2 -0
- thinkstack_core-4.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.context_collector
|
|
3
|
+
=======================================================
|
|
4
|
+
Collect enriched extraction context from a ReasoningCall and workspace.
|
|
5
|
+
|
|
6
|
+
Builds an ExtractionContext that contains not just the reasoning text
|
|
7
|
+
but also file paths, import changes, and library references — enabling
|
|
8
|
+
the LLM extractor to produce concrete, actionable learnings like
|
|
9
|
+
"use date-fns v3 instead of moment.js".
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from thinkstack_core.reasoning_plus.learning.models import ReasoningCall
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# Patterns for detecting import statements in text
|
|
22
|
+
_IMPORT_PATTERN = re.compile(
|
|
23
|
+
r'(?:^|[;.,:\s])(?:import\s+(?:(\w+(?:\.\w+)*)(?:\s+as\s+\w+)?)'
|
|
24
|
+
r'|from\s+(\w+(?:\.\w+)*)\s+import\s+[\w,\s*]+(?:as\s+\w+)?)',
|
|
25
|
+
re.MULTILINE,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Patterns for detecting file paths (common extensions)
|
|
29
|
+
_FILE_EXTENSIONS = {
|
|
30
|
+
".py", ".js", ".ts", ".tsx", ".jsx", ".java", ".kt", ".go",
|
|
31
|
+
".rs", ".cpp", ".c", ".h", ".hpp", ".css", ".scss", ".html",
|
|
32
|
+
".vue", ".svelte", ".rb", ".php", ".swift", ".m", ".mm",
|
|
33
|
+
".yaml", ".yml", ".json", ".toml", ".ini", ".cfg", ".conf",
|
|
34
|
+
".md", ".rst", ".txt", ".sh", ".bash", ".zsh", ".ps1",
|
|
35
|
+
".sql", ".graphql", ".proto", ".gradle", ".lock",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
# Patterns for dependency references
|
|
39
|
+
_PIP_PACKAGE_PATTERN = re.compile(
|
|
40
|
+
r'(?:pip\s+install\s+)([\w\-.\[\]]+)',
|
|
41
|
+
re.IGNORECASE,
|
|
42
|
+
)
|
|
43
|
+
_NPM_PACKAGE_PATTERN = re.compile(
|
|
44
|
+
r'(?:npm\s+(?:install|i|add)\s+)([\w@./\-]+)',
|
|
45
|
+
re.IGNORECASE,
|
|
46
|
+
)
|
|
47
|
+
_CARGO_PACKAGE_PATTERN = re.compile(
|
|
48
|
+
r'(?:cargo\s+add\s+)([\w\-]+)',
|
|
49
|
+
re.IGNORECASE,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class ExtractionContext:
|
|
55
|
+
"""Enriched context for learning extraction."""
|
|
56
|
+
|
|
57
|
+
reasoning_text: str = ""
|
|
58
|
+
files_changed: list[str] = field(default_factory=list)
|
|
59
|
+
imports_added: list[str] = field(default_factory=list)
|
|
60
|
+
imports_removed: list[str] = field(default_factory=list)
|
|
61
|
+
libraries_used: list[str] = field(default_factory=list)
|
|
62
|
+
call_type: str = "llm"
|
|
63
|
+
outcome: str = "unknown"
|
|
64
|
+
|
|
65
|
+
def is_significant(self) -> bool:
|
|
66
|
+
"""Return True if this context contains enough signal to warrant extraction."""
|
|
67
|
+
if self.outcome in ("failure", "partial"):
|
|
68
|
+
return True
|
|
69
|
+
if self.files_changed or self.imports_added or self.imports_removed:
|
|
70
|
+
return True
|
|
71
|
+
if self.libraries_used:
|
|
72
|
+
return True
|
|
73
|
+
return False
|
|
74
|
+
|
|
75
|
+
def to_prompt_context(self) -> str:
|
|
76
|
+
"""Format this context as a compact block for the extraction LLM prompt."""
|
|
77
|
+
parts: list[str] = []
|
|
78
|
+
|
|
79
|
+
if self.files_changed:
|
|
80
|
+
parts.append(f"Files changed/suggested: {', '.join(sorted(self.files_changed))}")
|
|
81
|
+
|
|
82
|
+
if self.imports_added:
|
|
83
|
+
parts.append(f"New imports: {', '.join(sorted(self.imports_added))}")
|
|
84
|
+
|
|
85
|
+
if self.imports_removed:
|
|
86
|
+
parts.append(f"Removed imports: {', '.join(sorted(self.imports_removed))}")
|
|
87
|
+
|
|
88
|
+
if self.libraries_used:
|
|
89
|
+
parts.append(f"Libraries referenced: {', '.join(sorted(self.libraries_used))}")
|
|
90
|
+
|
|
91
|
+
parts.append(f"Outcome: {self.outcome}")
|
|
92
|
+
parts.append(f"Call type: {self.call_type}")
|
|
93
|
+
|
|
94
|
+
return "\n".join(parts)
|
|
95
|
+
|
|
96
|
+
def to_dict(self) -> dict[str, Any]:
|
|
97
|
+
return {
|
|
98
|
+
"reasoning_text": self.reasoning_text,
|
|
99
|
+
"files_changed": list(self.files_changed),
|
|
100
|
+
"imports_added": list(self.imports_added),
|
|
101
|
+
"imports_removed": list(self.imports_removed),
|
|
102
|
+
"libraries_used": list(self.libraries_used),
|
|
103
|
+
"call_type": self.call_type,
|
|
104
|
+
"outcome": self.outcome,
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def collect_extraction_context(
|
|
109
|
+
call: ReasoningCall,
|
|
110
|
+
gcc_dir: Path | None = None,
|
|
111
|
+
) -> ExtractionContext:
|
|
112
|
+
"""Build an ExtractionContext from a ReasoningCall and optional workspace.
|
|
113
|
+
|
|
114
|
+
The collector parses the call's reasoning, input, and output text for:
|
|
115
|
+
- File paths mentioned
|
|
116
|
+
- Import statements
|
|
117
|
+
- Package/dependency references
|
|
118
|
+
|
|
119
|
+
If a gcc_dir is provided, it may use workspace metadata to enrich context
|
|
120
|
+
(e.g., reading current file content for diff analysis).
|
|
121
|
+
"""
|
|
122
|
+
combined = f"{call.reasoning} {call.input} {call.output}"
|
|
123
|
+
|
|
124
|
+
files_changed = _extract_file_paths(call)
|
|
125
|
+
imports_added = _extract_imports(combined)
|
|
126
|
+
# TODO S19: Populate imports_removed by comparing file imports before/after the call
|
|
127
|
+
imports_removed: list[str] = []
|
|
128
|
+
libraries_used = _extract_libraries(combined)
|
|
129
|
+
|
|
130
|
+
ctx = ExtractionContext(
|
|
131
|
+
reasoning_text=call.reasoning,
|
|
132
|
+
files_changed=files_changed,
|
|
133
|
+
imports_added=imports_added,
|
|
134
|
+
imports_removed=imports_removed,
|
|
135
|
+
libraries_used=libraries_used,
|
|
136
|
+
call_type=call.call_type,
|
|
137
|
+
outcome=call.outcome,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
if gcc_dir and files_changed:
|
|
141
|
+
_enrich_from_workspace(ctx, gcc_dir)
|
|
142
|
+
|
|
143
|
+
return ctx
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _extract_file_paths(call: ReasoningCall) -> list[str]:
|
|
147
|
+
"""Extract file paths from the call's reasoning and output text."""
|
|
148
|
+
paths: set[str] = set()
|
|
149
|
+
|
|
150
|
+
texts = [call.reasoning, call.output, call.input, call.name]
|
|
151
|
+
for text in texts:
|
|
152
|
+
if not isinstance(text, str):
|
|
153
|
+
continue
|
|
154
|
+
# Split on whitespace and check each word for known file extensions
|
|
155
|
+
for word in text.replace("/", " / ").split():
|
|
156
|
+
word = word.strip(".,;:!?\"'()[]{}<>")
|
|
157
|
+
ext = Path(word).suffix.lower()
|
|
158
|
+
if ext in _FILE_EXTENSIONS:
|
|
159
|
+
paths.add(word)
|
|
160
|
+
|
|
161
|
+
return sorted(paths)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _extract_imports(text: str) -> list[str]:
|
|
165
|
+
"""Extract import statements from text (pip-style Python imports)."""
|
|
166
|
+
imports: set[str] = set()
|
|
167
|
+
|
|
168
|
+
for match in _IMPORT_PATTERN.finditer(text):
|
|
169
|
+
if match.group(1):
|
|
170
|
+
imports.add(match.group(1).split(".")[0])
|
|
171
|
+
if match.group(2):
|
|
172
|
+
imports.add(match.group(2).split(".")[0])
|
|
173
|
+
|
|
174
|
+
return sorted(imports)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _extract_libraries(text: str) -> list[str]:
|
|
178
|
+
"""Extract library/package references from dependency install commands."""
|
|
179
|
+
libs: set[str] = set()
|
|
180
|
+
|
|
181
|
+
for pattern in [_PIP_PACKAGE_PATTERN, _NPM_PACKAGE_PATTERN, _CARGO_PACKAGE_PATTERN]:
|
|
182
|
+
for match in pattern.finditer(text):
|
|
183
|
+
libs.add(match.group(1).lower())
|
|
184
|
+
|
|
185
|
+
return sorted(libs)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _enrich_from_workspace(ctx: ExtractionContext, gcc_dir: Path) -> None:
|
|
189
|
+
"""Enrich context with workspace analysis.
|
|
190
|
+
|
|
191
|
+
# TODO S19+: Diff changed files against git HEAD to detect import
|
|
192
|
+
# adds/removes and library changes. This is deferred to keep
|
|
193
|
+
# extraction cost zero for the common case.
|
|
194
|
+
"""
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.cross_project
|
|
3
|
+
====================================================
|
|
4
|
+
Cross-project learning suggestions.
|
|
5
|
+
|
|
6
|
+
Maintains a machine-level registry of governed projects at
|
|
7
|
+
``~/.thinkstack/registry.json`` under the ``governed_projects`` key and lets a
|
|
8
|
+
project surface relevant learnings from sibling projects with matching tags or
|
|
9
|
+
names.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
from dataclasses import asdict, dataclass
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
from thinkstack_core.reasoning_plus.learning.models import Learning
|
|
22
|
+
from thinkstack_core.reasoning_plus.learning.relevance import RelevanceEngine
|
|
23
|
+
from thinkstack_core.reasoning_plus.learning.store import LearningStore
|
|
24
|
+
from thinkstack_core.reasoning_plus.context import extract_keywords
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning.cross_project")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
DEFAULT_REGISTRY_DIR = Path.home() / ".thinkstack"
|
|
30
|
+
DEFAULT_REGISTRY_FILE = "registry.json"
|
|
31
|
+
REGISTRY_PATH_ENV = "THINKSTACK_REGISTRY_PATH"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class GovernedProject:
|
|
36
|
+
"""One governed project entry in the cross-project learning registry."""
|
|
37
|
+
|
|
38
|
+
project_root: str
|
|
39
|
+
name: str
|
|
40
|
+
tags: list[str]
|
|
41
|
+
gcc_dir: str
|
|
42
|
+
added_at: str
|
|
43
|
+
updated_at: str
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class CrossProjectRegistry:
|
|
47
|
+
"""Machine-level registry of governed projects used for cross-project learning.
|
|
48
|
+
|
|
49
|
+
Stored under the ``governed_projects`` key in ``~/.thinkstack/registry.json``
|
|
50
|
+
so it coexists with the Sprint 19 project registry (which uses the
|
|
51
|
+
``projects`` key).
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
def __init__(self, registry_path: Path | str | None = None) -> None:
|
|
55
|
+
if registry_path is None:
|
|
56
|
+
env_path = os.environ.get(REGISTRY_PATH_ENV, "").strip()
|
|
57
|
+
if env_path:
|
|
58
|
+
registry_path = Path(env_path)
|
|
59
|
+
else:
|
|
60
|
+
registry_path = DEFAULT_REGISTRY_DIR / DEFAULT_REGISTRY_FILE
|
|
61
|
+
self.registry_path = Path(registry_path)
|
|
62
|
+
self._ensure_dir()
|
|
63
|
+
|
|
64
|
+
def _ensure_dir(self) -> None:
|
|
65
|
+
self.registry_path.parent.mkdir(parents=True, exist_ok=True)
|
|
66
|
+
|
|
67
|
+
def _read(self) -> dict[str, Any]:
|
|
68
|
+
if not self.registry_path.exists():
|
|
69
|
+
return {"version": "1", "projects": [], "governed_projects": []}
|
|
70
|
+
try:
|
|
71
|
+
data = json.loads(self.registry_path.read_text(encoding="utf-8"))
|
|
72
|
+
if not isinstance(data, dict):
|
|
73
|
+
return {"version": "1", "projects": [], "governed_projects": []}
|
|
74
|
+
return data
|
|
75
|
+
except (json.JSONDecodeError, OSError):
|
|
76
|
+
return {"version": "1", "projects": [], "governed_projects": []}
|
|
77
|
+
|
|
78
|
+
def _write(self, data: dict[str, Any]) -> None:
|
|
79
|
+
self._ensure_dir()
|
|
80
|
+
self.registry_path.write_text(
|
|
81
|
+
json.dumps(data, indent=2, sort_keys=True), encoding="utf-8"
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
def list(self) -> list[GovernedProject]:
|
|
85
|
+
"""Return all registered governed projects."""
|
|
86
|
+
data = self._read()
|
|
87
|
+
entries: list[GovernedProject] = []
|
|
88
|
+
for raw in data.get("governed_projects", []):
|
|
89
|
+
try:
|
|
90
|
+
entries.append(GovernedProject(**raw))
|
|
91
|
+
except (TypeError, ValueError):
|
|
92
|
+
logger.debug("thinkstack: skipping malformed governed project entry: %s", raw)
|
|
93
|
+
continue
|
|
94
|
+
return entries
|
|
95
|
+
|
|
96
|
+
def add(self, gcc_dir: Path | str, name: str, tags: list[str] | None = None) -> GovernedProject:
|
|
97
|
+
"""Add or update a governed project entry keyed by gcc_dir."""
|
|
98
|
+
gcc_dir = Path(gcc_dir).resolve()
|
|
99
|
+
project_root = gcc_dir.parent
|
|
100
|
+
name = name or project_root.name
|
|
101
|
+
tags = sorted(set((tags or [])))
|
|
102
|
+
|
|
103
|
+
data = self._read()
|
|
104
|
+
projects = data.setdefault("governed_projects", [])
|
|
105
|
+
|
|
106
|
+
resolved_root = str(project_root)
|
|
107
|
+
for raw in projects:
|
|
108
|
+
if Path(raw.get("gcc_dir", "")).resolve() == gcc_dir:
|
|
109
|
+
raw["project_root"] = resolved_root
|
|
110
|
+
raw["name"] = name
|
|
111
|
+
raw["tags"] = tags
|
|
112
|
+
raw["updated_at"] = _now_iso()
|
|
113
|
+
self._write(data)
|
|
114
|
+
return GovernedProject(**raw)
|
|
115
|
+
|
|
116
|
+
entry = GovernedProject(
|
|
117
|
+
project_root=resolved_root,
|
|
118
|
+
name=name,
|
|
119
|
+
tags=tags,
|
|
120
|
+
gcc_dir=str(gcc_dir),
|
|
121
|
+
added_at=_now_iso(),
|
|
122
|
+
updated_at=_now_iso(),
|
|
123
|
+
)
|
|
124
|
+
projects.append(asdict(entry))
|
|
125
|
+
self._write(data)
|
|
126
|
+
return entry
|
|
127
|
+
|
|
128
|
+
def remove(self, gcc_dir: Path | str) -> bool:
|
|
129
|
+
"""Remove a governed project by gcc_dir. Returns True if removed."""
|
|
130
|
+
gcc_dir = Path(gcc_dir).resolve()
|
|
131
|
+
data = self._read()
|
|
132
|
+
projects = data.get("governed_projects", [])
|
|
133
|
+
original_len = len(projects)
|
|
134
|
+
projects = [
|
|
135
|
+
raw for raw in projects
|
|
136
|
+
if Path(raw.get("gcc_dir", "")).resolve() != gcc_dir
|
|
137
|
+
]
|
|
138
|
+
if len(projects) == original_len:
|
|
139
|
+
return False
|
|
140
|
+
data["governed_projects"] = projects
|
|
141
|
+
self._write(data)
|
|
142
|
+
return True
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _registry() -> CrossProjectRegistry:
|
|
146
|
+
return CrossProjectRegistry()
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def register_project(gcc_dir: Path | str, name: str, tags: list[str] | None = None) -> GovernedProject:
|
|
150
|
+
"""Add or update the current project in the governed project registry."""
|
|
151
|
+
return _registry().add(gcc_dir, name, tags)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def list_projects() -> list[GovernedProject]:
|
|
155
|
+
"""Return all registered governed projects."""
|
|
156
|
+
return _registry().list()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _project_match_score(query: str, project: GovernedProject) -> float:
|
|
160
|
+
"""Keyword overlap score between the query and project name/tags."""
|
|
161
|
+
query_tokens = set(extract_keywords(query, max_keywords=20))
|
|
162
|
+
if not query_tokens:
|
|
163
|
+
return 0.0
|
|
164
|
+
|
|
165
|
+
project_tokens: set[str] = set()
|
|
166
|
+
project_tokens.update(extract_keywords(project.name, max_keywords=20))
|
|
167
|
+
for tag in project.tags:
|
|
168
|
+
project_tokens.update(extract_keywords(tag, max_keywords=20))
|
|
169
|
+
project_tokens.add(tag.lower())
|
|
170
|
+
|
|
171
|
+
if not project_tokens:
|
|
172
|
+
return 0.0
|
|
173
|
+
|
|
174
|
+
intersection = query_tokens & project_tokens
|
|
175
|
+
return len(intersection) / max(len(query_tokens), len(project_tokens))
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _load_active_learnings(gcc_dir: Path | str) -> list[Learning]:
|
|
179
|
+
"""Load active learnings from a project's .GCC directory."""
|
|
180
|
+
try:
|
|
181
|
+
store = LearningStore(gcc_dir)
|
|
182
|
+
return store.list(validity="active")
|
|
183
|
+
except Exception as exc:
|
|
184
|
+
logger.warning("thinkstack: failed to load learnings from %s — %s", gcc_dir, exc)
|
|
185
|
+
return []
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def cross_project_suggest(
|
|
189
|
+
query: str,
|
|
190
|
+
current_project_root: Path | str,
|
|
191
|
+
top_n: int = 3,
|
|
192
|
+
project_match_threshold: float = 0.1,
|
|
193
|
+
) -> list[Learning]:
|
|
194
|
+
"""Find relevant learnings from sibling governed projects.
|
|
195
|
+
|
|
196
|
+
Projects are matched by keyword overlap with the query using their name and
|
|
197
|
+
tags. Active learnings from matched projects are loaded via ``LearningStore``
|
|
198
|
+
and ranked with the keyword-based ``RelevanceEngine``. The current project is
|
|
199
|
+
excluded so suggestions come from other governed projects.
|
|
200
|
+
"""
|
|
201
|
+
current_project_root = Path(current_project_root).resolve()
|
|
202
|
+
registry = _registry()
|
|
203
|
+
projects = registry.list()
|
|
204
|
+
|
|
205
|
+
# Exclude current project to avoid suggesting its own learnings back to it.
|
|
206
|
+
sibling_projects = [
|
|
207
|
+
p for p in projects
|
|
208
|
+
if Path(p.project_root).resolve() != current_project_root
|
|
209
|
+
]
|
|
210
|
+
|
|
211
|
+
# Match projects by name/tags.
|
|
212
|
+
scored_projects = [
|
|
213
|
+
(score, p)
|
|
214
|
+
for p in sibling_projects
|
|
215
|
+
if (score := _project_match_score(query, p)) >= project_match_threshold
|
|
216
|
+
]
|
|
217
|
+
scored_projects.sort(key=lambda x: x[0], reverse=True)
|
|
218
|
+
|
|
219
|
+
# Collect active learnings from all matched projects.
|
|
220
|
+
all_learnings: list[Learning] = []
|
|
221
|
+
for _, project in scored_projects:
|
|
222
|
+
learnings = _load_active_learnings(project.gcc_dir)
|
|
223
|
+
all_learnings.extend(learnings)
|
|
224
|
+
|
|
225
|
+
if not all_learnings:
|
|
226
|
+
return []
|
|
227
|
+
|
|
228
|
+
# Rank across projects using the same keyword backend used by the facade.
|
|
229
|
+
engine = RelevanceEngine()
|
|
230
|
+
return engine.rank(query, all_learnings, top_n=top_n)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _now_iso() -> str:
|
|
234
|
+
return datetime.now(timezone.utc).isoformat()
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.deny_list
|
|
3
|
+
=================================================
|
|
4
|
+
Per-project learning deny lists.
|
|
5
|
+
|
|
6
|
+
Each project can maintain a ``.GCC/learning_config.json`` file that
|
|
7
|
+
lists concepts and learning IDs to exclude from cross-project sharing
|
|
8
|
+
and injection. The deny list is consulted by the five-layer filter
|
|
9
|
+
and by the org-store aggregation stage.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import logging
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class DenyList:
|
|
23
|
+
"""Exclusion rules for cross-project learning."""
|
|
24
|
+
|
|
25
|
+
excluded_concepts: list[str] = field(default_factory=list)
|
|
26
|
+
excluded_learnings: list[str] = field(default_factory=list)
|
|
27
|
+
project_root: str | None = None
|
|
28
|
+
|
|
29
|
+
def add_concept(self, concept: str) -> None:
|
|
30
|
+
if concept not in self.excluded_concepts:
|
|
31
|
+
self.excluded_concepts.append(concept)
|
|
32
|
+
|
|
33
|
+
def remove_concept(self, concept: str) -> bool:
|
|
34
|
+
lowered = [c.lower() for c in self.excluded_concepts]
|
|
35
|
+
try:
|
|
36
|
+
idx = lowered.index(concept.lower())
|
|
37
|
+
self.excluded_concepts.pop(idx)
|
|
38
|
+
return True
|
|
39
|
+
except ValueError:
|
|
40
|
+
return False
|
|
41
|
+
|
|
42
|
+
def add_learning(self, learning_id: str) -> None:
|
|
43
|
+
if learning_id not in self.excluded_learnings:
|
|
44
|
+
self.excluded_learnings.append(learning_id)
|
|
45
|
+
|
|
46
|
+
def remove_learning(self, learning_id: str) -> bool:
|
|
47
|
+
if learning_id in self.excluded_learnings:
|
|
48
|
+
self.excluded_learnings.remove(learning_id)
|
|
49
|
+
return True
|
|
50
|
+
return False
|
|
51
|
+
|
|
52
|
+
def to_dict(self) -> dict:
|
|
53
|
+
return {
|
|
54
|
+
"excluded_concepts": self.excluded_concepts,
|
|
55
|
+
"excluded_learnings": self.excluded_learnings,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def load_deny_list(gcc_dir_or_project_root: Path | str) -> DenyList:
|
|
60
|
+
"""Load the deny list from a project's ``.GCC/learning_config.json``."""
|
|
61
|
+
project_root = Path(gcc_dir_or_project_root)
|
|
62
|
+
if project_root.name == ".GCC":
|
|
63
|
+
config_path = project_root / "learning_config.json"
|
|
64
|
+
else:
|
|
65
|
+
config_path = project_root / ".GCC" / "learning_config.json"
|
|
66
|
+
if not config_path.exists():
|
|
67
|
+
return DenyList(project_root=str(project_root))
|
|
68
|
+
try:
|
|
69
|
+
data = json.loads(config_path.read_text(encoding="utf-8"))
|
|
70
|
+
exclude = data.get("exclude", {})
|
|
71
|
+
return DenyList(
|
|
72
|
+
excluded_concepts=exclude.get("excluded_concepts", exclude.get("concepts", [])),
|
|
73
|
+
excluded_learnings=exclude.get("excluded_learnings", exclude.get("learnings", [])),
|
|
74
|
+
project_root=str(project_root),
|
|
75
|
+
)
|
|
76
|
+
except Exception as exc:
|
|
77
|
+
logger.warning("thinkstack: failed to read deny list from %s — %s", config_path, exc)
|
|
78
|
+
return DenyList(project_root=str(project_root))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def save_deny_list(deny_list: DenyList, gcc_dir_or_project_root: Path | str) -> None:
|
|
82
|
+
"""Persist a deny list to ``.GCC/learning_config.json``."""
|
|
83
|
+
project_root = Path(gcc_dir_or_project_root)
|
|
84
|
+
if project_root.name == ".GCC":
|
|
85
|
+
config_path = project_root / "learning_config.json"
|
|
86
|
+
gcc_dir = project_root
|
|
87
|
+
else:
|
|
88
|
+
config_path = project_root / ".GCC" / "learning_config.json"
|
|
89
|
+
gcc_dir = project_root / ".GCC"
|
|
90
|
+
gcc_dir.mkdir(parents=True, exist_ok=True)
|
|
91
|
+
data = {
|
|
92
|
+
"exclude": deny_list.to_dict(),
|
|
93
|
+
"version": 1,
|
|
94
|
+
}
|
|
95
|
+
tmp = config_path.with_suffix(".tmp")
|
|
96
|
+
try:
|
|
97
|
+
with open(tmp, "w", encoding="utf-8") as f:
|
|
98
|
+
json.dump(data, f, indent=2)
|
|
99
|
+
tmp.replace(config_path)
|
|
100
|
+
except Exception as exc:
|
|
101
|
+
logger.warning("thinkstack: failed to save deny list to %s — %s", config_path, exc)
|
|
102
|
+
raise
|
|
103
|
+
finally:
|
|
104
|
+
if tmp.exists():
|
|
105
|
+
try:
|
|
106
|
+
tmp.unlink()
|
|
107
|
+
except Exception:
|
|
108
|
+
pass
|