thinkstack-core 4.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thinkstack_core/__init__.py +158 -0
- thinkstack_core/aggphi_textual.py +275 -0
- thinkstack_core/alerts/__init__.py +23 -0
- thinkstack_core/alerts/base.py +46 -0
- thinkstack_core/alerts/config.py +60 -0
- thinkstack_core/alerts/dispatcher.py +110 -0
- thinkstack_core/alerts/jira.py +96 -0
- thinkstack_core/alerts/linear.py +72 -0
- thinkstack_core/alerts/pagerduty.py +66 -0
- thinkstack_core/alerts/slack.py +81 -0
- thinkstack_core/alerts/teams.py +70 -0
- thinkstack_core/audit/__init__.py +43 -0
- thinkstack_core/audit/exporter.py +297 -0
- thinkstack_core/audit/privacy.py +101 -0
- thinkstack_core/audit/scrubber.py +149 -0
- thinkstack_core/audit/service.py +67 -0
- thinkstack_core/audit/signing.py +127 -0
- thinkstack_core/broadcast/__init__.py +4 -0
- thinkstack_core/broadcast/broadcaster.py +100 -0
- thinkstack_core/broadcast/watcher.py +71 -0
- thinkstack_core/capability.py +639 -0
- thinkstack_core/cloud/__init__.py +1 -0
- thinkstack_core/cloud/client_config.py +472 -0
- thinkstack_core/cloud/client_configs/.claude-opencode-fallback.json +8 -0
- thinkstack_core/cloud/client_configs/.claude-stdio.json +13 -0
- thinkstack_core/cloud/client_configs/.cursor-mcp.json +13 -0
- thinkstack_core/cloud/client_configs/.opencode-bridge.json +13 -0
- thinkstack_core/cloud/client_configs/.opencode.json +15 -0
- thinkstack_core/cloud/client_configs/.vscode-mcp.json +13 -0
- thinkstack_core/cloud/mcp_client.py +229 -0
- thinkstack_core/cloud/setup.py +144 -0
- thinkstack_core/cloud/sync.py +143 -0
- thinkstack_core/cloud/sync_bundle.py +639 -0
- thinkstack_core/cloud/sync_conflicts.py +183 -0
- thinkstack_core/cloud/sync_state.py +159 -0
- thinkstack_core/cloud/team_sync.py +337 -0
- thinkstack_core/cloud/thinkstack-mcp-bridge.js +357 -0
- thinkstack_core/codex/__init__.py +9 -0
- thinkstack_core/codex/__main__.py +97 -0
- thinkstack_core/codex/capture.py +208 -0
- thinkstack_core/codex/proxy.py +412 -0
- thinkstack_core/compat.py +103 -0
- thinkstack_core/concept_catalog.py +209 -0
- thinkstack_core/consolidation/__init__.py +3 -0
- thinkstack_core/consolidation/synthesizer.py +87 -0
- thinkstack_core/consolidation/workflow.py +175 -0
- thinkstack_core/daemon/__init__.py +27 -0
- thinkstack_core/daemon/supervisor.py +293 -0
- thinkstack_core/daemon/watcher.py +244 -0
- thinkstack_core/dashboard_api.py +2012 -0
- thinkstack_core/deltaf.py +97 -0
- thinkstack_core/disclosure.py +50 -0
- thinkstack_core/divergence/__init__.py +3 -0
- thinkstack_core/divergence/detector.py +166 -0
- thinkstack_core/gateway/__init__.py +32 -0
- thinkstack_core/gateway/key_manager.py +124 -0
- thinkstack_core/gateway/metrics_webhook.py +252 -0
- thinkstack_core/gateway/policy.py +262 -0
- thinkstack_core/gateway/server.py +727 -0
- thinkstack_core/gateway/sso.py +233 -0
- thinkstack_core/gcc.py +1246 -0
- thinkstack_core/github/__init__.py +35 -0
- thinkstack_core/github/app.py +240 -0
- thinkstack_core/github/comment_builder.py +113 -0
- thinkstack_core/github/pat.py +76 -0
- thinkstack_core/github/pr_parser.py +82 -0
- thinkstack_core/github/pr_reporter.py +555 -0
- thinkstack_core/gitlab/__init__.py +177 -0
- thinkstack_core/hitl/__init__.py +4 -0
- thinkstack_core/hitl/channels.py +129 -0
- thinkstack_core/hitl/orchestrator.py +95 -0
- thinkstack_core/hooks/__init__.py +17 -0
- thinkstack_core/hooks/claude_code.py +228 -0
- thinkstack_core/hooks/git_capture.py +341 -0
- thinkstack_core/hooks/git_commit.py +182 -0
- thinkstack_core/hooks/installer.py +850 -0
- thinkstack_core/hooks/pre_commit.py +157 -0
- thinkstack_core/hooks/runner.py +386 -0
- thinkstack_core/identity/__init__.py +4 -0
- thinkstack_core/identity/agent.py +86 -0
- thinkstack_core/identity/providers.py +85 -0
- thinkstack_core/invariants.py +182 -0
- thinkstack_core/mcp/__init__.py +10 -0
- thinkstack_core/mcp/auth.py +177 -0
- thinkstack_core/mcp/server.py +1215 -0
- thinkstack_core/metrics/__init__.py +35 -0
- thinkstack_core/metrics/aggregate.py +215 -0
- thinkstack_core/metrics/calibrate.py +198 -0
- thinkstack_core/metrics/calibration.py +125 -0
- thinkstack_core/metrics/credibility.py +288 -0
- thinkstack_core/metrics/delivery_time.py +70 -0
- thinkstack_core/metrics/dhs.py +126 -0
- thinkstack_core/metrics/mcs.py +96 -0
- thinkstack_core/metrics/roi.py +88 -0
- thinkstack_core/metrics/session_writer.py +81 -0
- thinkstack_core/metrics/shadow_ai.py +117 -0
- thinkstack_core/metrics/sprint_writer.py +243 -0
- thinkstack_core/observability/__init__.py +78 -0
- thinkstack_core/observability/datadog.py +157 -0
- thinkstack_core/observability/formatter.py +119 -0
- thinkstack_core/observability/report.py +264 -0
- thinkstack_core/observability/servicenow.py +147 -0
- thinkstack_core/observability/splunk.py +218 -0
- thinkstack_core/observability/webhook.py +227 -0
- thinkstack_core/parser/__init__.py +30 -0
- thinkstack_core/parser/blocks.py +216 -0
- thinkstack_core/parser/inference.py +159 -0
- thinkstack_core/parser/thinking.py +112 -0
- thinkstack_core/projects.py +169 -0
- thinkstack_core/prompt_artifact.py +76 -0
- thinkstack_core/proxy/__init__.py +9 -0
- thinkstack_core/proxy/routes/__init__.py +1 -0
- thinkstack_core/proxy/routes/anthropic.py +264 -0
- thinkstack_core/proxy/routes/azure_openai.py +336 -0
- thinkstack_core/proxy/routes/gemini.py +331 -0
- thinkstack_core/proxy/routes/groq.py +284 -0
- thinkstack_core/proxy/routes/ollama.py +279 -0
- thinkstack_core/proxy/routes/openai.py +287 -0
- thinkstack_core/proxy/server.py +356 -0
- thinkstack_core/query/__init__.py +15 -0
- thinkstack_core/query/grep.py +181 -0
- thinkstack_core/query/hybrid.py +86 -0
- thinkstack_core/query/semantic.py +157 -0
- thinkstack_core/rdp.py +105 -0
- thinkstack_core/reasoning/__init__.py +4 -0
- thinkstack_core/reasoning/entry.py +31 -0
- thinkstack_core/reasoning/store.py +122 -0
- thinkstack_core/reasoning_plus/__init__.py +70 -0
- thinkstack_core/reasoning_plus/augmenter.py +337 -0
- thinkstack_core/reasoning_plus/capture.py +51 -0
- thinkstack_core/reasoning_plus/config.py +313 -0
- thinkstack_core/reasoning_plus/context.py +262 -0
- thinkstack_core/reasoning_plus/learning/__init__.py +125 -0
- thinkstack_core/reasoning_plus/learning/analytics.py +141 -0
- thinkstack_core/reasoning_plus/learning/api.py +784 -0
- thinkstack_core/reasoning_plus/learning/chain.py +285 -0
- thinkstack_core/reasoning_plus/learning/composer.py +141 -0
- thinkstack_core/reasoning_plus/learning/conflicts.py +184 -0
- thinkstack_core/reasoning_plus/learning/context_collector.py +194 -0
- thinkstack_core/reasoning_plus/learning/cross_project.py +234 -0
- thinkstack_core/reasoning_plus/learning/deny_list.py +108 -0
- thinkstack_core/reasoning_plus/learning/embeddings.py +209 -0
- thinkstack_core/reasoning_plus/learning/evolution.py +119 -0
- thinkstack_core/reasoning_plus/learning/extractor.py +271 -0
- thinkstack_core/reasoning_plus/learning/filter_five_layer.py +95 -0
- thinkstack_core/reasoning_plus/learning/models.py +149 -0
- thinkstack_core/reasoning_plus/learning/org_store.py +156 -0
- thinkstack_core/reasoning_plus/learning/pii.py +142 -0
- thinkstack_core/reasoning_plus/learning/promotion.py +58 -0
- thinkstack_core/reasoning_plus/learning/provenance.py +126 -0
- thinkstack_core/reasoning_plus/learning/recorder.py +81 -0
- thinkstack_core/reasoning_plus/learning/relevance.py +122 -0
- thinkstack_core/reasoning_plus/learning/state.py +86 -0
- thinkstack_core/reasoning_plus/learning/store.py +178 -0
- thinkstack_core/reasoning_plus/learning/theta_learning_bridge.py +94 -0
- thinkstack_core/reasoning_plus/prompt.py +90 -0
- thinkstack_core/rep.py +134 -0
- thinkstack_core/rep_network/__init__.py +25 -0
- thinkstack_core/rep_network/merge.py +70 -0
- thinkstack_core/rep_network/node.py +137 -0
- thinkstack_core/rep_network/server.py +140 -0
- thinkstack_core/rep_network/sync.py +207 -0
- thinkstack_core/sensitivity.py +182 -0
- thinkstack_core/serve.py +258 -0
- thinkstack_core/session/__init__.py +39 -0
- thinkstack_core/session/disagreement.py +188 -0
- thinkstack_core/session/models.py +114 -0
- thinkstack_core/session/orchestrator.py +182 -0
- thinkstack_core/session/planner.py +169 -0
- thinkstack_core/session/simulator.py +132 -0
- thinkstack_core/signing.py +290 -0
- thinkstack_core/sis.py +197 -0
- thinkstack_core/skills/pr-reviewer/SKILL.md +204 -0
- thinkstack_core/skills/thinkstack-auto-sync/SKILL.md +175 -0
- thinkstack_core/skills/thinkstack-session-start/SKILL.md +136 -0
- thinkstack_core/storage.py +308 -0
- thinkstack_core/templates/__init__.py +6 -0
- thinkstack_core/templates/engine.py +122 -0
- thinkstack_core/templates/go.py +18 -0
- thinkstack_core/templates/infra.py +19 -0
- thinkstack_core/templates/library/__init__.py +18 -0
- thinkstack_core/templates/library/api_design.md +27 -0
- thinkstack_core/templates/library/bug_fix.md +27 -0
- thinkstack_core/templates/library/decision_record.md +27 -0
- thinkstack_core/templates/library/engine.py +228 -0
- thinkstack_core/templates/library/security_review.md +30 -0
- thinkstack_core/templates/python.py +19 -0
- thinkstack_core/templates/react.py +18 -0
- thinkstack_core/templates/typescript.py +18 -0
- thinkstack_core/theta.py +221 -0
- thinkstack_core/theta_synthesis.py +268 -0
- thinkstack_core/topics.py +320 -0
- thinkstack_core/variance.py +219 -0
- thinkstack_core/wrapper/__init__.py +52 -0
- thinkstack_core/wrapper/anthropic.py +487 -0
- thinkstack_core/wrapper/base.py +562 -0
- thinkstack_core/wrapper/bedrock.py +342 -0
- thinkstack_core/wrapper/gemini.py +422 -0
- thinkstack_core/wrapper/ollama.py +527 -0
- thinkstack_core/wrapper/openai.py +461 -0
- thinkstack_core-4.0.0.dist-info/METADATA +868 -0
- thinkstack_core-4.0.0.dist-info/RECORD +205 -0
- thinkstack_core-4.0.0.dist-info/WHEEL +5 -0
- thinkstack_core-4.0.0.dist-info/entry_points.txt +2 -0
- thinkstack_core-4.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.models
|
|
3
|
+
===========================================
|
|
4
|
+
Data models for Reasoning Plus Learning (DRPL).
|
|
5
|
+
|
|
6
|
+
A ReasoningCall is a single recorded step (LLM, tool, memory, or action).
|
|
7
|
+
A Learning is a distilled, reusable insight derived from one or more calls.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import uuid
|
|
14
|
+
from dataclasses import asdict, dataclass, field, fields
|
|
15
|
+
from datetime import datetime, timezone
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
VALID_CALL_TYPES = ("llm", "tool", "memory", "action")
|
|
20
|
+
VALID_OUTCOMES = ("success", "failure", "partial", "unknown")
|
|
21
|
+
VALID_LEARNING_TYPES = ("insight", "correction", "pattern", "avoid", "confirm", "prescription", "canon")
|
|
22
|
+
VALID_VALIDITIES = ("active", "stale", "deprecated")
|
|
23
|
+
VALID_DISCLOSURE_LEVELS = ("PUBLIC", "PROTECTED", "PRIVATE")
|
|
24
|
+
VALID_SCOPES = ("branch", "project", "org")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _now_iso() -> str:
|
|
28
|
+
return datetime.now(timezone.utc).isoformat()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _make_id(prefix: str = "drpl") -> str:
|
|
32
|
+
return f"{prefix}_{uuid.uuid4().hex[:16]}"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class ReasoningCall:
|
|
37
|
+
"""One recorded reasoning step."""
|
|
38
|
+
|
|
39
|
+
reasoning: str
|
|
40
|
+
call_type: str = "llm"
|
|
41
|
+
name: str = ""
|
|
42
|
+
input: str = ""
|
|
43
|
+
output: str = ""
|
|
44
|
+
outcome: str = "unknown"
|
|
45
|
+
state_hash: str = ""
|
|
46
|
+
concepts: list[str] = field(default_factory=list)
|
|
47
|
+
confidence: float = 1.0
|
|
48
|
+
sensitivity: str = "PUBLIC"
|
|
49
|
+
scope: str = "project"
|
|
50
|
+
session_id: str = ""
|
|
51
|
+
timestamp: str = field(default_factory=_now_iso)
|
|
52
|
+
id: str = field(default_factory=lambda: _make_id("call"))
|
|
53
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
def __post_init__(self) -> None:
|
|
56
|
+
if self.call_type not in VALID_CALL_TYPES:
|
|
57
|
+
raise ValueError(f"Invalid call_type: {self.call_type}")
|
|
58
|
+
if self.outcome not in VALID_OUTCOMES:
|
|
59
|
+
raise ValueError(f"Invalid outcome: {self.outcome}")
|
|
60
|
+
if self.sensitivity not in VALID_DISCLOSURE_LEVELS:
|
|
61
|
+
raise ValueError(f"Invalid sensitivity: {self.sensitivity}")
|
|
62
|
+
if self.scope not in VALID_SCOPES:
|
|
63
|
+
raise ValueError(f"Invalid scope: {self.scope}")
|
|
64
|
+
self.confidence = max(0.0, min(1.0, float(self.confidence)))
|
|
65
|
+
|
|
66
|
+
def to_dict(self) -> dict[str, Any]:
|
|
67
|
+
return asdict(self)
|
|
68
|
+
|
|
69
|
+
@classmethod
|
|
70
|
+
def from_dict(cls, data: dict[str, Any]) -> "ReasoningCall":
|
|
71
|
+
valid = {f.name for f in fields(cls)}
|
|
72
|
+
return cls(**{k: v for k, v in data.items() if k in valid})
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class Learning:
|
|
77
|
+
"""A distilled, reusable learning from one or more reasoning calls."""
|
|
78
|
+
|
|
79
|
+
content: str
|
|
80
|
+
type: str = "insight"
|
|
81
|
+
trigger_concepts: list[str] = field(default_factory=list)
|
|
82
|
+
state_hash: str = ""
|
|
83
|
+
validity: str = "active"
|
|
84
|
+
confidence: float = 1.0
|
|
85
|
+
source_reasoning_ids: list[str] = field(default_factory=list)
|
|
86
|
+
sensitivity: str = "PUBLIC"
|
|
87
|
+
scope: str = "project"
|
|
88
|
+
supersedes: list[str] = field(default_factory=list)
|
|
89
|
+
superseded_by: str | None = None
|
|
90
|
+
evolution_chain: str | None = None
|
|
91
|
+
promoted_by: str | None = None
|
|
92
|
+
promoted_at: str | None = None
|
|
93
|
+
source_project_root: str | None = None
|
|
94
|
+
created_at: str = field(default_factory=_now_iso)
|
|
95
|
+
updated_at: str = field(default_factory=_now_iso)
|
|
96
|
+
id: str = field(default_factory=lambda: _make_id("learn"))
|
|
97
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
98
|
+
|
|
99
|
+
def promote(
|
|
100
|
+
self,
|
|
101
|
+
to_scope: str = "org",
|
|
102
|
+
promoted_by: str | None = None,
|
|
103
|
+
boost_confidence: bool = True,
|
|
104
|
+
) -> None:
|
|
105
|
+
"""Promote this learning to a broader scope.
|
|
106
|
+
|
|
107
|
+
Sets scope to ``to_scope``, sets type to ``"canon"``, boosts confidence
|
|
108
|
+
to 1.0 (when ``boost_confidence`` is True), and records promotion
|
|
109
|
+
metadata.
|
|
110
|
+
"""
|
|
111
|
+
if boost_confidence:
|
|
112
|
+
self.confidence = 1.0
|
|
113
|
+
self.scope = to_scope
|
|
114
|
+
self.type = "canon"
|
|
115
|
+
self.promoted_by = promoted_by
|
|
116
|
+
self.promoted_at = _now_iso()
|
|
117
|
+
self.updated_at = _now_iso()
|
|
118
|
+
|
|
119
|
+
def __post_init__(self) -> None:
|
|
120
|
+
if self.type not in VALID_LEARNING_TYPES:
|
|
121
|
+
raise ValueError(f"Invalid learning type: {self.type}")
|
|
122
|
+
if self.validity not in VALID_VALIDITIES:
|
|
123
|
+
raise ValueError(f"Invalid validity: {self.validity}")
|
|
124
|
+
if self.sensitivity not in VALID_DISCLOSURE_LEVELS:
|
|
125
|
+
raise ValueError(f"Invalid sensitivity: {self.sensitivity}")
|
|
126
|
+
if self.scope not in VALID_SCOPES:
|
|
127
|
+
raise ValueError(f"Invalid scope: {self.scope}")
|
|
128
|
+
self.confidence = max(0.0, min(1.0, float(self.confidence)))
|
|
129
|
+
|
|
130
|
+
def to_dict(self) -> dict[str, Any]:
|
|
131
|
+
return asdict(self)
|
|
132
|
+
|
|
133
|
+
@classmethod
|
|
134
|
+
def from_dict(cls, data: dict[str, Any]) -> "Learning":
|
|
135
|
+
valid = {f.name for f in fields(cls)}
|
|
136
|
+
return cls(**{k: v for k, v in data.items() if k in valid})
|
|
137
|
+
|
|
138
|
+
def short_form(self, max_chars: int = 200) -> str:
|
|
139
|
+
"""Compact, prompt-ready representation."""
|
|
140
|
+
text = self.content.strip().replace("\n", " ")
|
|
141
|
+
if len(text) > max_chars:
|
|
142
|
+
text = text[: max_chars - 3].rstrip() + "..."
|
|
143
|
+
return text
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def hash_state(*items: str) -> str:
|
|
147
|
+
"""Stable hash for a set of state tokens."""
|
|
148
|
+
joined = "|".join(sorted(items))
|
|
149
|
+
return hashlib.sha256(joined.encode("utf-8")).hexdigest()[:16]
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.org_store
|
|
3
|
+
=================================================
|
|
4
|
+
Org-level learning store aggregation.
|
|
5
|
+
|
|
6
|
+
Merges learnings from multiple governed projects with conflict-aware
|
|
7
|
+
aggregation: max confidence and most-restrictive validity. Used by the
|
|
8
|
+
five-layer filter and learning_search to produce an org-wide view of
|
|
9
|
+
reusable insights.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import logging
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from thinkstack_core.reasoning_plus.learning.models import Learning
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning")
|
|
22
|
+
|
|
23
|
+
DEFAULT_CROSS_PROJECT_THRESHOLD = 0.3
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class OrgLearningAggregate:
|
|
28
|
+
"""A single row in the aggregated org-learning view."""
|
|
29
|
+
|
|
30
|
+
content: str
|
|
31
|
+
type: str = "insight"
|
|
32
|
+
trigger_concepts: list[str] = field(default_factory=list)
|
|
33
|
+
validity: str = "active"
|
|
34
|
+
confidence: float = 1.0
|
|
35
|
+
scope: str = "org"
|
|
36
|
+
sensitivity: str = "PUBLIC"
|
|
37
|
+
source_project_roots: list[str] = field(default_factory=list)
|
|
38
|
+
source_learning_ids: list[str] = field(default_factory=list)
|
|
39
|
+
evolution_chain: str | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _merge_into_aggregate(
|
|
43
|
+
agg: OrgLearningAggregate,
|
|
44
|
+
learning: Learning,
|
|
45
|
+
project_root: str,
|
|
46
|
+
) -> None:
|
|
47
|
+
"""Merge one learning into an aggregate, taking the best per field."""
|
|
48
|
+
agg.confidence = max(agg.confidence, learning.confidence)
|
|
49
|
+
s_map = {"deprecated": 0, "stale": 1, "active": 2}
|
|
50
|
+
if s_map.get(learning.validity, 0) > s_map.get(agg.validity, 0):
|
|
51
|
+
agg.validity = learning.validity
|
|
52
|
+
agg.sensitivity = learning.sensitivity
|
|
53
|
+
if project_root not in agg.source_project_roots:
|
|
54
|
+
agg.source_project_roots.append(project_root)
|
|
55
|
+
if learning.id not in agg.source_learning_ids:
|
|
56
|
+
agg.source_learning_ids.append(learning.id)
|
|
57
|
+
if not agg.evolution_chain and learning.evolution_chain:
|
|
58
|
+
agg.evolution_chain = learning.evolution_chain
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def aggregate_org_learnings(
|
|
62
|
+
projects_dir: Path | str,
|
|
63
|
+
*,
|
|
64
|
+
min_confidence: float = 0.3,
|
|
65
|
+
top_n: int = 5,
|
|
66
|
+
) -> list[OrgLearningAggregate]:
|
|
67
|
+
"""Aggregate PUBLIC-scope learnings from multiple governed projects.
|
|
68
|
+
|
|
69
|
+
Reads ``.GCC/reasoning_learnings/learnings/`` from each project directory,
|
|
70
|
+
groups learnings by overlapping trigger concepts, and merges them with
|
|
71
|
+
max-confidence / most-restrictive-validity semantics.
|
|
72
|
+
|
|
73
|
+
Returns up to ``top_n`` :class:`OrgLearningAggregate` sorted by confidence
|
|
74
|
+
descending.
|
|
75
|
+
"""
|
|
76
|
+
out: dict[str, list[OrgLearningAggregate]] = {}
|
|
77
|
+
|
|
78
|
+
for project_root in _resolve_project_dirs(projects_dir):
|
|
79
|
+
try:
|
|
80
|
+
learnings = _load_active_learnings(project_root)
|
|
81
|
+
except Exception as exc:
|
|
82
|
+
logger.debug("thinkstack: skipping project %s — %s", project_root, exc)
|
|
83
|
+
continue
|
|
84
|
+
for learning in learnings:
|
|
85
|
+
concepts_key = _concept_key(learning.trigger_concepts)
|
|
86
|
+
if concepts_key not in out:
|
|
87
|
+
out[concepts_key] = []
|
|
88
|
+
existing = out[concepts_key]
|
|
89
|
+
for agg in existing:
|
|
90
|
+
cset = set(agg.trigger_concepts)
|
|
91
|
+
lset = set(learning.trigger_concepts)
|
|
92
|
+
overlap = cset & lset
|
|
93
|
+
if overlap and _content_similarity(agg.content, learning.content) > 0.6:
|
|
94
|
+
_merge_into_aggregate(agg, learning, str(project_root))
|
|
95
|
+
break
|
|
96
|
+
else:
|
|
97
|
+
agg = OrgLearningAggregate(
|
|
98
|
+
content=learning.content,
|
|
99
|
+
type=learning.type,
|
|
100
|
+
trigger_concepts=list(learning.trigger_concepts),
|
|
101
|
+
validity=learning.validity,
|
|
102
|
+
confidence=learning.confidence,
|
|
103
|
+
scope=learning.scope,
|
|
104
|
+
sensitivity=learning.sensitivity,
|
|
105
|
+
source_project_roots=[str(project_root)],
|
|
106
|
+
source_learning_ids=[learning.id],
|
|
107
|
+
evolution_chain=learning.evolution_chain,
|
|
108
|
+
)
|
|
109
|
+
out[concepts_key].append(agg)
|
|
110
|
+
|
|
111
|
+
results: list[OrgLearningAggregate] = []
|
|
112
|
+
for aggs in out.values():
|
|
113
|
+
for agg in aggs:
|
|
114
|
+
if agg.confidence >= min_confidence:
|
|
115
|
+
results.append(agg)
|
|
116
|
+
results.sort(key=lambda x: x.confidence, reverse=True)
|
|
117
|
+
return results[:top_n]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _concept_key(concepts: list[str]) -> str:
|
|
121
|
+
return "|".join(sorted(c.lower() for c in concepts))
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _content_similarity(a: str, b: str) -> float:
|
|
125
|
+
a_words = set(a.lower().split())
|
|
126
|
+
b_words = set(b.lower().split())
|
|
127
|
+
if not a_words or not b_words:
|
|
128
|
+
return 0.0
|
|
129
|
+
intersection = a_words & b_words
|
|
130
|
+
return len(intersection) / max(len(a_words), len(b_words))
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _resolve_project_dirs(projects_dir: Path | str) -> list[Path]:
|
|
134
|
+
dirs: list[Path] = []
|
|
135
|
+
registry_path = Path(projects_dir)
|
|
136
|
+
if not registry_path.exists():
|
|
137
|
+
return []
|
|
138
|
+
try:
|
|
139
|
+
data = json.loads(registry_path.read_text(encoding="utf-8"))
|
|
140
|
+
except (json.JSONDecodeError, OSError):
|
|
141
|
+
return []
|
|
142
|
+
governed = data.get("governed_projects", [])
|
|
143
|
+
if isinstance(governed, list):
|
|
144
|
+
for entry in governed:
|
|
145
|
+
if isinstance(entry, dict) and "project_root" in entry:
|
|
146
|
+
p = Path(entry["project_root"])
|
|
147
|
+
if p.exists():
|
|
148
|
+
dirs.append(p)
|
|
149
|
+
return dirs
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _load_active_learnings(project_root: Path) -> list[Learning]:
|
|
153
|
+
from thinkstack_core.reasoning_plus.learning.store import LearningStore
|
|
154
|
+
gcc_dir = project_root / ".GCC"
|
|
155
|
+
store = LearningStore(gcc_dir)
|
|
156
|
+
return store.list(validity="active")
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.pii
|
|
3
|
+
=========================================
|
|
4
|
+
PII detection, sanitization, and secret scanning for the learning pipeline.
|
|
5
|
+
|
|
6
|
+
Two layers of protection:
|
|
7
|
+
1. **Pre-extraction sanitization** — strip PII from reasoning text before
|
|
8
|
+
it reaches the LLM extractor. Uses presidio-analyzer when available,
|
|
9
|
+
with a pure-regex fallback.
|
|
10
|
+
2. **Post-extraction secret scanning** — detect credentials, tokens, and
|
|
11
|
+
API keys in extracted learning content. Learnings that match are
|
|
12
|
+
quarantined (not stored).
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import logging
|
|
17
|
+
import re
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from thinkstack_core.reasoning_plus.learning.models import Learning
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning.pii")
|
|
23
|
+
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
# Pre-extraction PII sanitization
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _get_presidio_engine() -> Any:
|
|
30
|
+
"""Return a presidio AnalyzerEngine, or None if presidio is unavailable."""
|
|
31
|
+
try:
|
|
32
|
+
from presidio_analyzer import AnalyzerEngine
|
|
33
|
+
return AnalyzerEngine()
|
|
34
|
+
except ImportError:
|
|
35
|
+
logger.debug("thinkstack: presidio not available, using regex PII fallback")
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
_PRESIDIO_ENGINE: Any | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _ensure_engine() -> Any:
|
|
43
|
+
global _PRESIDIO_ENGINE
|
|
44
|
+
if _PRESIDIO_ENGINE is None:
|
|
45
|
+
_PRESIDIO_ENGINE = _get_presidio_engine()
|
|
46
|
+
return _PRESIDIO_ENGINE
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# Fallback regex patterns for PII detection (used when presidio is unavailable)
|
|
50
|
+
_PII_REGEX_PATTERNS: list[tuple[str, str]] = [
|
|
51
|
+
("CREDIT_CARD", r"\b(?:\d{4}[-\s]?){3}\d{4}\b"),
|
|
52
|
+
("EMAIL", r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b"),
|
|
53
|
+
("PHONE", r"\b\+?\d{1,3}[-.\s]?\(?\d{1,4}\)?[-.\s]?\d{1,4}[-.\s]?\d{1,9}\b"),
|
|
54
|
+
("SSN", r"\b\d{3}-\d{2}-\d{4}\b"),
|
|
55
|
+
("API_KEY_GENERIC", r"\b(?:sk-[A-Za-z0-9]{20,}|[A-Za-z0-9]{32,})\b"),
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def sanitize_for_extraction(text: str) -> str:
|
|
60
|
+
"""Strip PII from *text*, replacing it with ``[REDACTED:<type>]``.
|
|
61
|
+
|
|
62
|
+
Presidio-analyzer runs first (when available) for NLP-based detection.
|
|
63
|
+
Regex patterns always run second as a catch-all for API keys, emails,
|
|
64
|
+
credit cards, phones, and SSNs. The regex pass never overwrites
|
|
65
|
+
presidio's output because it operates on the result of the presidio pass.
|
|
66
|
+
"""
|
|
67
|
+
if not text:
|
|
68
|
+
return text
|
|
69
|
+
|
|
70
|
+
result = text
|
|
71
|
+
|
|
72
|
+
# Presidio — optional NLP-based detection for additional coverage.
|
|
73
|
+
engine = _ensure_engine()
|
|
74
|
+
if engine is not None:
|
|
75
|
+
try:
|
|
76
|
+
presidio_results = engine.analyze(text=result, language="en")
|
|
77
|
+
if presidio_results:
|
|
78
|
+
sanitized = list(result)
|
|
79
|
+
for r in sorted(presidio_results, key=lambda x: x.start, reverse=True):
|
|
80
|
+
entity_type = r.entity_type or "PII"
|
|
81
|
+
placeholder = f"[REDACTED:{entity_type}]"
|
|
82
|
+
sanitized[r.start : r.end] = list(placeholder)
|
|
83
|
+
result = "".join(sanitized)
|
|
84
|
+
except Exception as exc:
|
|
85
|
+
logger.debug("thinkstack: presidio analysis failed — %s", exc)
|
|
86
|
+
|
|
87
|
+
# Apply regex patterns — always runs, catches API keys and other patterns
|
|
88
|
+
# that presidio may have missed. Operates on presidio's output (or the
|
|
89
|
+
# original text if presidio is unavailable), so it never overwrites
|
|
90
|
+
# presidio redactions.
|
|
91
|
+
for entity_type, pattern in _PII_REGEX_PATTERNS:
|
|
92
|
+
try:
|
|
93
|
+
result = re.sub(pattern, f"[REDACTED:{entity_type}]", result)
|
|
94
|
+
except Exception:
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
return result
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# ---------------------------------------------------------------------------
|
|
101
|
+
# Post-extraction secret scanning
|
|
102
|
+
# ---------------------------------------------------------------------------
|
|
103
|
+
|
|
104
|
+
_SECRET_PATTERNS: list[tuple[str, str]] = [
|
|
105
|
+
("OPENAI_KEY", r"sk-[A-Za-z0-9]{20,}"),
|
|
106
|
+
("GITHUB_TOKEN", r"(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9_]{36,}"),
|
|
107
|
+
("AWS_ACCESS_KEY", r"AKIA[0-9A-Z]{16}"),
|
|
108
|
+
("BEARER_TOKEN", r"Bearer\s+[A-Za-z0-9\-._~+/]{20,}"),
|
|
109
|
+
("GENERIC_SECRET", r"(?:secret|password|token|key)\s*[:=]\s*['\"]?[A-Za-z0-9_\-]{16,}"),
|
|
110
|
+
]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def scan_for_secrets(text: str) -> list[str]:
|
|
114
|
+
"""Scan *text* for credentials, tokens, and API keys.
|
|
115
|
+
|
|
116
|
+
Returns a list of secret type strings that were matched.
|
|
117
|
+
Empty list means no secrets detected.
|
|
118
|
+
"""
|
|
119
|
+
if not text:
|
|
120
|
+
return []
|
|
121
|
+
found: list[str] = []
|
|
122
|
+
for name, pattern in _SECRET_PATTERNS:
|
|
123
|
+
try:
|
|
124
|
+
if re.search(pattern, text):
|
|
125
|
+
found.append(name)
|
|
126
|
+
except Exception:
|
|
127
|
+
continue
|
|
128
|
+
return found
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# ---------------------------------------------------------------------------
|
|
132
|
+
# Sync filtering helper
|
|
133
|
+
# ---------------------------------------------------------------------------
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def filter_learnings_for_sync(learnings: list[Learning]) -> list[Learning]:
|
|
137
|
+
"""Filter learnings for sync: exclude PRIVATE-disclosure learnings.
|
|
138
|
+
|
|
139
|
+
Only PUBLIC and PROTECTED learnings are included in sync payloads.
|
|
140
|
+
PRIVATE learnings stay local.
|
|
141
|
+
"""
|
|
142
|
+
return [l for l in learnings if l.sensitivity != "PRIVATE"]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.promotion
|
|
3
|
+
=================================================
|
|
4
|
+
Learning scope promotion workflow.
|
|
5
|
+
|
|
6
|
+
Promotion elevates a learning from the project scope to the org scope,
|
|
7
|
+
making it eligible for cross-project injection via the five-layer filter.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from thinkstack_core.reasoning_plus.learning.models import Learning
|
|
15
|
+
from thinkstack_core.reasoning_plus.learning.store import LearningStore
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def promote_learning(
|
|
21
|
+
learning_id: str,
|
|
22
|
+
gcc_dir: Path | str,
|
|
23
|
+
*,
|
|
24
|
+
to_scope: str = "org",
|
|
25
|
+
promoted_by: str | None = None,
|
|
26
|
+
boost_confidence: bool = True,
|
|
27
|
+
) -> Learning | None:
|
|
28
|
+
"""Promote a learning to a broader scope and persist the change.
|
|
29
|
+
|
|
30
|
+
The learning is:
|
|
31
|
+
- reassigned to ``to_scope`` (default ``"org"``)
|
|
32
|
+
- its type becomes ``"canon"``
|
|
33
|
+
- its confidence is boosted to 1.0 (when ``boost_confidence``)
|
|
34
|
+
- promotion metadata (``promoted_by``, ``promoted_at``) is recorded
|
|
35
|
+
"""
|
|
36
|
+
gcc_dir = Path(gcc_dir)
|
|
37
|
+
store = LearningStore(gcc_dir)
|
|
38
|
+
learning = store.get(learning_id)
|
|
39
|
+
if not learning:
|
|
40
|
+
logger.warning("thinkstack: promotion — learning %s not found", learning_id)
|
|
41
|
+
return None
|
|
42
|
+
|
|
43
|
+
if learning.sensitivity == "PRIVATE":
|
|
44
|
+
logger.warning(
|
|
45
|
+
"thinkstack: cannot promote PRIVATE learning %s — "
|
|
46
|
+
"PRIVATE learnings must remain project-local",
|
|
47
|
+
learning_id,
|
|
48
|
+
)
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
learning.promote(to_scope=to_scope, promoted_by=promoted_by, boost_confidence=boost_confidence)
|
|
52
|
+
learning.sensitivity = "PUBLIC"
|
|
53
|
+
store.save(learning, deduplicate=False)
|
|
54
|
+
logger.info(
|
|
55
|
+
"thinkstack: promoted learning %s to scope=%s type=%s",
|
|
56
|
+
learning_id, to_scope, learning.type,
|
|
57
|
+
)
|
|
58
|
+
return learning
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""
|
|
2
|
+
thinkstack_core.reasoning_plus.learning.provenance
|
|
3
|
+
===============================================
|
|
4
|
+
Provenance tracking for injected learnings.
|
|
5
|
+
|
|
6
|
+
Records which Learning records were injected into each LLM/tool call so that
|
|
7
|
+
observed outcomes can be attributed back to the learnings that influenced the
|
|
8
|
+
call.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import hashlib
|
|
13
|
+
import json
|
|
14
|
+
import logging
|
|
15
|
+
from dataclasses import asdict, dataclass, field, fields
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from thinkstack_core.reasoning_plus.learning.models import _make_id, _now_iso
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger("thinkstack.reasoning_plus.learning")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class ProvenanceRecord:
|
|
26
|
+
"""Which learnings were injected into a single call, and what the outcome was."""
|
|
27
|
+
|
|
28
|
+
call_id: str
|
|
29
|
+
learning_ids: list[str] = field(default_factory=list)
|
|
30
|
+
session_id: str = ""
|
|
31
|
+
prompt_hash: str = ""
|
|
32
|
+
prompt_excerpt: str = ""
|
|
33
|
+
model_name: str = ""
|
|
34
|
+
outcome: str | None = None
|
|
35
|
+
injected_at: str = field(default_factory=_now_iso)
|
|
36
|
+
updated_at: str = field(default_factory=_now_iso)
|
|
37
|
+
id: str = field(default_factory=lambda: _make_id("prov"))
|
|
38
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
def to_dict(self) -> dict[str, Any]:
|
|
41
|
+
return asdict(self)
|
|
42
|
+
|
|
43
|
+
@classmethod
|
|
44
|
+
def from_dict(cls, data: dict[str, Any]) -> "ProvenanceRecord":
|
|
45
|
+
valid = {f.name for f in fields(cls)}
|
|
46
|
+
return cls(**{k: v for k, v in data.items() if k in valid})
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class ProvenanceStore:
|
|
50
|
+
"""Persist ProvenanceRecord records under .GCC/reasoning_learnings/provenance/."""
|
|
51
|
+
|
|
52
|
+
def __init__(self, gcc_dir: Path | str) -> None:
|
|
53
|
+
self._gcc_dir = Path(gcc_dir)
|
|
54
|
+
self._provenance_dir = self._gcc_dir / "reasoning_learnings" / "provenance"
|
|
55
|
+
|
|
56
|
+
def save(self, record: ProvenanceRecord) -> Path:
|
|
57
|
+
self._provenance_dir.mkdir(parents=True, exist_ok=True)
|
|
58
|
+
path = self._provenance_dir / f"{record.id}.json"
|
|
59
|
+
_atomic_json_write(path, record.to_dict())
|
|
60
|
+
return path
|
|
61
|
+
|
|
62
|
+
def get(self, record_id: str) -> ProvenanceRecord | None:
|
|
63
|
+
path = self._provenance_dir / f"{record_id}.json"
|
|
64
|
+
if not path.exists():
|
|
65
|
+
return None
|
|
66
|
+
try:
|
|
67
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
68
|
+
return ProvenanceRecord.from_dict(data)
|
|
69
|
+
except Exception as exc:
|
|
70
|
+
logger.warning("thinkstack: failed to read provenance %s — %s", record_id, exc)
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
def get_by_call(self, call_id: str) -> ProvenanceRecord | None:
|
|
74
|
+
for record in self.list():
|
|
75
|
+
if record.call_id == call_id:
|
|
76
|
+
return record
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
def list(self, outcome: str | None = None, learning_id: str | None = None) -> list[ProvenanceRecord]:
|
|
80
|
+
out: list[ProvenanceRecord] = []
|
|
81
|
+
if not self._provenance_dir.exists():
|
|
82
|
+
return out
|
|
83
|
+
for path in sorted(self._provenance_dir.glob("*.json")):
|
|
84
|
+
try:
|
|
85
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
86
|
+
record = ProvenanceRecord.from_dict(data)
|
|
87
|
+
if outcome and record.outcome != outcome:
|
|
88
|
+
continue
|
|
89
|
+
if learning_id and learning_id not in record.learning_ids:
|
|
90
|
+
continue
|
|
91
|
+
out.append(record)
|
|
92
|
+
except Exception as exc:
|
|
93
|
+
logger.debug("thinkstack: skipping malformed provenance %s — %s", path, exc)
|
|
94
|
+
return out
|
|
95
|
+
|
|
96
|
+
def update_outcome(self, call_id: str, outcome: str) -> ProvenanceRecord | None:
|
|
97
|
+
record = self.get_by_call(call_id)
|
|
98
|
+
if not record:
|
|
99
|
+
return None
|
|
100
|
+
record.outcome = outcome
|
|
101
|
+
record.updated_at = _now_iso()
|
|
102
|
+
self.save(record)
|
|
103
|
+
return record
|
|
104
|
+
|
|
105
|
+
def count(self) -> int:
|
|
106
|
+
if not self._provenance_dir.exists():
|
|
107
|
+
return 0
|
|
108
|
+
return len(list(self._provenance_dir.glob("*.json")))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _atomic_json_write(path: Path, data: dict[str, Any]) -> None:
|
|
112
|
+
tmp = path.with_suffix(path.suffix + ".tmp")
|
|
113
|
+
try:
|
|
114
|
+
with open(tmp, "w", encoding="utf-8") as f:
|
|
115
|
+
json.dump(data, f, indent=2)
|
|
116
|
+
tmp.replace(path)
|
|
117
|
+
finally:
|
|
118
|
+
if tmp.exists():
|
|
119
|
+
try:
|
|
120
|
+
tmp.unlink()
|
|
121
|
+
except Exception:
|
|
122
|
+
pass
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _hash_prompt(prompt_text: str) -> str:
|
|
126
|
+
return hashlib.sha256(prompt_text.encode("utf-8")).hexdigest()[:16]
|