continuum-toolkit 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. cli/__init__.py +12 -0
  2. cli/main.py +415 -0
  3. confidence/__init__.py +13 -0
  4. confidence/calculator.py +308 -0
  5. confidence/models.py +40 -0
  6. context/__init__.py +17 -0
  7. context/models.py +108 -0
  8. context/pruner.py +228 -0
  9. context/selector.py +193 -0
  10. continuum_toolkit-1.0.0.dist-info/METADATA +511 -0
  11. continuum_toolkit-1.0.0.dist-info/RECORD +72 -0
  12. continuum_toolkit-1.0.0.dist-info/WHEEL +5 -0
  13. continuum_toolkit-1.0.0.dist-info/entry_points.txt +2 -0
  14. continuum_toolkit-1.0.0.dist-info/licenses/LICENSE +21 -0
  15. continuum_toolkit-1.0.0.dist-info/top_level.txt +13 -0
  16. contradictions/__init__.py +19 -0
  17. contradictions/detector.py +442 -0
  18. contradictions/models.py +75 -0
  19. core/__init__.py +97 -0
  20. core/enums.py +130 -0
  21. core/evidence.py +117 -0
  22. core/interfaces.py +209 -0
  23. core/schema.py +391 -0
  24. core/serializer.py +84 -0
  25. core/state_models.py +530 -0
  26. daemon/__init__.py +12 -0
  27. daemon/service.py +170 -0
  28. extractors/__init__.py +51 -0
  29. extractors/base.py +117 -0
  30. extractors/config/parsers.py +288 -0
  31. extractors/config/secret_sanitizer.py +91 -0
  32. extractors/config_extractor.py +172 -0
  33. extractors/conversation/analyzers.py +193 -0
  34. extractors/conversation/models.py +148 -0
  35. extractors/conversation_extractor.py +166 -0
  36. extractors/git_extractor.py +305 -0
  37. extractors/parsers/base.py +91 -0
  38. extractors/parsers/comment_parser.py +51 -0
  39. extractors/parsers/js_ts_parser.py +171 -0
  40. extractors/parsers/python_parser.py +180 -0
  41. extractors/verification/runners.py +278 -0
  42. extractors/verification_extractor.py +263 -0
  43. extractors/workspace_extractor.py +221 -0
  44. graph/__init__.py +24 -0
  45. graph/diff.py +109 -0
  46. graph/manager.py +473 -0
  47. graph/models.py +62 -0
  48. graph/propagator.py +194 -0
  49. graph/query.py +86 -0
  50. graph/snapshot.py +85 -0
  51. handoff/__init__.py +28 -0
  52. handoff/adapters/__init__.py +45 -0
  53. handoff/adapters/base.py +90 -0
  54. handoff/adapters/claude_adapter.py +176 -0
  55. handoff/adapters/codex_gpt_adapter.py +151 -0
  56. handoff/adapters/gemini_adapter.py +151 -0
  57. handoff/adapters/local_model_adapter.py +130 -0
  58. handoff/models.py +88 -0
  59. handoff/packager.py +288 -0
  60. pipeline/__init__.py +9 -0
  61. pipeline/orchestrator.py +260 -0
  62. resolution/__init__.py +15 -0
  63. resolution/resolver.py +311 -0
  64. storage/__init__.py +18 -0
  65. storage/hooks.py +125 -0
  66. storage/manager.py +127 -0
  67. storage/models.py +39 -0
  68. storage/recovery.py +98 -0
  69. watcher/__init__.py +17 -0
  70. watcher/detector.py +136 -0
  71. watcher/models.py +62 -0
  72. watcher/updater.py +163 -0
context/pruner.py ADDED
@@ -0,0 +1,228 @@
1
+ """
2
+ Project Continuum - Context Pruning & Token Budget Optimization
3
+ ===============================================================
4
+ Milestone 5 - Phase 13: Context Pruning & Token Budget Optimization.
5
+ Implements greedy priority-tiered context pruning under configurable token budgets.
6
+ Guarantees that active architectural constraints and blockers are never silently dropped.
7
+ """
8
+
9
+ from dataclasses import dataclass, field, asdict
10
+ from enum import Enum
11
+ import math
12
+ from typing import Any, Dict, List, Optional, Set, Tuple
13
+
14
+ from context.models import TaskContext
15
+
16
+
17
+ class ContextPriorityTier(int, Enum):
18
+ """Priority levels for task context retention."""
19
+ TIER_1_CRITICAL_CONSTRAINTS = 1 # Active architectural constraints, blockers, goal (NEVER dropped)
20
+ TIER_2_PRIMARY_ENTITIES = 2 # Primary components, failing tests, key interfaces
21
+ TIER_3_DIRECT_DEPENDENCIES = 3 # Direct prerequisite components, passing tests
22
+ TIER_4_SOURCE_SNIPPETS = 4 # Detailed symbol line spans & secondary files
23
+ TIER_5_HISTORICAL_CONTEXT = 5 # Architectural background narratives, secondary decisions
24
+
25
+
26
+ @dataclass
27
+ class PrunedContextResult:
28
+ """
29
+ Result of budget-constrained context optimization.
30
+ """
31
+ task_description: str
32
+ token_budget: Optional[int]
33
+ estimated_tokens_before: int
34
+ estimated_tokens_after: int
35
+ reduction_percentage: float
36
+ retained_sections: List[str] = field(default_factory=list)
37
+ omitted_items: List[Dict[str, str]] = field(default_factory=list)
38
+ constraint_preservation_guarantee: bool = True
39
+ formatted_prompt: str = ""
40
+
41
+ def to_dict(self) -> Dict[str, Any]:
42
+ return asdict(self)
43
+
44
+
45
+ class ContextPruner:
46
+ """
47
+ Optimizes and prunes TaskContext to fit within a target token budget while
48
+ strictly preserving Tier 1 critical constraints and reporting all omissions.
49
+ """
50
+
51
+ DEFAULT_CHAR_PER_TOKEN = 3.8 # Standard heuristic for code + English markdown
52
+
53
+ def __init__(self, char_per_token: float = DEFAULT_CHAR_PER_TOKEN):
54
+ self._char_per_token = char_per_token
55
+
56
+ def estimate_tokens(self, text: str) -> int:
57
+ """Estimates token count from text using character ratio heuristic."""
58
+ if not text:
59
+ return 0
60
+ return math.ceil(len(text) / self._char_per_token)
61
+
62
+ def prune_to_budget(
63
+ self,
64
+ task_context: TaskContext,
65
+ token_budget: Optional[int] = None
66
+ ) -> PrunedContextResult:
67
+ """
68
+ Prunes context items based on priority tiers until within token_budget.
69
+ Tier 1 (Constraints & Blockers) is strictly protected.
70
+ """
71
+ full_markdown = task_context.to_markdown()
72
+ initial_tokens = self.estimate_tokens(full_markdown)
73
+
74
+ if token_budget is None or initial_tokens <= token_budget:
75
+ return PrunedContextResult(
76
+ task_description=task_context.task_description,
77
+ token_budget=token_budget,
78
+ estimated_tokens_before=initial_tokens,
79
+ estimated_tokens_after=initial_tokens,
80
+ reduction_percentage=0.0,
81
+ retained_sections=["all"],
82
+ omitted_items=[],
83
+ constraint_preservation_guarantee=True,
84
+ formatted_prompt=full_markdown
85
+ )
86
+
87
+ # Token budget is constrained: selectively assemble from Tier 1 to Tier 5
88
+ omitted_items: List[Dict[str, str]] = []
89
+ retained_sections: List[str] = []
90
+
91
+ # Tier 1 (Mandatory - Critical Constraints & Blockers)
92
+ tier1_lines = [
93
+ f"# Task Briefing: {task_context.task_description}",
94
+ ""
95
+ ]
96
+
97
+ if task_context.active_constraints:
98
+ tier1_lines.append("## 🛡️ Active Constraints (CRITICAL)")
99
+ for c in task_context.active_constraints:
100
+ tier1_lines.append(f"- {c}")
101
+ retained_sections.append("constraints")
102
+
103
+ if task_context.known_blockers:
104
+ tier1_lines.append("\n## ⚠️ Active Blockers & Discrepancies")
105
+ for b in task_context.known_blockers:
106
+ tier1_lines.append(f"- **[{b.severity}]**: {b.explanation}")
107
+ retained_sections.append("blockers")
108
+
109
+ current_text = "\n".join(tier1_lines)
110
+ current_tokens = self.estimate_tokens(current_text)
111
+
112
+ # Tier 2 (Primary Components & Failing Tests)
113
+ tier2_lines = []
114
+ if task_context.primary_nodes:
115
+ tier2_lines.append("\n## 🎯 Primary Components")
116
+ for n in task_context.primary_nodes:
117
+ tier2_lines.append(f"- **{n.name}** (`{n.node_type.value}`): Status `[{n.status.value}]` (Confidence: {n.confidence_score:.0f}%)")
118
+
119
+ failing_tests = [t for t in task_context.relevant_tests if t.status.value != "VERIFIED"]
120
+ if failing_tests:
121
+ tier2_lines.append("\n## ❌ Failing Test Evidence")
122
+ for t in failing_tests:
123
+ tier2_lines.append(f"- ❌ **{t.name}** (`{t.suite}`): exit code {t.exit_code} - `{t.error_message or ''}`")
124
+
125
+ tier2_text = "\n".join(tier2_lines)
126
+ if current_tokens + self.estimate_tokens(tier2_text) <= token_budget:
127
+ current_text += tier2_text
128
+ current_tokens = self.estimate_tokens(current_text)
129
+ retained_sections.append("primary_components")
130
+ else:
131
+ omitted_items.append({
132
+ "tier": "Tier 2",
133
+ "item": "Primary components detailed list",
134
+ "reason": "Exceeded token budget"
135
+ })
136
+
137
+ # Tier 3 (Direct Dependencies & Passing Tests)
138
+ tier3_lines = []
139
+ if task_context.dependency_nodes:
140
+ tier3_lines.append("\n## 🔗 Direct Prerequisites")
141
+ for d in task_context.dependency_nodes:
142
+ tier3_lines.append(f"- **{d.name}** (`{d.node_type.value}`): Status `[{d.status.value}]`")
143
+
144
+ passing_tests = [t for t in task_context.relevant_tests if t.status.value == "VERIFIED"]
145
+ if passing_tests:
146
+ tier3_lines.append("\n## ✅ Verified Test Baseline")
147
+ for t in passing_tests[:3]: # Top passing tests
148
+ tier3_lines.append(f"- ✅ **{t.name}** (`{t.suite}`)")
149
+
150
+ tier3_text = "\n".join(tier3_lines)
151
+ if current_tokens + self.estimate_tokens(tier3_text) <= token_budget:
152
+ current_text += tier3_text
153
+ current_tokens = self.estimate_tokens(current_text)
154
+ retained_sections.append("dependencies")
155
+ else:
156
+ if task_context.dependency_nodes:
157
+ omitted_items.append({
158
+ "tier": "Tier 3",
159
+ "item": f"{len(task_context.dependency_nodes)} dependency nodes",
160
+ "reason": "Omitted to fit token budget"
161
+ })
162
+
163
+ # Tier 4 (Source Code Symbols & Line Spans)
164
+ tier4_lines = []
165
+ if task_context.relevant_symbols:
166
+ tier4_lines.append("\n## 🧬 Relevant Symbols")
167
+ for s in task_context.relevant_symbols[:5]:
168
+ tier4_lines.append(f"- `{s.kind} {s.name}` in `{s.file_path}` (lines {s.line_start}-{s.line_end})")
169
+
170
+ tier4_text = "\n".join(tier4_lines)
171
+ if current_tokens + self.estimate_tokens(tier4_text) <= token_budget:
172
+ current_text += tier4_text
173
+ current_tokens = self.estimate_tokens(current_text)
174
+ retained_sections.append("symbols")
175
+ else:
176
+ if task_context.relevant_symbols:
177
+ omitted_items.append({
178
+ "tier": "Tier 4",
179
+ "item": f"{len(task_context.relevant_symbols)} code symbols",
180
+ "reason": "Omitted to fit token budget"
181
+ })
182
+
183
+ # Tier 5 (Background Decisions & Historical Narrative)
184
+ tier5_lines = []
185
+ if task_context.relevant_decisions:
186
+ tier5_lines.append("\n## 📐 Architectural Decisions")
187
+ for d in task_context.relevant_decisions:
188
+ tier5_lines.append(f"- **{d.title}**: {d.rationale}")
189
+
190
+ tier5_text = "\n".join(tier5_lines)
191
+ if current_tokens + self.estimate_tokens(tier5_text) <= token_budget:
192
+ current_text += tier5_text
193
+ current_tokens = self.estimate_tokens(current_text)
194
+ retained_sections.append("decisions")
195
+ else:
196
+ if task_context.relevant_decisions:
197
+ omitted_items.append({
198
+ "tier": "Tier 5",
199
+ "item": f"{len(task_context.relevant_decisions)} architectural decisions",
200
+ "reason": "Omitted to fit token budget"
201
+ })
202
+
203
+ # Add explicit omission ledger footer
204
+ footer_lines = [
205
+ "",
206
+ "---",
207
+ f"**[Continuum Context Budget Audit: {current_tokens} tokens (Budget: {token_budget}) | {len(omitted_items)} section(s) pruned]**"
208
+ ]
209
+ if omitted_items:
210
+ footer_lines.append("**Omitted Items due to budget constraints:**")
211
+ for o in omitted_items:
212
+ footer_lines.append(f"- *{o['tier']}*: {o['item']} ({o['reason']})")
213
+
214
+ current_text += "\n".join(footer_lines)
215
+ final_tokens = self.estimate_tokens(current_text)
216
+ reduction = round(max(0.0, ((initial_tokens - final_tokens) / initial_tokens) * 100.0), 1)
217
+
218
+ return PrunedContextResult(
219
+ task_description=task_context.task_description,
220
+ token_budget=token_budget,
221
+ estimated_tokens_before=initial_tokens,
222
+ estimated_tokens_after=final_tokens,
223
+ reduction_percentage=reduction,
224
+ retained_sections=retained_sections,
225
+ omitted_items=omitted_items,
226
+ constraint_preservation_guarantee=True,
227
+ formatted_prompt=current_text
228
+ )
context/selector.py ADDED
@@ -0,0 +1,193 @@
1
+ """
2
+ Project Continuum - Task-Driven Context Selection Engine
3
+ ========================================================
4
+ Milestone 5 - Phase 12: Task-Driven Context Selection.
5
+ Identifies and extracts the minimum sufficient, high-signal project slice
6
+ required to execute a specific software development task.
7
+ """
8
+
9
+ import re
10
+ from typing import Any, Dict, List, Optional, Set, Tuple
11
+
12
+ from core.enums import NodeType, RelationType, Status
13
+ from core.interfaces import IContextSelector
14
+ from core.state_models import (
15
+ ArchitecturalDecision,
16
+ AstSymbol,
17
+ CanonicalProjectState,
18
+ ContradictionRecord,
19
+ GraphNode,
20
+ TestResult,
21
+ )
22
+ from context.models import TaskContext
23
+ from graph.manager import StateGraphManager
24
+
25
+
26
+ class TaskContextSelector(IContextSelector):
27
+ """
28
+ Selects task-specific project context from CanonicalProjectState and DAG.
29
+ Implements IContextSelector Protocol.
30
+ """
31
+
32
+ STOP_WORDS = {
33
+ "a", "an", "the", "and", "or", "in", "on", "at", "to", "for", "with",
34
+ "by", "of", "is", "are", "be", "this", "that", "it", "from", "as", "fix",
35
+ "add", "update", "implement", "create", "make", "refactor", "check", "test"
36
+ }
37
+
38
+ def select_context_for_task(
39
+ self,
40
+ task_description: str,
41
+ canonical_state: CanonicalProjectState,
42
+ token_budget: Optional[int] = None
43
+ ) -> Dict[str, Any]:
44
+ """
45
+ Extracts task-relevant context slice and applies token budget optimization if requested.
46
+ Implements IContextSelector interface.
47
+ """
48
+ task_ctx = self.extract_task_context(task_description, canonical_state, token_budget)
49
+ if token_budget is not None:
50
+ from context.pruner import ContextPruner
51
+ pruner = ContextPruner()
52
+ pruned_res = pruner.prune_to_budget(task_ctx, token_budget=token_budget)
53
+ d = task_ctx.to_dict()
54
+ d["pruning"] = pruned_res.to_dict()
55
+ return d
56
+
57
+ return task_ctx.to_dict()
58
+
59
+ def extract_task_context(
60
+ self,
61
+ task_description: str,
62
+ canonical_state: CanonicalProjectState,
63
+ token_budget: Optional[int] = None
64
+ ) -> TaskContext:
65
+ """
66
+ Returns strongly-typed TaskContext object.
67
+ """
68
+ # Build state graph manager
69
+ graph_mgr = StateGraphManager(canonical_state)
70
+ if not graph_mgr.nodes:
71
+ graph_mgr.build_from_canonical_state(canonical_state)
72
+
73
+ # 1. Extract keywords and semantic tokens from task prompt
74
+ keywords = self._extract_keywords(task_description)
75
+
76
+ # 2. Identify Primary Graph Nodes
77
+ primary_nodes: List[GraphNode] = []
78
+ matched_node_ids: Set[str] = set()
79
+
80
+ for node in graph_mgr.nodes.values():
81
+ node_name_clean = node.name.lower()
82
+ if any(kw in node_name_clean for kw in keywords) or any(kw in node.id.lower() for kw in keywords):
83
+ primary_nodes.append(node)
84
+ matched_node_ids.add(node.id)
85
+
86
+ # If no nodes matched directly, fall back to matching symbols or requirements
87
+ if not primary_nodes and graph_mgr.nodes:
88
+ primary_nodes = list(graph_mgr.nodes.values())[:3]
89
+ for n in primary_nodes:
90
+ matched_node_ids.add(n.id)
91
+
92
+ # 3. Locate Direct Dependencies & Prerequisites
93
+ dependency_nodes: List[GraphNode] = []
94
+ dep_ids: Set[str] = set()
95
+
96
+ for p_node in primary_nodes:
97
+ # Direct dependencies
98
+ deps = graph_mgr.get_dependencies(p_node.id, recursive=False)
99
+ for d in deps:
100
+ if d.id not in matched_node_ids and d.id not in dep_ids:
101
+ dependency_nodes.append(d)
102
+ dep_ids.add(d.id)
103
+
104
+ # 4. Filter Relevant AST Symbols and Physical Files
105
+ relevant_symbols: List[AstSymbol] = []
106
+ relevant_files: Set[str] = set()
107
+
108
+ all_target_names = {n.name.lower() for n in primary_nodes + dependency_nodes}
109
+ all_target_names.update(keywords)
110
+
111
+ for sym in canonical_state.project_state.symbols:
112
+ sym_lower = sym.name.lower()
113
+ if any(t in sym_lower or sym_lower in t for t in all_target_names):
114
+ relevant_symbols.append(sym)
115
+ relevant_files.add(sym.file_path)
116
+
117
+ for f in canonical_state.project_state.files:
118
+ f_lower = f.lower()
119
+ if any(kw in f_lower for kw in keywords):
120
+ relevant_files.add(f)
121
+
122
+ # 5. Filter Relevant Test Results & Suites
123
+ relevant_tests: List[TestResult] = []
124
+ for t in canonical_state.project_state.test_results:
125
+ t_lower = (t.name + " " + t.suite).lower()
126
+ if any(t_name in t_lower for t_name in all_target_names):
127
+ relevant_tests.append(t)
128
+ elif t.status == Status.FAILED:
129
+ # Include failing tests if they might relate
130
+ relevant_tests.append(t)
131
+
132
+ # Deduplicate tests
133
+ unique_tests = []
134
+ seen_test_ids = set()
135
+ for t in relevant_tests:
136
+ if t.test_id not in seen_test_ids:
137
+ seen_test_ids.add(t.test_id)
138
+ unique_tests.append(t)
139
+
140
+ # 6. Locate Relevant Known Blockers & Contradictions
141
+ relevant_blockers: List[ContradictionRecord] = []
142
+ for c in canonical_state.contradictions:
143
+ if not c.resolved:
144
+ c_text = (c.claim_text + " " + c.explanation).lower()
145
+ if any(t in c_text for t in all_target_names):
146
+ relevant_blockers.append(c)
147
+
148
+ # 7. Extract Architectural Decisions & Active Constraints
149
+ relevant_decisions: List[ArchitecturalDecision] = []
150
+ active_constraints: List[str] = []
151
+
152
+ for d in canonical_state.conversational_state.architectural_decisions:
153
+ d_text = (d.title + " " + d.rationale).lower()
154
+ if any(kw in d_text for kw in keywords):
155
+ relevant_decisions.append(d)
156
+ active_constraints.extend(d.constraints)
157
+
158
+ # 8. Compute Omitted Node Count
159
+ total_graph_nodes = len(graph_mgr.nodes)
160
+ included_count = len(matched_node_ids) + len(dep_ids)
161
+ omitted_count = max(0, total_graph_nodes - included_count)
162
+
163
+ verification_summary = {
164
+ "primary_node_count": len(primary_nodes),
165
+ "dependency_node_count": len(dependency_nodes),
166
+ "relevant_tests_count": len(unique_tests),
167
+ "failing_tests_count": sum(1 for t in unique_tests if t.status == Status.FAILED),
168
+ "active_blockers_count": len(relevant_blockers),
169
+ }
170
+
171
+ return TaskContext(
172
+ task_description=task_description,
173
+ primary_nodes=primary_nodes,
174
+ dependency_nodes=dependency_nodes,
175
+ relevant_files=sorted(list(relevant_files)),
176
+ relevant_symbols=relevant_symbols,
177
+ relevant_tests=unique_tests,
178
+ active_constraints=active_constraints,
179
+ relevant_decisions=relevant_decisions,
180
+ known_blockers=relevant_blockers,
181
+ verification_status_summary=verification_summary,
182
+ omitted_node_count=omitted_count
183
+ )
184
+
185
+ def _extract_keywords(self, text: str) -> List[str]:
186
+ """Extracts significant keywords and identifier tokens from prompt text."""
187
+ # Find alphanumeric words with length >= 3
188
+ words = re.findall(r"[A-Za-z0-9_]{3,}", text)
189
+ keywords = [
190
+ w.lower() for w in words
191
+ if w.lower() not in self.STOP_WORDS
192
+ ]
193
+ return list(dict.fromkeys(keywords)) # Preserve order deduplication