continuum-toolkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +12 -0
- cli/main.py +415 -0
- confidence/__init__.py +13 -0
- confidence/calculator.py +308 -0
- confidence/models.py +40 -0
- context/__init__.py +17 -0
- context/models.py +108 -0
- context/pruner.py +228 -0
- context/selector.py +193 -0
- continuum_toolkit-1.0.0.dist-info/METADATA +511 -0
- continuum_toolkit-1.0.0.dist-info/RECORD +72 -0
- continuum_toolkit-1.0.0.dist-info/WHEEL +5 -0
- continuum_toolkit-1.0.0.dist-info/entry_points.txt +2 -0
- continuum_toolkit-1.0.0.dist-info/licenses/LICENSE +21 -0
- continuum_toolkit-1.0.0.dist-info/top_level.txt +13 -0
- contradictions/__init__.py +19 -0
- contradictions/detector.py +442 -0
- contradictions/models.py +75 -0
- core/__init__.py +97 -0
- core/enums.py +130 -0
- core/evidence.py +117 -0
- core/interfaces.py +209 -0
- core/schema.py +391 -0
- core/serializer.py +84 -0
- core/state_models.py +530 -0
- daemon/__init__.py +12 -0
- daemon/service.py +170 -0
- extractors/__init__.py +51 -0
- extractors/base.py +117 -0
- extractors/config/parsers.py +288 -0
- extractors/config/secret_sanitizer.py +91 -0
- extractors/config_extractor.py +172 -0
- extractors/conversation/analyzers.py +193 -0
- extractors/conversation/models.py +148 -0
- extractors/conversation_extractor.py +166 -0
- extractors/git_extractor.py +305 -0
- extractors/parsers/base.py +91 -0
- extractors/parsers/comment_parser.py +51 -0
- extractors/parsers/js_ts_parser.py +171 -0
- extractors/parsers/python_parser.py +180 -0
- extractors/verification/runners.py +278 -0
- extractors/verification_extractor.py +263 -0
- extractors/workspace_extractor.py +221 -0
- graph/__init__.py +24 -0
- graph/diff.py +109 -0
- graph/manager.py +473 -0
- graph/models.py +62 -0
- graph/propagator.py +194 -0
- graph/query.py +86 -0
- graph/snapshot.py +85 -0
- handoff/__init__.py +28 -0
- handoff/adapters/__init__.py +45 -0
- handoff/adapters/base.py +90 -0
- handoff/adapters/claude_adapter.py +176 -0
- handoff/adapters/codex_gpt_adapter.py +151 -0
- handoff/adapters/gemini_adapter.py +151 -0
- handoff/adapters/local_model_adapter.py +130 -0
- handoff/models.py +88 -0
- handoff/packager.py +288 -0
- pipeline/__init__.py +9 -0
- pipeline/orchestrator.py +260 -0
- resolution/__init__.py +15 -0
- resolution/resolver.py +311 -0
- storage/__init__.py +18 -0
- storage/hooks.py +125 -0
- storage/manager.py +127 -0
- storage/models.py +39 -0
- storage/recovery.py +98 -0
- watcher/__init__.py +17 -0
- watcher/detector.py +136 -0
- watcher/models.py +62 -0
- watcher/updater.py +163 -0
context/pruner.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Project Continuum - Context Pruning & Token Budget Optimization
|
|
3
|
+
===============================================================
|
|
4
|
+
Milestone 5 - Phase 13: Context Pruning & Token Budget Optimization.
|
|
5
|
+
Implements greedy priority-tiered context pruning under configurable token budgets.
|
|
6
|
+
Guarantees that active architectural constraints and blockers are never silently dropped.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field, asdict
|
|
10
|
+
from enum import Enum
|
|
11
|
+
import math
|
|
12
|
+
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
13
|
+
|
|
14
|
+
from context.models import TaskContext
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ContextPriorityTier(int, Enum):
|
|
18
|
+
"""Priority levels for task context retention."""
|
|
19
|
+
TIER_1_CRITICAL_CONSTRAINTS = 1 # Active architectural constraints, blockers, goal (NEVER dropped)
|
|
20
|
+
TIER_2_PRIMARY_ENTITIES = 2 # Primary components, failing tests, key interfaces
|
|
21
|
+
TIER_3_DIRECT_DEPENDENCIES = 3 # Direct prerequisite components, passing tests
|
|
22
|
+
TIER_4_SOURCE_SNIPPETS = 4 # Detailed symbol line spans & secondary files
|
|
23
|
+
TIER_5_HISTORICAL_CONTEXT = 5 # Architectural background narratives, secondary decisions
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class PrunedContextResult:
|
|
28
|
+
"""
|
|
29
|
+
Result of budget-constrained context optimization.
|
|
30
|
+
"""
|
|
31
|
+
task_description: str
|
|
32
|
+
token_budget: Optional[int]
|
|
33
|
+
estimated_tokens_before: int
|
|
34
|
+
estimated_tokens_after: int
|
|
35
|
+
reduction_percentage: float
|
|
36
|
+
retained_sections: List[str] = field(default_factory=list)
|
|
37
|
+
omitted_items: List[Dict[str, str]] = field(default_factory=list)
|
|
38
|
+
constraint_preservation_guarantee: bool = True
|
|
39
|
+
formatted_prompt: str = ""
|
|
40
|
+
|
|
41
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
42
|
+
return asdict(self)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ContextPruner:
|
|
46
|
+
"""
|
|
47
|
+
Optimizes and prunes TaskContext to fit within a target token budget while
|
|
48
|
+
strictly preserving Tier 1 critical constraints and reporting all omissions.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
DEFAULT_CHAR_PER_TOKEN = 3.8 # Standard heuristic for code + English markdown
|
|
52
|
+
|
|
53
|
+
def __init__(self, char_per_token: float = DEFAULT_CHAR_PER_TOKEN):
|
|
54
|
+
self._char_per_token = char_per_token
|
|
55
|
+
|
|
56
|
+
def estimate_tokens(self, text: str) -> int:
|
|
57
|
+
"""Estimates token count from text using character ratio heuristic."""
|
|
58
|
+
if not text:
|
|
59
|
+
return 0
|
|
60
|
+
return math.ceil(len(text) / self._char_per_token)
|
|
61
|
+
|
|
62
|
+
def prune_to_budget(
|
|
63
|
+
self,
|
|
64
|
+
task_context: TaskContext,
|
|
65
|
+
token_budget: Optional[int] = None
|
|
66
|
+
) -> PrunedContextResult:
|
|
67
|
+
"""
|
|
68
|
+
Prunes context items based on priority tiers until within token_budget.
|
|
69
|
+
Tier 1 (Constraints & Blockers) is strictly protected.
|
|
70
|
+
"""
|
|
71
|
+
full_markdown = task_context.to_markdown()
|
|
72
|
+
initial_tokens = self.estimate_tokens(full_markdown)
|
|
73
|
+
|
|
74
|
+
if token_budget is None or initial_tokens <= token_budget:
|
|
75
|
+
return PrunedContextResult(
|
|
76
|
+
task_description=task_context.task_description,
|
|
77
|
+
token_budget=token_budget,
|
|
78
|
+
estimated_tokens_before=initial_tokens,
|
|
79
|
+
estimated_tokens_after=initial_tokens,
|
|
80
|
+
reduction_percentage=0.0,
|
|
81
|
+
retained_sections=["all"],
|
|
82
|
+
omitted_items=[],
|
|
83
|
+
constraint_preservation_guarantee=True,
|
|
84
|
+
formatted_prompt=full_markdown
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
# Token budget is constrained: selectively assemble from Tier 1 to Tier 5
|
|
88
|
+
omitted_items: List[Dict[str, str]] = []
|
|
89
|
+
retained_sections: List[str] = []
|
|
90
|
+
|
|
91
|
+
# Tier 1 (Mandatory - Critical Constraints & Blockers)
|
|
92
|
+
tier1_lines = [
|
|
93
|
+
f"# Task Briefing: {task_context.task_description}",
|
|
94
|
+
""
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
if task_context.active_constraints:
|
|
98
|
+
tier1_lines.append("## 🛡️ Active Constraints (CRITICAL)")
|
|
99
|
+
for c in task_context.active_constraints:
|
|
100
|
+
tier1_lines.append(f"- {c}")
|
|
101
|
+
retained_sections.append("constraints")
|
|
102
|
+
|
|
103
|
+
if task_context.known_blockers:
|
|
104
|
+
tier1_lines.append("\n## ⚠️ Active Blockers & Discrepancies")
|
|
105
|
+
for b in task_context.known_blockers:
|
|
106
|
+
tier1_lines.append(f"- **[{b.severity}]**: {b.explanation}")
|
|
107
|
+
retained_sections.append("blockers")
|
|
108
|
+
|
|
109
|
+
current_text = "\n".join(tier1_lines)
|
|
110
|
+
current_tokens = self.estimate_tokens(current_text)
|
|
111
|
+
|
|
112
|
+
# Tier 2 (Primary Components & Failing Tests)
|
|
113
|
+
tier2_lines = []
|
|
114
|
+
if task_context.primary_nodes:
|
|
115
|
+
tier2_lines.append("\n## 🎯 Primary Components")
|
|
116
|
+
for n in task_context.primary_nodes:
|
|
117
|
+
tier2_lines.append(f"- **{n.name}** (`{n.node_type.value}`): Status `[{n.status.value}]` (Confidence: {n.confidence_score:.0f}%)")
|
|
118
|
+
|
|
119
|
+
failing_tests = [t for t in task_context.relevant_tests if t.status.value != "VERIFIED"]
|
|
120
|
+
if failing_tests:
|
|
121
|
+
tier2_lines.append("\n## ❌ Failing Test Evidence")
|
|
122
|
+
for t in failing_tests:
|
|
123
|
+
tier2_lines.append(f"- ❌ **{t.name}** (`{t.suite}`): exit code {t.exit_code} - `{t.error_message or ''}`")
|
|
124
|
+
|
|
125
|
+
tier2_text = "\n".join(tier2_lines)
|
|
126
|
+
if current_tokens + self.estimate_tokens(tier2_text) <= token_budget:
|
|
127
|
+
current_text += tier2_text
|
|
128
|
+
current_tokens = self.estimate_tokens(current_text)
|
|
129
|
+
retained_sections.append("primary_components")
|
|
130
|
+
else:
|
|
131
|
+
omitted_items.append({
|
|
132
|
+
"tier": "Tier 2",
|
|
133
|
+
"item": "Primary components detailed list",
|
|
134
|
+
"reason": "Exceeded token budget"
|
|
135
|
+
})
|
|
136
|
+
|
|
137
|
+
# Tier 3 (Direct Dependencies & Passing Tests)
|
|
138
|
+
tier3_lines = []
|
|
139
|
+
if task_context.dependency_nodes:
|
|
140
|
+
tier3_lines.append("\n## 🔗 Direct Prerequisites")
|
|
141
|
+
for d in task_context.dependency_nodes:
|
|
142
|
+
tier3_lines.append(f"- **{d.name}** (`{d.node_type.value}`): Status `[{d.status.value}]`")
|
|
143
|
+
|
|
144
|
+
passing_tests = [t for t in task_context.relevant_tests if t.status.value == "VERIFIED"]
|
|
145
|
+
if passing_tests:
|
|
146
|
+
tier3_lines.append("\n## ✅ Verified Test Baseline")
|
|
147
|
+
for t in passing_tests[:3]: # Top passing tests
|
|
148
|
+
tier3_lines.append(f"- ✅ **{t.name}** (`{t.suite}`)")
|
|
149
|
+
|
|
150
|
+
tier3_text = "\n".join(tier3_lines)
|
|
151
|
+
if current_tokens + self.estimate_tokens(tier3_text) <= token_budget:
|
|
152
|
+
current_text += tier3_text
|
|
153
|
+
current_tokens = self.estimate_tokens(current_text)
|
|
154
|
+
retained_sections.append("dependencies")
|
|
155
|
+
else:
|
|
156
|
+
if task_context.dependency_nodes:
|
|
157
|
+
omitted_items.append({
|
|
158
|
+
"tier": "Tier 3",
|
|
159
|
+
"item": f"{len(task_context.dependency_nodes)} dependency nodes",
|
|
160
|
+
"reason": "Omitted to fit token budget"
|
|
161
|
+
})
|
|
162
|
+
|
|
163
|
+
# Tier 4 (Source Code Symbols & Line Spans)
|
|
164
|
+
tier4_lines = []
|
|
165
|
+
if task_context.relevant_symbols:
|
|
166
|
+
tier4_lines.append("\n## 🧬 Relevant Symbols")
|
|
167
|
+
for s in task_context.relevant_symbols[:5]:
|
|
168
|
+
tier4_lines.append(f"- `{s.kind} {s.name}` in `{s.file_path}` (lines {s.line_start}-{s.line_end})")
|
|
169
|
+
|
|
170
|
+
tier4_text = "\n".join(tier4_lines)
|
|
171
|
+
if current_tokens + self.estimate_tokens(tier4_text) <= token_budget:
|
|
172
|
+
current_text += tier4_text
|
|
173
|
+
current_tokens = self.estimate_tokens(current_text)
|
|
174
|
+
retained_sections.append("symbols")
|
|
175
|
+
else:
|
|
176
|
+
if task_context.relevant_symbols:
|
|
177
|
+
omitted_items.append({
|
|
178
|
+
"tier": "Tier 4",
|
|
179
|
+
"item": f"{len(task_context.relevant_symbols)} code symbols",
|
|
180
|
+
"reason": "Omitted to fit token budget"
|
|
181
|
+
})
|
|
182
|
+
|
|
183
|
+
# Tier 5 (Background Decisions & Historical Narrative)
|
|
184
|
+
tier5_lines = []
|
|
185
|
+
if task_context.relevant_decisions:
|
|
186
|
+
tier5_lines.append("\n## 📐 Architectural Decisions")
|
|
187
|
+
for d in task_context.relevant_decisions:
|
|
188
|
+
tier5_lines.append(f"- **{d.title}**: {d.rationale}")
|
|
189
|
+
|
|
190
|
+
tier5_text = "\n".join(tier5_lines)
|
|
191
|
+
if current_tokens + self.estimate_tokens(tier5_text) <= token_budget:
|
|
192
|
+
current_text += tier5_text
|
|
193
|
+
current_tokens = self.estimate_tokens(current_text)
|
|
194
|
+
retained_sections.append("decisions")
|
|
195
|
+
else:
|
|
196
|
+
if task_context.relevant_decisions:
|
|
197
|
+
omitted_items.append({
|
|
198
|
+
"tier": "Tier 5",
|
|
199
|
+
"item": f"{len(task_context.relevant_decisions)} architectural decisions",
|
|
200
|
+
"reason": "Omitted to fit token budget"
|
|
201
|
+
})
|
|
202
|
+
|
|
203
|
+
# Add explicit omission ledger footer
|
|
204
|
+
footer_lines = [
|
|
205
|
+
"",
|
|
206
|
+
"---",
|
|
207
|
+
f"**[Continuum Context Budget Audit: {current_tokens} tokens (Budget: {token_budget}) | {len(omitted_items)} section(s) pruned]**"
|
|
208
|
+
]
|
|
209
|
+
if omitted_items:
|
|
210
|
+
footer_lines.append("**Omitted Items due to budget constraints:**")
|
|
211
|
+
for o in omitted_items:
|
|
212
|
+
footer_lines.append(f"- *{o['tier']}*: {o['item']} ({o['reason']})")
|
|
213
|
+
|
|
214
|
+
current_text += "\n".join(footer_lines)
|
|
215
|
+
final_tokens = self.estimate_tokens(current_text)
|
|
216
|
+
reduction = round(max(0.0, ((initial_tokens - final_tokens) / initial_tokens) * 100.0), 1)
|
|
217
|
+
|
|
218
|
+
return PrunedContextResult(
|
|
219
|
+
task_description=task_context.task_description,
|
|
220
|
+
token_budget=token_budget,
|
|
221
|
+
estimated_tokens_before=initial_tokens,
|
|
222
|
+
estimated_tokens_after=final_tokens,
|
|
223
|
+
reduction_percentage=reduction,
|
|
224
|
+
retained_sections=retained_sections,
|
|
225
|
+
omitted_items=omitted_items,
|
|
226
|
+
constraint_preservation_guarantee=True,
|
|
227
|
+
formatted_prompt=current_text
|
|
228
|
+
)
|
context/selector.py
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Project Continuum - Task-Driven Context Selection Engine
|
|
3
|
+
========================================================
|
|
4
|
+
Milestone 5 - Phase 12: Task-Driven Context Selection.
|
|
5
|
+
Identifies and extracts the minimum sufficient, high-signal project slice
|
|
6
|
+
required to execute a specific software development task.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from typing import Any, Dict, List, Optional, Set, Tuple
|
|
11
|
+
|
|
12
|
+
from core.enums import NodeType, RelationType, Status
|
|
13
|
+
from core.interfaces import IContextSelector
|
|
14
|
+
from core.state_models import (
|
|
15
|
+
ArchitecturalDecision,
|
|
16
|
+
AstSymbol,
|
|
17
|
+
CanonicalProjectState,
|
|
18
|
+
ContradictionRecord,
|
|
19
|
+
GraphNode,
|
|
20
|
+
TestResult,
|
|
21
|
+
)
|
|
22
|
+
from context.models import TaskContext
|
|
23
|
+
from graph.manager import StateGraphManager
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class TaskContextSelector(IContextSelector):
|
|
27
|
+
"""
|
|
28
|
+
Selects task-specific project context from CanonicalProjectState and DAG.
|
|
29
|
+
Implements IContextSelector Protocol.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
STOP_WORDS = {
|
|
33
|
+
"a", "an", "the", "and", "or", "in", "on", "at", "to", "for", "with",
|
|
34
|
+
"by", "of", "is", "are", "be", "this", "that", "it", "from", "as", "fix",
|
|
35
|
+
"add", "update", "implement", "create", "make", "refactor", "check", "test"
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
def select_context_for_task(
|
|
39
|
+
self,
|
|
40
|
+
task_description: str,
|
|
41
|
+
canonical_state: CanonicalProjectState,
|
|
42
|
+
token_budget: Optional[int] = None
|
|
43
|
+
) -> Dict[str, Any]:
|
|
44
|
+
"""
|
|
45
|
+
Extracts task-relevant context slice and applies token budget optimization if requested.
|
|
46
|
+
Implements IContextSelector interface.
|
|
47
|
+
"""
|
|
48
|
+
task_ctx = self.extract_task_context(task_description, canonical_state, token_budget)
|
|
49
|
+
if token_budget is not None:
|
|
50
|
+
from context.pruner import ContextPruner
|
|
51
|
+
pruner = ContextPruner()
|
|
52
|
+
pruned_res = pruner.prune_to_budget(task_ctx, token_budget=token_budget)
|
|
53
|
+
d = task_ctx.to_dict()
|
|
54
|
+
d["pruning"] = pruned_res.to_dict()
|
|
55
|
+
return d
|
|
56
|
+
|
|
57
|
+
return task_ctx.to_dict()
|
|
58
|
+
|
|
59
|
+
def extract_task_context(
|
|
60
|
+
self,
|
|
61
|
+
task_description: str,
|
|
62
|
+
canonical_state: CanonicalProjectState,
|
|
63
|
+
token_budget: Optional[int] = None
|
|
64
|
+
) -> TaskContext:
|
|
65
|
+
"""
|
|
66
|
+
Returns strongly-typed TaskContext object.
|
|
67
|
+
"""
|
|
68
|
+
# Build state graph manager
|
|
69
|
+
graph_mgr = StateGraphManager(canonical_state)
|
|
70
|
+
if not graph_mgr.nodes:
|
|
71
|
+
graph_mgr.build_from_canonical_state(canonical_state)
|
|
72
|
+
|
|
73
|
+
# 1. Extract keywords and semantic tokens from task prompt
|
|
74
|
+
keywords = self._extract_keywords(task_description)
|
|
75
|
+
|
|
76
|
+
# 2. Identify Primary Graph Nodes
|
|
77
|
+
primary_nodes: List[GraphNode] = []
|
|
78
|
+
matched_node_ids: Set[str] = set()
|
|
79
|
+
|
|
80
|
+
for node in graph_mgr.nodes.values():
|
|
81
|
+
node_name_clean = node.name.lower()
|
|
82
|
+
if any(kw in node_name_clean for kw in keywords) or any(kw in node.id.lower() for kw in keywords):
|
|
83
|
+
primary_nodes.append(node)
|
|
84
|
+
matched_node_ids.add(node.id)
|
|
85
|
+
|
|
86
|
+
# If no nodes matched directly, fall back to matching symbols or requirements
|
|
87
|
+
if not primary_nodes and graph_mgr.nodes:
|
|
88
|
+
primary_nodes = list(graph_mgr.nodes.values())[:3]
|
|
89
|
+
for n in primary_nodes:
|
|
90
|
+
matched_node_ids.add(n.id)
|
|
91
|
+
|
|
92
|
+
# 3. Locate Direct Dependencies & Prerequisites
|
|
93
|
+
dependency_nodes: List[GraphNode] = []
|
|
94
|
+
dep_ids: Set[str] = set()
|
|
95
|
+
|
|
96
|
+
for p_node in primary_nodes:
|
|
97
|
+
# Direct dependencies
|
|
98
|
+
deps = graph_mgr.get_dependencies(p_node.id, recursive=False)
|
|
99
|
+
for d in deps:
|
|
100
|
+
if d.id not in matched_node_ids and d.id not in dep_ids:
|
|
101
|
+
dependency_nodes.append(d)
|
|
102
|
+
dep_ids.add(d.id)
|
|
103
|
+
|
|
104
|
+
# 4. Filter Relevant AST Symbols and Physical Files
|
|
105
|
+
relevant_symbols: List[AstSymbol] = []
|
|
106
|
+
relevant_files: Set[str] = set()
|
|
107
|
+
|
|
108
|
+
all_target_names = {n.name.lower() for n in primary_nodes + dependency_nodes}
|
|
109
|
+
all_target_names.update(keywords)
|
|
110
|
+
|
|
111
|
+
for sym in canonical_state.project_state.symbols:
|
|
112
|
+
sym_lower = sym.name.lower()
|
|
113
|
+
if any(t in sym_lower or sym_lower in t for t in all_target_names):
|
|
114
|
+
relevant_symbols.append(sym)
|
|
115
|
+
relevant_files.add(sym.file_path)
|
|
116
|
+
|
|
117
|
+
for f in canonical_state.project_state.files:
|
|
118
|
+
f_lower = f.lower()
|
|
119
|
+
if any(kw in f_lower for kw in keywords):
|
|
120
|
+
relevant_files.add(f)
|
|
121
|
+
|
|
122
|
+
# 5. Filter Relevant Test Results & Suites
|
|
123
|
+
relevant_tests: List[TestResult] = []
|
|
124
|
+
for t in canonical_state.project_state.test_results:
|
|
125
|
+
t_lower = (t.name + " " + t.suite).lower()
|
|
126
|
+
if any(t_name in t_lower for t_name in all_target_names):
|
|
127
|
+
relevant_tests.append(t)
|
|
128
|
+
elif t.status == Status.FAILED:
|
|
129
|
+
# Include failing tests if they might relate
|
|
130
|
+
relevant_tests.append(t)
|
|
131
|
+
|
|
132
|
+
# Deduplicate tests
|
|
133
|
+
unique_tests = []
|
|
134
|
+
seen_test_ids = set()
|
|
135
|
+
for t in relevant_tests:
|
|
136
|
+
if t.test_id not in seen_test_ids:
|
|
137
|
+
seen_test_ids.add(t.test_id)
|
|
138
|
+
unique_tests.append(t)
|
|
139
|
+
|
|
140
|
+
# 6. Locate Relevant Known Blockers & Contradictions
|
|
141
|
+
relevant_blockers: List[ContradictionRecord] = []
|
|
142
|
+
for c in canonical_state.contradictions:
|
|
143
|
+
if not c.resolved:
|
|
144
|
+
c_text = (c.claim_text + " " + c.explanation).lower()
|
|
145
|
+
if any(t in c_text for t in all_target_names):
|
|
146
|
+
relevant_blockers.append(c)
|
|
147
|
+
|
|
148
|
+
# 7. Extract Architectural Decisions & Active Constraints
|
|
149
|
+
relevant_decisions: List[ArchitecturalDecision] = []
|
|
150
|
+
active_constraints: List[str] = []
|
|
151
|
+
|
|
152
|
+
for d in canonical_state.conversational_state.architectural_decisions:
|
|
153
|
+
d_text = (d.title + " " + d.rationale).lower()
|
|
154
|
+
if any(kw in d_text for kw in keywords):
|
|
155
|
+
relevant_decisions.append(d)
|
|
156
|
+
active_constraints.extend(d.constraints)
|
|
157
|
+
|
|
158
|
+
# 8. Compute Omitted Node Count
|
|
159
|
+
total_graph_nodes = len(graph_mgr.nodes)
|
|
160
|
+
included_count = len(matched_node_ids) + len(dep_ids)
|
|
161
|
+
omitted_count = max(0, total_graph_nodes - included_count)
|
|
162
|
+
|
|
163
|
+
verification_summary = {
|
|
164
|
+
"primary_node_count": len(primary_nodes),
|
|
165
|
+
"dependency_node_count": len(dependency_nodes),
|
|
166
|
+
"relevant_tests_count": len(unique_tests),
|
|
167
|
+
"failing_tests_count": sum(1 for t in unique_tests if t.status == Status.FAILED),
|
|
168
|
+
"active_blockers_count": len(relevant_blockers),
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
return TaskContext(
|
|
172
|
+
task_description=task_description,
|
|
173
|
+
primary_nodes=primary_nodes,
|
|
174
|
+
dependency_nodes=dependency_nodes,
|
|
175
|
+
relevant_files=sorted(list(relevant_files)),
|
|
176
|
+
relevant_symbols=relevant_symbols,
|
|
177
|
+
relevant_tests=unique_tests,
|
|
178
|
+
active_constraints=active_constraints,
|
|
179
|
+
relevant_decisions=relevant_decisions,
|
|
180
|
+
known_blockers=relevant_blockers,
|
|
181
|
+
verification_status_summary=verification_summary,
|
|
182
|
+
omitted_node_count=omitted_count
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
def _extract_keywords(self, text: str) -> List[str]:
|
|
186
|
+
"""Extracts significant keywords and identifier tokens from prompt text."""
|
|
187
|
+
# Find alphanumeric words with length >= 3
|
|
188
|
+
words = re.findall(r"[A-Za-z0-9_]{3,}", text)
|
|
189
|
+
keywords = [
|
|
190
|
+
w.lower() for w in words
|
|
191
|
+
if w.lower() not in self.STOP_WORDS
|
|
192
|
+
]
|
|
193
|
+
return list(dict.fromkeys(keywords)) # Preserve order deduplication
|