skeletongraph 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- skeletongraph/__init__.py +35 -0
- skeletongraph/__main__.py +6 -0
- skeletongraph/assembly/__init__.py +0 -0
- skeletongraph/assembly/constraint_store.py +301 -0
- skeletongraph/assembly/context_routing.py +173 -0
- skeletongraph/assembly/modifier.py +130 -0
- skeletongraph/assembly/prompt_builder.py +791 -0
- skeletongraph/build.py +704 -0
- skeletongraph/cli/__init__.py +0 -0
- skeletongraph/cli/init.py +350 -0
- skeletongraph/cli/main.py +2050 -0
- skeletongraph/cli/prepare.py +208 -0
- skeletongraph/cli/run_exec.py +161 -0
- skeletongraph/config.py +576 -0
- skeletongraph/daemon.py +116 -0
- skeletongraph/engine.py +740 -0
- skeletongraph/eval/__init__.py +1 -0
- skeletongraph/eval/token_counter.py +194 -0
- skeletongraph/graph/__init__.py +0 -0
- skeletongraph/graph/bloom.py +125 -0
- skeletongraph/graph/bm25.py +136 -0
- skeletongraph/graph/dependency.py +415 -0
- skeletongraph/graph/embeddings.py +458 -0
- skeletongraph/graph/inverted_index.py +403 -0
- skeletongraph/graph/pagerank.py +131 -0
- skeletongraph/hooks/__init__.py +1 -0
- skeletongraph/hooks/claude_code.py +433 -0
- skeletongraph/install/__init__.py +11 -0
- skeletongraph/install/claude_code.py +310 -0
- skeletongraph/install/cursor.py +173 -0
- skeletongraph/install/detect.py +111 -0
- skeletongraph/install/mcp_only.py +181 -0
- skeletongraph/llm/__init__.py +0 -0
- skeletongraph/llm/provider.py +138 -0
- skeletongraph/llm/summarizer.py +208 -0
- skeletongraph/parser/__init__.py +0 -0
- skeletongraph/parser/ast_extractor.py +389 -0
- skeletongraph/parser/edge_extractor.py +295 -0
- skeletongraph/parser/import_resolver.py +268 -0
- skeletongraph/parser/languages/__init__.py +0 -0
- skeletongraph/parser/languages/cpp.py +250 -0
- skeletongraph/parser/languages/csharp.py +201 -0
- skeletongraph/parser/languages/go.py +265 -0
- skeletongraph/parser/languages/java.py +215 -0
- skeletongraph/parser/languages/php.py +192 -0
- skeletongraph/parser/languages/python.py +548 -0
- skeletongraph/parser/languages/ruby.py +173 -0
- skeletongraph/parser/languages/rust.py +231 -0
- skeletongraph/parser/languages/typescript.py +782 -0
- skeletongraph/parser/node_kinds.py +114 -0
- skeletongraph/parser/skeleton.py +364 -0
- skeletongraph/retrieval/__init__.py +10 -0
- skeletongraph/retrieval/bm25_flat.py +187 -0
- skeletongraph/retrieval/budget.py +132 -0
- skeletongraph/retrieval/classifier.py +554 -0
- skeletongraph/retrieval/confidence.py +295 -0
- skeletongraph/retrieval/dense.py +227 -0
- skeletongraph/retrieval/detect_changes.py +229 -0
- skeletongraph/retrieval/fusion.py +192 -0
- skeletongraph/retrieval/intent.py +350 -0
- skeletongraph/retrieval/model_router.py +109 -0
- skeletongraph/retrieval/ranker.py +142 -0
- skeletongraph/retrieval/resolver.py +913 -0
- skeletongraph/retrieval/session.py +308 -0
- skeletongraph/server/__init__.py +0 -0
- skeletongraph/server/mcp.py +2223 -0
- skeletongraph/session/__init__.py +1 -0
- skeletongraph/session/decision_log.py +163 -0
- skeletongraph/session/log.py +111 -0
- skeletongraph/storage/__init__.py +0 -0
- skeletongraph/storage/dirty.py +215 -0
- skeletongraph/storage/local.py +312 -0
- skeletongraph/storage/staleness.py +109 -0
- skeletongraph/summary/__init__.py +49 -0
- skeletongraph/summary/local.py +117 -0
- skeletongraph/summary/ollama.py +264 -0
- skeletongraph/summary/queue.py +356 -0
- skeletongraph/summary/summary_store.py +116 -0
- skeletongraph-0.1.0.dist-info/METADATA +540 -0
- skeletongraph-0.1.0.dist-info/RECORD +83 -0
- skeletongraph-0.1.0.dist-info/WHEEL +4 -0
- skeletongraph-0.1.0.dist-info/entry_points.txt +3 -0
- skeletongraph-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""
|
|
2
|
+
SkeletonGraph — Token-minimal, constraint-preserving context assembly for AI coding agents.
|
|
3
|
+
|
|
4
|
+
Quick start:
|
|
5
|
+
from skeletongraph import SGEngine
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
engine = SGEngine(Path("."))
|
|
9
|
+
result = engine.query("fix validate_token")
|
|
10
|
+
print(result.context_text)
|
|
11
|
+
|
|
12
|
+
Features:
|
|
13
|
+
- 5-layer attention-aware SLM-orchestrated context assembly
|
|
14
|
+
- Elastic token budget with progressive compression
|
|
15
|
+
- Session memory for cross-turn context deduplication
|
|
16
|
+
- Per-directory constraint scoping
|
|
17
|
+
- 10-language support via Tree-sitter
|
|
18
|
+
- MCP server with 11 tools for IDE integration
|
|
19
|
+
- PR blast-radius analysis with risk scoring
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
__version__ = "0.1.0"
|
|
23
|
+
|
|
24
|
+
from .build import build_index, update_index
|
|
25
|
+
from .engine import SGEngine
|
|
26
|
+
from .config import SGConfig, load_config
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"build_index",
|
|
30
|
+
"update_index",
|
|
31
|
+
"SGEngine",
|
|
32
|
+
"SGConfig",
|
|
33
|
+
"load_config",
|
|
34
|
+
"__version__",
|
|
35
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Constraint store: hierarchical scoping + IDE rule aggregation + propose/confirm.
|
|
3
|
+
|
|
4
|
+
Sources (three paths per the plan):
|
|
5
|
+
1. sg init --constraints "..." (init-arg)
|
|
6
|
+
2. IDE rule files aggregated at index time → constraints.md
|
|
7
|
+
3. Model-driven proposals via sg_constraint(action="propose")
|
|
8
|
+
|
|
9
|
+
Constraints are stored in .skeletongraph/constraints.md with structured markers
|
|
10
|
+
so SG can manage individual items (confirm/remove) while keeping the file
|
|
11
|
+
human-readable and editable.
|
|
12
|
+
|
|
13
|
+
Never strictly enforced — Zone 1 visibility is the mechanism.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import re
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
import time
|
|
22
|
+
from pathlib import Path, PurePosixPath
|
|
23
|
+
from typing import Dict, List, Optional
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# ── Constraint dataclass ─────────────────────────────────────────────────
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Constraint:
|
|
31
|
+
"""Single constraint entry."""
|
|
32
|
+
id: str
|
|
33
|
+
text: str
|
|
34
|
+
provenance: str # e.g. "CLAUDE.md", "init-arg", "model-proposed"
|
|
35
|
+
confirmed: bool = True
|
|
36
|
+
|
|
37
|
+
def to_block(self) -> str:
|
|
38
|
+
confirmed_str = "true" if self.confirmed else "false"
|
|
39
|
+
return (
|
|
40
|
+
f"<!-- sg:constraint id={self.id} confirmed={confirmed_str}"
|
|
41
|
+
f" provenance={self.provenance} -->\n"
|
|
42
|
+
f"{self.text.strip()}\n"
|
|
43
|
+
f"<!-- /sg:constraint -->"
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ── ConstraintStore ──────────────────────────────────────────────────────
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ConstraintStore:
|
|
51
|
+
"""Hierarchical constraint manager with IDE rule aggregation.
|
|
52
|
+
|
|
53
|
+
Load once at build time, query per-file at assembly time.
|
|
54
|
+
The existing load()/get_constraints_for_file()/get_all_constraints() interface
|
|
55
|
+
is preserved for storage/local.py compatibility.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(self) -> None:
|
|
59
|
+
self._global_constraints: str = ""
|
|
60
|
+
self._scoped_constraints: Dict[str, str] = {} # dir_path → raw text
|
|
61
|
+
self._items: List[Constraint] = [] # structured items
|
|
62
|
+
|
|
63
|
+
# ── Load / Save ──────────────────────────────────────────────────────
|
|
64
|
+
|
|
65
|
+
def load(self, project_root: Path) -> None:
|
|
66
|
+
"""Scan project for constraint files and load them.
|
|
67
|
+
|
|
68
|
+
Looks for:
|
|
69
|
+
- {project_root}/.skeletongraph/constraints.md (global)
|
|
70
|
+
- {project_root}/<any_dir>/.skeletongraph/constraints.md (scoped)
|
|
71
|
+
"""
|
|
72
|
+
global_file = project_root / ".skeletongraph" / "constraints.md"
|
|
73
|
+
if global_file.exists():
|
|
74
|
+
raw = global_file.read_text(encoding="utf-8", errors="replace").strip()
|
|
75
|
+
self._items = _parse_items(raw)
|
|
76
|
+
self._global_constraints = raw
|
|
77
|
+
|
|
78
|
+
for constraints_file in project_root.rglob(".skeletongraph/constraints.md"):
|
|
79
|
+
if constraints_file == global_file:
|
|
80
|
+
continue
|
|
81
|
+
sg_dir = constraints_file.parent
|
|
82
|
+
scope_dir = sg_dir.parent
|
|
83
|
+
rel_scope = scope_dir.relative_to(project_root).as_posix()
|
|
84
|
+
self._scoped_constraints[rel_scope] = constraints_file.read_text(
|
|
85
|
+
encoding="utf-8", errors="replace"
|
|
86
|
+
).strip()
|
|
87
|
+
|
|
88
|
+
def save_global(self, project_root: Path) -> None:
|
|
89
|
+
"""Persist global constraints to disk."""
|
|
90
|
+
sg_dir = project_root / ".skeletongraph"
|
|
91
|
+
sg_dir.mkdir(parents=True, exist_ok=True)
|
|
92
|
+
target = sg_dir / "constraints.md"
|
|
93
|
+
# Rebuild from structured items + any free-form preamble
|
|
94
|
+
blocks = [item.to_block() for item in self._items]
|
|
95
|
+
target.write_text("\n\n".join(blocks) + "\n", encoding="utf-8")
|
|
96
|
+
self._global_constraints = target.read_text(encoding="utf-8")
|
|
97
|
+
|
|
98
|
+
# ── Query interface (used by assembly) ───────────────────────────────
|
|
99
|
+
|
|
100
|
+
def get_constraints_for_file(self, file_path: str) -> str:
|
|
101
|
+
"""Get merged constraints for a specific file (global + scoped)."""
|
|
102
|
+
parts = []
|
|
103
|
+
|
|
104
|
+
confirmed_text = self._confirmed_text()
|
|
105
|
+
if confirmed_text:
|
|
106
|
+
parts.append(confirmed_text)
|
|
107
|
+
|
|
108
|
+
file_dir = str(PurePosixPath(file_path).parent)
|
|
109
|
+
for scope_dir, constraints in sorted(self._scoped_constraints.items()):
|
|
110
|
+
if file_dir == scope_dir or file_dir.startswith(scope_dir + "/"):
|
|
111
|
+
parts.append(f"# [{scope_dir}/ scope]\n{constraints}")
|
|
112
|
+
|
|
113
|
+
return "\n\n".join(parts)
|
|
114
|
+
|
|
115
|
+
def get_all_constraints(self) -> str:
|
|
116
|
+
"""Get confirmed global constraints. Used when file scope is unknown."""
|
|
117
|
+
return self._confirmed_text() or self._global_constraints
|
|
118
|
+
|
|
119
|
+
def get_all_for_overview(self) -> str:
|
|
120
|
+
"""Human-readable list for sg_overview Zone 1, includes proposed."""
|
|
121
|
+
if not self._items:
|
|
122
|
+
return self._global_constraints
|
|
123
|
+
lines = []
|
|
124
|
+
for item in self._items:
|
|
125
|
+
prefix = "✓" if item.confirmed else "?"
|
|
126
|
+
lines.append(f"[{prefix}] ({item.provenance}) {item.text.strip()}")
|
|
127
|
+
return "\n".join(lines)
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def has_constraints(self) -> bool:
|
|
131
|
+
return bool(self._global_constraints) or bool(self._scoped_constraints)
|
|
132
|
+
|
|
133
|
+
@property
|
|
134
|
+
def scope_count(self) -> int:
|
|
135
|
+
return len(self._scoped_constraints)
|
|
136
|
+
|
|
137
|
+
# ── CRUD ─────────────────────────────────────────────────────────────
|
|
138
|
+
|
|
139
|
+
def add_constraint(
|
|
140
|
+
self,
|
|
141
|
+
text: str,
|
|
142
|
+
provenance: str = "manual",
|
|
143
|
+
confirmed: bool = True,
|
|
144
|
+
) -> Constraint:
|
|
145
|
+
"""Add a confirmed constraint (from init-arg or manual CLI)."""
|
|
146
|
+
cid = _make_id(text)
|
|
147
|
+
c = Constraint(id=cid, text=text.strip(), provenance=provenance, confirmed=confirmed)
|
|
148
|
+
self._items.append(c)
|
|
149
|
+
return c
|
|
150
|
+
|
|
151
|
+
def propose_constraint(self, text: str, source: str = "model-proposed") -> Constraint:
|
|
152
|
+
"""Add an unconfirmed proposal (from sg_constraint action=propose)."""
|
|
153
|
+
return self.add_constraint(text, provenance=source, confirmed=False)
|
|
154
|
+
|
|
155
|
+
def confirm_constraint(
|
|
156
|
+
self,
|
|
157
|
+
constraint_id: str,
|
|
158
|
+
project_root: Optional[Path] = None,
|
|
159
|
+
) -> bool:
|
|
160
|
+
"""Mark a proposed constraint as confirmed. Returns True if found.
|
|
161
|
+
|
|
162
|
+
If project_root is given, also promotes to decisions.md.
|
|
163
|
+
"""
|
|
164
|
+
for item in self._items:
|
|
165
|
+
if item.id == constraint_id or item.id.startswith(constraint_id):
|
|
166
|
+
item.confirmed = True
|
|
167
|
+
if project_root is not None:
|
|
168
|
+
_promote_to_decisions(item, project_root)
|
|
169
|
+
return True
|
|
170
|
+
return False
|
|
171
|
+
|
|
172
|
+
def remove_constraint(self, constraint_id: str) -> bool:
|
|
173
|
+
"""Remove a constraint by id. Returns True if found."""
|
|
174
|
+
before = len(self._items)
|
|
175
|
+
self._items = [
|
|
176
|
+
c for c in self._items
|
|
177
|
+
if not (c.id == constraint_id or c.id.startswith(constraint_id))
|
|
178
|
+
]
|
|
179
|
+
return len(self._items) < before
|
|
180
|
+
|
|
181
|
+
def list_constraints(self, include_proposed: bool = True) -> List[Constraint]:
|
|
182
|
+
"""Return all constraints, optionally filtering out proposals."""
|
|
183
|
+
if include_proposed:
|
|
184
|
+
return list(self._items)
|
|
185
|
+
return [c for c in self._items if c.confirmed]
|
|
186
|
+
|
|
187
|
+
# ── IDE rule aggregation ──────────────────────────────────────────────
|
|
188
|
+
|
|
189
|
+
def aggregate_from_ide_rules(self, project_root: Path) -> int:
|
|
190
|
+
"""Scan IDE rule files and import their content as confirmed constraints.
|
|
191
|
+
|
|
192
|
+
Files checked (in order):
|
|
193
|
+
CLAUDE.md, AGENTS.md, .github/copilot-instructions.md,
|
|
194
|
+
.windsurfrules, .roorules, .rules,
|
|
195
|
+
.cursor/rules/*.mdc (all matched)
|
|
196
|
+
|
|
197
|
+
Returns number of new constraints added.
|
|
198
|
+
"""
|
|
199
|
+
sources: List[tuple[str, Path]] = [
|
|
200
|
+
("CLAUDE.md", project_root / "CLAUDE.md"),
|
|
201
|
+
("AGENTS.md", project_root / "AGENTS.md"),
|
|
202
|
+
("copilot-instructions.md", project_root / ".github" / "copilot-instructions.md"),
|
|
203
|
+
(".windsurfrules", project_root / ".windsurfrules"),
|
|
204
|
+
(".roorules", project_root / ".roorules"),
|
|
205
|
+
(".rules", project_root / ".rules"),
|
|
206
|
+
]
|
|
207
|
+
|
|
208
|
+
# .cursor/rules/*.mdc — all files
|
|
209
|
+
cursor_rules_dir = project_root / ".cursor" / "rules"
|
|
210
|
+
if cursor_rules_dir.exists():
|
|
211
|
+
for mdc in sorted(cursor_rules_dir.glob("*.mdc")):
|
|
212
|
+
sources.append((f".cursor/rules/{mdc.name}", mdc))
|
|
213
|
+
|
|
214
|
+
added = 0
|
|
215
|
+
existing_texts = {c.text.strip() for c in self._items}
|
|
216
|
+
|
|
217
|
+
for provenance, path in sources:
|
|
218
|
+
if not path.exists():
|
|
219
|
+
continue
|
|
220
|
+
text = path.read_text(encoding="utf-8", errors="replace").strip()
|
|
221
|
+
if not text:
|
|
222
|
+
continue
|
|
223
|
+
# Deduplicate — skip if same text already stored
|
|
224
|
+
if text in existing_texts:
|
|
225
|
+
continue
|
|
226
|
+
c = Constraint(
|
|
227
|
+
id=_make_id(text),
|
|
228
|
+
text=text,
|
|
229
|
+
provenance=provenance,
|
|
230
|
+
confirmed=True,
|
|
231
|
+
)
|
|
232
|
+
self._items.append(c)
|
|
233
|
+
existing_texts.add(text)
|
|
234
|
+
added += 1
|
|
235
|
+
|
|
236
|
+
return added
|
|
237
|
+
|
|
238
|
+
# ── Legacy helpers ────────────────────────────────────────────────────
|
|
239
|
+
|
|
240
|
+
def set_global(self, text: str) -> None:
|
|
241
|
+
"""Set global constraints programmatically (for API/test use)."""
|
|
242
|
+
self._global_constraints = text.strip()
|
|
243
|
+
# Also parse items from provided text
|
|
244
|
+
self._items = _parse_items(text)
|
|
245
|
+
|
|
246
|
+
# ── Internal ──────────────────────────────────────────────────────────
|
|
247
|
+
|
|
248
|
+
def _confirmed_text(self) -> str:
|
|
249
|
+
confirmed = [c.text.strip() for c in self._items if c.confirmed]
|
|
250
|
+
return "\n\n".join(confirmed)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ── Helpers ──────────────────────────────────────────────────────────────
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _make_id(text: str) -> str:
|
|
257
|
+
"""Stable 8-char id from text hash."""
|
|
258
|
+
return hashlib.sha1(text.strip().encode()).hexdigest()[:8]
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
_BLOCK_RE = re.compile(
|
|
262
|
+
r"<!-- sg:constraint\s+id=(\S+)\s+confirmed=(\S+)\s+provenance=(\S+)\s*-->\n(.*?)\n<!-- /sg:constraint -->",
|
|
263
|
+
re.DOTALL,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _promote_to_decisions(item: Constraint, project_root: Path) -> None:
|
|
268
|
+
"""Append a confirmed constraint to decisions.md as a persistent decision record."""
|
|
269
|
+
sg_dir = project_root / ".skeletongraph"
|
|
270
|
+
sg_dir.mkdir(parents=True, exist_ok=True)
|
|
271
|
+
decisions_path = sg_dir / "decisions.md"
|
|
272
|
+
|
|
273
|
+
date_str = time.strftime("%Y-%m-%d")
|
|
274
|
+
entry = (
|
|
275
|
+
f"\n## [{item.id}] {date_str} — {item.provenance}\n\n"
|
|
276
|
+
f"{item.text.strip()}\n"
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
if decisions_path.exists():
|
|
280
|
+
existing = decisions_path.read_text(encoding="utf-8", errors="replace")
|
|
281
|
+
# Skip if already promoted
|
|
282
|
+
if item.id in existing:
|
|
283
|
+
return
|
|
284
|
+
decisions_path.write_text(existing.rstrip() + "\n" + entry, encoding="utf-8")
|
|
285
|
+
else:
|
|
286
|
+
header = "# Decisions\n\nPromoted constraints and architectural decisions.\n"
|
|
287
|
+
decisions_path.write_text(header + entry, encoding="utf-8")
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _parse_items(raw: str) -> List[Constraint]:
|
|
291
|
+
"""Parse structured constraint blocks from constraints.md text."""
|
|
292
|
+
items = []
|
|
293
|
+
for m in _BLOCK_RE.finditer(raw):
|
|
294
|
+
cid, confirmed_str, provenance, text = m.groups()
|
|
295
|
+
items.append(Constraint(
|
|
296
|
+
id=cid,
|
|
297
|
+
text=text.strip(),
|
|
298
|
+
provenance=provenance,
|
|
299
|
+
confirmed=(confirmed_str.lower() == "true"),
|
|
300
|
+
))
|
|
301
|
+
return items
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Smart context routing — pick which MD files / sections to include per prompt.
|
|
2
|
+
|
|
3
|
+
The full context is expensive: constraints + architecture + project + session digest
|
|
4
|
+
+ top functions adds 1k–3k tokens to every prompt. Most prompts don't need all of it.
|
|
5
|
+
|
|
6
|
+
This module classifies the prompt by keyword family and returns ONLY the sections
|
|
7
|
+
that actually help for that kind of task.
|
|
8
|
+
|
|
9
|
+
Always-on:
|
|
10
|
+
• constraints (compact) — Zone 1, never skip
|
|
11
|
+
• session digest (5 turns) — short, useful for continuity
|
|
12
|
+
|
|
13
|
+
Conditional:
|
|
14
|
+
• architecture.md — for design/refactor/migrate/explain queries
|
|
15
|
+
• project.md — for "what is this codebase" queries
|
|
16
|
+
• decisions.md — for "why did we..." queries
|
|
17
|
+
|
|
18
|
+
This is heuristic — no LLM call — so it's cheap to run on every hook invocation.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import re
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Dict, List, Tuple
|
|
26
|
+
|
|
27
|
+
# Keyword families. Order matters — first match wins for the primary mode.
|
|
28
|
+
_KEYWORD_FAMILIES: List[Tuple[str, List[str]]] = [
|
|
29
|
+
(
|
|
30
|
+
"architecture",
|
|
31
|
+
[
|
|
32
|
+
"architecture", "architect", "design", "structure", "organize",
|
|
33
|
+
"refactor", "restructure", "migrate", "module", "layer", "pattern",
|
|
34
|
+
"dependency", "coupling", "abstraction",
|
|
35
|
+
],
|
|
36
|
+
),
|
|
37
|
+
(
|
|
38
|
+
"explain",
|
|
39
|
+
[
|
|
40
|
+
"what is", "what does", "explain", "describe", "overview",
|
|
41
|
+
"how does this", "walk me through", "tell me about",
|
|
42
|
+
],
|
|
43
|
+
),
|
|
44
|
+
(
|
|
45
|
+
"decision",
|
|
46
|
+
[
|
|
47
|
+
"why did", "why do we", "why are we", "rationale", "history of",
|
|
48
|
+
"decision", "tradeoff", "trade-off",
|
|
49
|
+
],
|
|
50
|
+
),
|
|
51
|
+
(
|
|
52
|
+
"debug",
|
|
53
|
+
[
|
|
54
|
+
"fix", "bug", "broken", "error", "fail", "crash", "exception",
|
|
55
|
+
"traceback", "regression", "wrong", "not working",
|
|
56
|
+
],
|
|
57
|
+
),
|
|
58
|
+
(
|
|
59
|
+
"test",
|
|
60
|
+
[
|
|
61
|
+
"test", "coverage", "spec", "pytest", "unittest", "mock", "fixture",
|
|
62
|
+
],
|
|
63
|
+
),
|
|
64
|
+
(
|
|
65
|
+
"review",
|
|
66
|
+
[
|
|
67
|
+
"review", "audit", "check", "validate", "inspect", "lint",
|
|
68
|
+
"security", "vulnerability",
|
|
69
|
+
],
|
|
70
|
+
),
|
|
71
|
+
]
|
|
72
|
+
|
|
73
|
+
# Section name → (filename, max_chars)
|
|
74
|
+
_SECTION_FILES: Dict[str, Tuple[str, int]] = {
|
|
75
|
+
"architecture": ("architecture.md", 3200), # ~800 tokens
|
|
76
|
+
"project": ("project.md", 1600), # ~400 tokens
|
|
77
|
+
"decisions": ("decisions.md", 2400), # ~600 tokens
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def classify_prompt(prompt: str) -> str:
|
|
82
|
+
"""Return the primary keyword family for a prompt.
|
|
83
|
+
|
|
84
|
+
Returns 'general' if no family matches.
|
|
85
|
+
"""
|
|
86
|
+
p = (prompt or "").lower()
|
|
87
|
+
if not p.strip():
|
|
88
|
+
return "general"
|
|
89
|
+
for family, keywords in _KEYWORD_FAMILIES:
|
|
90
|
+
for kw in keywords:
|
|
91
|
+
# Word-boundary for single words; phrase match for multi-word
|
|
92
|
+
if " " in kw:
|
|
93
|
+
if kw in p:
|
|
94
|
+
return family
|
|
95
|
+
else:
|
|
96
|
+
if re.search(rf"\b{re.escape(kw)}\b", p):
|
|
97
|
+
return family
|
|
98
|
+
return "general"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def route_context_sections(
|
|
102
|
+
prompt: str,
|
|
103
|
+
sg_dir: Path,
|
|
104
|
+
) -> Dict[str, str]:
|
|
105
|
+
"""Pick which optional MD sections to include, based on prompt classification.
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
Dict mapping section_name → trimmed content. May be empty for general queries.
|
|
109
|
+
Caller is responsible for actually injecting these sections into the
|
|
110
|
+
prompt/overview output.
|
|
111
|
+
|
|
112
|
+
Always-on sections (constraints, session digest, top functions) are NOT
|
|
113
|
+
handled here — they're always included by the hook/tool caller.
|
|
114
|
+
"""
|
|
115
|
+
sections: Dict[str, str] = {}
|
|
116
|
+
family = classify_prompt(prompt)
|
|
117
|
+
|
|
118
|
+
if family == "architecture":
|
|
119
|
+
_add_section(sections, "architecture", sg_dir)
|
|
120
|
+
|
|
121
|
+
elif family == "explain":
|
|
122
|
+
# "what is this codebase" — include project + architecture (short)
|
|
123
|
+
_add_section(sections, "project", sg_dir)
|
|
124
|
+
_add_section(sections, "architecture", sg_dir, char_override=1600)
|
|
125
|
+
|
|
126
|
+
elif family == "decision":
|
|
127
|
+
_add_section(sections, "decisions", sg_dir)
|
|
128
|
+
|
|
129
|
+
# debug / test / review / general: no extra MD — constraints + session +
|
|
130
|
+
# top functions are usually enough. The caller adds those unconditionally.
|
|
131
|
+
|
|
132
|
+
return sections
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _add_section(
|
|
136
|
+
sections: Dict[str, str],
|
|
137
|
+
key: str,
|
|
138
|
+
sg_dir: Path,
|
|
139
|
+
char_override: int = 0,
|
|
140
|
+
) -> None:
|
|
141
|
+
"""Read the MD file for `key`, trim, store under sections[key]."""
|
|
142
|
+
if key not in _SECTION_FILES:
|
|
143
|
+
return
|
|
144
|
+
filename, default_chars = _SECTION_FILES[key]
|
|
145
|
+
cap = char_override or default_chars
|
|
146
|
+
path = sg_dir / filename
|
|
147
|
+
if not path.exists():
|
|
148
|
+
return
|
|
149
|
+
try:
|
|
150
|
+
text = path.read_text(encoding="utf-8", errors="replace").strip()
|
|
151
|
+
if not text:
|
|
152
|
+
return
|
|
153
|
+
if len(text) > cap:
|
|
154
|
+
text = text[:cap].rstrip() + "\n... (truncated)"
|
|
155
|
+
sections[key] = text
|
|
156
|
+
except Exception:
|
|
157
|
+
pass
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def format_routed_sections(sections: Dict[str, str]) -> str:
|
|
161
|
+
"""Render routed sections as markdown blocks. Returns '' if empty."""
|
|
162
|
+
if not sections:
|
|
163
|
+
return ""
|
|
164
|
+
parts = []
|
|
165
|
+
titles = {
|
|
166
|
+
"architecture": "## Architecture",
|
|
167
|
+
"project": "## Project",
|
|
168
|
+
"decisions": "## Decisions",
|
|
169
|
+
}
|
|
170
|
+
for key, text in sections.items():
|
|
171
|
+
title = titles.get(key, f"## {key.title()}")
|
|
172
|
+
parts.append(f"{title}\n{text}")
|
|
173
|
+
return "\n\n".join(parts)
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mode modifiers: reasoning instructions injected into assembled context.
|
|
3
|
+
|
|
4
|
+
7 modifiers total:
|
|
5
|
+
- 6 instruction-level (text appended near top of context)
|
|
6
|
+
- 1 API-level (EXTENDED_THINKING — flag only, not text)
|
|
7
|
+
|
|
8
|
+
Modifiers shape HOW the LLM reasons, not WHAT it sees.
|
|
9
|
+
Selected automatically by classifier.py. Max 2 instruction-level per query.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Dict, List
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# ── Modifier Templates ──────────────────────────────────────────────────
|
|
18
|
+
# Each is ~60-100 tokens. Placed at position 3 in assembly
|
|
19
|
+
# (after task + constraints, before session/architecture/code).
|
|
20
|
+
|
|
21
|
+
MODIFIER_TEMPLATES: Dict[str, str] = {
|
|
22
|
+
"BRAINSTORM": (
|
|
23
|
+
"## Reasoning Mode: Brainstorm\n"
|
|
24
|
+
"Before recommending an approach:\n"
|
|
25
|
+
"1. Generate at least 3 distinct approaches, including at least one you'd normally dismiss\n"
|
|
26
|
+
"2. For each: state the core tradeoff, not just pros/cons\n"
|
|
27
|
+
"3. Flag which approach fits best given the constraints above and WHY\n"
|
|
28
|
+
"4. Only then give a recommendation\n"
|
|
29
|
+
"\n"
|
|
30
|
+
"Do not recommend the first approach you think of. The right answer for this\n"
|
|
31
|
+
"project may not be the obvious one."
|
|
32
|
+
),
|
|
33
|
+
|
|
34
|
+
"BLAST_FIRST": (
|
|
35
|
+
"## Reasoning Mode: Blast-First\n"
|
|
36
|
+
"Before making any changes:\n"
|
|
37
|
+
"1. List every caller, consumer, and dependent of the code being refactored\n"
|
|
38
|
+
"2. For each: state whether the change will break it, may break it, or is safe\n"
|
|
39
|
+
"3. Identify the highest-risk change (most callers, least test coverage)\n"
|
|
40
|
+
"4. Start with the highest-risk change and confirm approach before proceeding\n"
|
|
41
|
+
"\n"
|
|
42
|
+
"Do not write code until the blast radius analysis is complete."
|
|
43
|
+
),
|
|
44
|
+
|
|
45
|
+
"VERIFY_ASSUMPTIONS": (
|
|
46
|
+
"## Reasoning Mode: Verify Assumptions\n"
|
|
47
|
+
"Before diagnosing:\n"
|
|
48
|
+
"1. List at least 3 possible causes, ordered by likelihood given recent changes\n"
|
|
49
|
+
"2. For each cause: what evidence would confirm or rule it out?\n"
|
|
50
|
+
"3. Check the evidence available in the context above\n"
|
|
51
|
+
"4. Only after ruling out alternatives: state your diagnosis\n"
|
|
52
|
+
"\n"
|
|
53
|
+
"The most obvious cause is often not the actual cause in debugging.\n"
|
|
54
|
+
"Favor hypotheses that explain the symptoms given RECENT CHANGES (see session memory)."
|
|
55
|
+
),
|
|
56
|
+
|
|
57
|
+
"STEP_COMMIT": (
|
|
58
|
+
"## Reasoning Mode: Step-Commit\n"
|
|
59
|
+
"This is a multi-step implementation. Use this process:\n"
|
|
60
|
+
"1. State the implementation plan (files to create/modify, order of changes)\n"
|
|
61
|
+
"2. Wait for confirmation before writing code (just state the plan first)\n"
|
|
62
|
+
"3. Implement one logical unit at a time\n"
|
|
63
|
+
"4. After each unit: state what's complete, what's next, what could break\n"
|
|
64
|
+
"\n"
|
|
65
|
+
"Do not attempt to implement everything in one response.\n"
|
|
66
|
+
"A correct partial implementation is better than a broken complete one."
|
|
67
|
+
),
|
|
68
|
+
|
|
69
|
+
"MINIMAL": (
|
|
70
|
+
"## Reasoning Mode: Minimal Change\n"
|
|
71
|
+
"Make the smallest correct change that achieves the goal.\n"
|
|
72
|
+
"Do not refactor surrounding code unless it's required for correctness.\n"
|
|
73
|
+
"Do not improve unrelated things you notice.\n"
|
|
74
|
+
"If you see something that should be fixed but is out of scope, note it briefly — don't fix it."
|
|
75
|
+
),
|
|
76
|
+
|
|
77
|
+
"THINK_ALOUD": (
|
|
78
|
+
"## Reasoning Mode: Think Aloud\n"
|
|
79
|
+
"For this task, show your reasoning before your answer:\n"
|
|
80
|
+
"- What are you uncertain about?\n"
|
|
81
|
+
"- What assumptions are you making?\n"
|
|
82
|
+
"- What would change your answer if it turned out to be wrong?\n"
|
|
83
|
+
"\n"
|
|
84
|
+
"Keep reasoning concise. The goal is to surface hidden assumptions,\n"
|
|
85
|
+
"not to write an essay. Then give your answer."
|
|
86
|
+
),
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
# EXTENDED_THINKING is NOT a template — it's an API-level flag.
|
|
90
|
+
# The assembler sets metadata.extended_thinking = True.
|
|
91
|
+
# The caller (MCP server, hook, CLI) uses the flag if the model supports it.
|
|
92
|
+
# Trigger: PLANNING or DEBUG_INVESTIGATE + cross_file > 3 or dep_depth > 3.
|
|
93
|
+
# Only works for: Claude Code API-direct, sg-agent, sg prompt → Claude.ai.
|
|
94
|
+
# Does NOT work for: Cursor, Copilot, Antigravity (they control the API call).
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def render_modifiers(modifier_names: List[str]) -> str:
|
|
98
|
+
"""Render selected modifiers into a single text block for context injection.
|
|
99
|
+
|
|
100
|
+
Args:
|
|
101
|
+
modifier_names: List of modifier names (e.g., ["BRAINSTORM", "THINK_ALOUD"])
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
Combined modifier text, or empty string if no modifiers.
|
|
105
|
+
EXTENDED_THINKING is silently skipped (it's API-level, not text).
|
|
106
|
+
"""
|
|
107
|
+
parts = []
|
|
108
|
+
for name in modifier_names:
|
|
109
|
+
if name == "EXTENDED_THINKING":
|
|
110
|
+
continue # API-level, not rendered as text
|
|
111
|
+
template = MODIFIER_TEMPLATES.get(name)
|
|
112
|
+
if template:
|
|
113
|
+
parts.append(template)
|
|
114
|
+
|
|
115
|
+
if not parts:
|
|
116
|
+
return ""
|
|
117
|
+
|
|
118
|
+
return "\n\n".join(parts)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def estimate_modifier_tokens(modifier_names: List[str]) -> int:
|
|
122
|
+
"""Estimate token count for selected modifiers.
|
|
123
|
+
|
|
124
|
+
Rough estimate: ~4 chars per token for English instruction text.
|
|
125
|
+
"""
|
|
126
|
+
text = render_modifiers(modifier_names)
|
|
127
|
+
if not text:
|
|
128
|
+
return 0
|
|
129
|
+
# ~4 chars per token is a conservative estimate for English instructions
|
|
130
|
+
return len(text) // 4
|