codelith 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. backend/__init__.py +1 -0
  2. backend/agents/__init__.py +6 -0
  3. backend/agents/assessment_agent.py +273 -0
  4. backend/agents/coding_agent.py +779 -0
  5. backend/agents/concept_categories.py +131 -0
  6. backend/agents/concept_detector.py +1217 -0
  7. backend/agents/debug_agent.py +166 -0
  8. backend/agents/teacher_agent.py +179 -0
  9. backend/cli/__init__.py +1 -0
  10. backend/cli/config_cmd.py +135 -0
  11. backend/cli/main.py +606 -0
  12. backend/daemon/__init__.py +1 -0
  13. backend/daemon/launcher.py +243 -0
  14. backend/daemon/server.py +453 -0
  15. backend/daemon/state.py +110 -0
  16. backend/daemon/static/assets/Gambarino-Regular-BjbcsURA.otf +0 -0
  17. backend/daemon/static/assets/abnfDiagram-VCTEODGH-CCJBE2aE.js +1 -0
  18. backend/daemon/static/assets/arc-BEvzHx4o.js +1 -0
  19. backend/daemon/static/assets/architecture-7GRP2DOG-DaWrPggL.js +1 -0
  20. backend/daemon/static/assets/architectureDiagram-5GKGNRK7-pR-klcZv.js +36 -0
  21. backend/daemon/static/assets/array-BifhSqXX.js +1 -0
  22. backend/daemon/static/assets/blockDiagram-I7D4REHJ-C504Gj6_.js +129 -0
  23. backend/daemon/static/assets/c4Diagram-7LVT6UL2-BjM04Mni.js +38 -0
  24. backend/daemon/static/assets/channel-DzSauwD3.js +1 -0
  25. backend/daemon/static/assets/chunk-2Q5K7J3B-C1jixKkw.js +1 -0
  26. backend/daemon/static/assets/chunk-4HAMMTFA-EgoP78tp.js +62 -0
  27. backend/daemon/static/assets/chunk-5VM5RSS4-ZNzvKenW.js +15 -0
  28. backend/daemon/static/assets/chunk-75Z2AOVW-EXNbuzun.js +2 -0
  29. backend/daemon/static/assets/chunk-DU6HZSFF-CF3OK3MZ.js +127 -0
  30. backend/daemon/static/assets/chunk-F27PBJKO-G71ylWJa.js +1 -0
  31. backend/daemon/static/assets/chunk-FOHPRMQF-DHwB1DNv.js +161 -0
  32. backend/daemon/static/assets/chunk-GMAD6QVW-2yfGg28o.js +72 -0
  33. backend/daemon/static/assets/chunk-GVQU2GXP-C_VeaX4U.js +1 -0
  34. backend/daemon/static/assets/chunk-IMKFNOWR-CNexRjjn.js +231 -0
  35. backend/daemon/static/assets/chunk-JWPE2WC7-DVXcaiue.js +1 -0
  36. backend/daemon/static/assets/chunk-P2QGCYS3-E4AByfsD.js +1 -0
  37. backend/daemon/static/assets/chunk-POPQ4Y6H-Bisbc2-3.js +1 -0
  38. backend/daemon/static/assets/chunk-PWAF6VOD-DaoPxZAa.js +1 -0
  39. backend/daemon/static/assets/chunk-SHT3W25Y-DarPToto.js +168 -0
  40. backend/daemon/static/assets/chunk-SVP7TREG-DvMOAiwI.js +88 -0
  41. backend/daemon/static/assets/chunk-TICWLB2K-DheuvyGM.js +206 -0
  42. backend/daemon/static/assets/chunk-XXDRQBXY-DFBUG-OT.js +1 -0
  43. backend/daemon/static/assets/chunk-Y2CYZVJY-DsF7k-Jl.js +1 -0
  44. backend/daemon/static/assets/classDiagram-ZZMXUADV-Ys5zkCXW.js +1 -0
  45. backend/daemon/static/assets/classDiagram-v2-VYDZK3BY-Ys5zkCXW.js +1 -0
  46. backend/daemon/static/assets/cose-bilkent-JH36ORCC-DLPLnxrP.js +1 -0
  47. backend/daemon/static/assets/cynefin-OW5HDTMX-Dv1OY_0y.js +1 -0
  48. backend/daemon/static/assets/cynefinDiagram-5FMLGOSQ-Ur7MTCmF.js +62 -0
  49. backend/daemon/static/assets/cytoscape.esm-CECbKnxF.js +321 -0
  50. backend/daemon/static/assets/dagre-CJLTJMFW.js +1 -0
  51. backend/daemon/static/assets/dagre-GXQ25YYZ-R3BwTvng.js +4 -0
  52. backend/daemon/static/assets/defaultLocale-BFoDCU3G.js +1 -0
  53. backend/daemon/static/assets/diagram-S7CK7UJ4-BxIoEKb4.js +30 -0
  54. backend/daemon/static/assets/diagram-UQ7AKVKN-DO4cuWN-.js +41 -0
  55. backend/daemon/static/assets/diagram-VSXAHHWV-DW5imp5t.js +3 -0
  56. backend/daemon/static/assets/diagram-VX7I27RA-CdZ3k7wQ.js +24 -0
  57. backend/daemon/static/assets/diagram-Z3DM3KII-DPyjbneL.js +24 -0
  58. backend/daemon/static/assets/dist-DTg6UBE_.js +1 -0
  59. backend/daemon/static/assets/ebnfDiagram-PWID7BFC-BO7VQsye.js +1 -0
  60. backend/daemon/static/assets/erDiagram-RLTQ6QDP-CevvjECq.js +99 -0
  61. backend/daemon/static/assets/eventmodeling-NTZA5JFV-yNfKR6-v.js +1 -0
  62. backend/daemon/static/assets/flowDiagram-HODETNUW-B4GT41mU.js +1 -0
  63. backend/daemon/static/assets/ganttDiagram-EL5Y4UJY-DNW5fWw1.js +292 -0
  64. backend/daemon/static/assets/gitGraph-4MIJSDKK-DKgVkWaZ.js +1 -0
  65. backend/daemon/static/assets/gitGraphDiagram-WWUBYQGX-0S7OF9Aj.js +106 -0
  66. backend/daemon/static/assets/index-D3vj8REa.js +63 -0
  67. backend/daemon/static/assets/index-D4lMFaiv.css +1 -0
  68. backend/daemon/static/assets/info-A6RAGUB7-Bxy-SzRN.js +1 -0
  69. backend/daemon/static/assets/infoDiagram-27XIBGKW-ClzQji6X.js +2 -0
  70. backend/daemon/static/assets/init-C-OQMol4.js +1 -0
  71. backend/daemon/static/assets/ishikawaDiagram-5VMMS53U-B3Lo-sS3.js +70 -0
  72. backend/daemon/static/assets/journeyDiagram-3NMN7TZE-0KL6R2Rz.js +139 -0
  73. backend/daemon/static/assets/kanban-definition-UXKFOSKX-zt5NbEep.js +89 -0
  74. backend/daemon/static/assets/katex-CXMH3UgJ.js +257 -0
  75. backend/daemon/static/assets/line-CiAFRJVJ.js +1 -0
  76. backend/daemon/static/assets/linear-BI6yqEPV.js +1 -0
  77. backend/daemon/static/assets/logo_darkmode-BPDdj6GZ.png +0 -0
  78. backend/daemon/static/assets/logo_lightmode-C3ZWMgAH.png +0 -0
  79. backend/daemon/static/assets/mermaid-parser.core-DEadI1Ja.js +7 -0
  80. backend/daemon/static/assets/mindmap-definition-YA3MSWOX-TGKGYg5n.js +96 -0
  81. backend/daemon/static/assets/ordinal-BDEzSJ7C.js +1 -0
  82. backend/daemon/static/assets/packet-AYTQ26CC-CZTSuh5x.js +1 -0
  83. backend/daemon/static/assets/path-fybaL0A-.js +1 -0
  84. backend/daemon/static/assets/pegDiagram-XKGWAZYB-DGd8LACA.js +1 -0
  85. backend/daemon/static/assets/pie-WAS4IAKB-B59sPr3Z.js +1 -0
  86. backend/daemon/static/assets/pieDiagram-E7YTZNPT-CpwxCR3L.js +39 -0
  87. backend/daemon/static/assets/quadrantDiagram-AXDQQJYC-BwSeF_E_.js +7 -0
  88. backend/daemon/static/assets/radar-RG4KPBEZ-DAa4JvTb.js +1 -0
  89. backend/daemon/static/assets/railroad-74A4TZTK-BitdNgDt.js +1 -0
  90. backend/daemon/static/assets/railroad-abnf-HS5TGJTU-DCrNKqAH.js +1 -0
  91. backend/daemon/static/assets/railroad-ebnf-LZEXJU2U-DmEwx8OK.js +1 -0
  92. backend/daemon/static/assets/railroad-peg-WCYAUIDC-CPc8dTCP.js +1 -0
  93. backend/daemon/static/assets/railroadDiagram-O6MQD6OU-DuizuzwD.js +1 -0
  94. backend/daemon/static/assets/requirementDiagram-BXWQKSXE-BjMk0yS8.js +84 -0
  95. backend/daemon/static/assets/rough.esm-Dy-Kn_BL.js +1 -0
  96. backend/daemon/static/assets/sankeyDiagram-P5KCCOFB-0T_bhkmz.js +40 -0
  97. backend/daemon/static/assets/sequenceDiagram-WJ2MYXX4-Cwa-1Stp.js +162 -0
  98. backend/daemon/static/assets/sizeCapture-INFHLROL-B0uUizjq.js +1 -0
  99. backend/daemon/static/assets/src-BH-TyZbA.js +1 -0
  100. backend/daemon/static/assets/stateDiagram-D77RDMKH-BpQSg_QL.js +1 -0
  101. backend/daemon/static/assets/stateDiagram-v2-MP3YSRHH-BItVXKof.js +1 -0
  102. backend/daemon/static/assets/swimlanes-42K2YHIH-h_ED18Vy.js +1 -0
  103. backend/daemon/static/assets/swimlanesDiagram-VR7AAH4N-D0fo0LN-.js +8 -0
  104. backend/daemon/static/assets/timeline-definition-24CTP7MA-DKfSO33a.js +120 -0
  105. backend/daemon/static/assets/treeView-Q6P3EWNA-DAj9fxfC.js +1 -0
  106. backend/daemon/static/assets/treemap-WGGIJYW6-5IIXD9Zu.js +1 -0
  107. backend/daemon/static/assets/vennDiagram-4TSXK5OY-BoBvVEci.js +34 -0
  108. backend/daemon/static/assets/wardley-WFR3VGLG-CGsd7s_-.js +1 -0
  109. backend/daemon/static/assets/wardleyDiagram-VM6X3IG4-QHdK5NsY.js +78 -0
  110. backend/daemon/static/assets/xychartDiagram-S5SC5T6Z-MN_fdKCJ.js +7 -0
  111. backend/daemon/static/index.html +49 -0
  112. backend/database/__init__.py +1 -0
  113. backend/database/concept_slug.py +39 -0
  114. backend/database/concepts.py +804 -0
  115. backend/llm/__init__.py +5 -0
  116. backend/llm/client.py +333 -0
  117. backend/llm/config.py +254 -0
  118. backend/llm/key_setup.py +237 -0
  119. backend/main.py +13 -0
  120. backend/orchestrator/__init__.py +1 -0
  121. backend/orchestrator/events.py +52 -0
  122. backend/orchestrator/graph.py +316 -0
  123. backend/orchestrator/modes.py +125 -0
  124. codelith-0.1.0.dist-info/METADATA +301 -0
  125. codelith-0.1.0.dist-info/RECORD +129 -0
  126. codelith-0.1.0.dist-info/WHEEL +5 -0
  127. codelith-0.1.0.dist-info/entry_points.txt +2 -0
  128. codelith-0.1.0.dist-info/licenses/LICENSE +21 -0
  129. codelith-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,237 @@
1
+ """First-run API key setup via the OS credential store.
2
+
3
+ Reuses the storage layer in :mod:`backend.llm.client` (service name
4
+ ``codelith``, one keyring account per provider) and adds what the CLI
5
+ needs:
6
+
7
+ - **Validation** — a minimal authenticated request to the provider is
8
+ made BEFORE the key is stored; an invalid key is never saved.
9
+ - **Interactive setup** — terminal prompt naming the provider, with the
10
+ provider's official key page, via :func:`collect_and_store_key`.
11
+ - **Startup check** — :func:`ensure_keys_at_startup` runs at the start
12
+ of ``codelith`` (after subcommand dispatch, before the daemon) and
13
+ prompts only for providers whose key cannot be resolved from any
14
+ layer (env var → keyring → .env). Non-interactive sessions (scripts,
15
+ CI) are never blocked: they get a hint instead of a prompt.
16
+
17
+ No key is ever hardcoded, written to a file, or logged. If the OS
18
+ credential store is unavailable, setup fails loudly rather than
19
+ leaking the key into plaintext.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import sys
25
+ from dataclasses import dataclass
26
+ from urllib.error import HTTPError, URLError
27
+ from urllib.request import Request, urlopen
28
+
29
+ from backend.llm.client import (
30
+ KEYRING_ACCOUNTS,
31
+ OPENROUTER_API_KEY_ENV,
32
+ GROQ_API_KEY_ENV,
33
+ resolve_agent_api_key,
34
+ resolve_api_key,
35
+ store_api_key,
36
+ )
37
+
38
+ VALIDATION_TIMEOUT = 15 # seconds
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class Provider:
43
+ """One API provider's setup metadata and key-page pointer."""
44
+
45
+ name: str # keyring account name, matches KEYRING_ACCOUNTS values
46
+ display: str
47
+ key_page: str
48
+ env_var: str
49
+ validation_url: str
50
+ resolver: object # () -> Optional[str]
51
+
52
+
53
+ PROVIDERS: dict[str, Provider] = {
54
+ "groq": Provider(
55
+ name="groq",
56
+ display="Groq",
57
+ key_page="https://console.groq.com/keys",
58
+ env_var=GROQ_API_KEY_ENV,
59
+ validation_url="https://api.groq.com/openai/v1/models",
60
+ resolver=resolve_api_key,
61
+ ),
62
+ "openrouter": Provider(
63
+ name="openrouter",
64
+ display="OpenRouter",
65
+ key_page="https://openrouter.ai/keys",
66
+ env_var=OPENROUTER_API_KEY_ENV,
67
+ validation_url="https://openrouter.ai/api/v1/key",
68
+ resolver=resolve_agent_api_key,
69
+ ),
70
+ }
71
+
72
+ SETUP_ORDER = ("groq", "openrouter")
73
+
74
+
75
+ def validate_api_key(provider: Provider, api_key: str) -> tuple[bool, str]:
76
+ """Make one minimal authenticated request to *provider*.
77
+
78
+ Returns ``(ok, message)``. Fail-closed: any ambiguity (network
79
+ down, unknown status) counts as invalid so an unverifiable key is
80
+ never stored.
81
+ """
82
+ request = Request(
83
+ provider.validation_url,
84
+ headers={
85
+ "Authorization": f"Bearer {api_key}",
86
+ # Cloudflare (Groq's edge) blocks Python-urllib's default
87
+ # signature with "error code: 1010" before the key is ever
88
+ # checked — a 403 that looks exactly like a bad key. A
89
+ # proper User-Agent gets through; the OpenAI SDK does the
90
+ # same, which is why the agents never saw this.
91
+ "User-Agent": "codelith/0.1 (api-key validation)",
92
+ # OpenRouter prefers identifying headers; harmless for Groq.
93
+ "HTTP-Referer": "https://github.com/codelith",
94
+ "X-Title": "CodeLith",
95
+ },
96
+ )
97
+ try:
98
+ with urlopen(request, timeout=VALIDATION_TIMEOUT) as response:
99
+ response.read(64)
100
+ if 200 <= response.status < 300:
101
+ return True, f"{provider.display} key validated."
102
+ return False, f"{provider.display} returned HTTP {response.status}."
103
+ except HTTPError as exc:
104
+ try:
105
+ body = exc.read(200).decode("utf-8", "replace")
106
+ except Exception: # noqa: BLE001 - body is best-effort context
107
+ body = ""
108
+ if exc.code == 401:
109
+ return False, f"{provider.display} rejected the key (invalid or revoked)."
110
+ if exc.code == 403 and "error code:" in body:
111
+ return False, (
112
+ f"{provider.display}'s firewall blocked the check before the "
113
+ "key could be verified — try again in a moment."
114
+ )
115
+ if exc.code == 403:
116
+ return False, f"{provider.display} rejected the key (forbidden)."
117
+ return False, f"{provider.display} returned HTTP {exc.code} — try again later."
118
+ except (URLError, TimeoutError, OSError) as exc:
119
+ reason = getattr(exc, "reason", exc)
120
+ return False, f"Could not reach {provider.display} to validate the key ({reason})."
121
+
122
+
123
+ def collect_and_store_key(provider: Provider, prompt_func=None) -> bool:
124
+ """Interactively obtain, validate, and store one provider's key.
125
+
126
+ Prompts name the provider and point at its official key page. The
127
+ key is stored ONLY after successful validation; validation failure
128
+ re-prompts (blank input cancels). Returns True when a valid key is
129
+ in the OS credential store.
130
+
131
+ ``prompt_func`` is injectable for tests; it receives the prompt
132
+ text and returns the entered key.
133
+ """
134
+ if prompt_func is None:
135
+ from getpass import getpass
136
+
137
+ prompt_func = getpass
138
+
139
+ print(f"\nNo {provider.display} API key found.")
140
+ print(f"{provider.display} powers "
141
+ + ("teaching, explanations, and grading."
142
+ if provider.name == "groq"
143
+ else "the coding and debugging agents.")
144
+ + f" Create a key at {provider.key_page}")
145
+
146
+ while True:
147
+ try:
148
+ key = prompt_func(f"Paste your {provider.display} API key (Enter to cancel): ")
149
+ except (EOFError, KeyboardInterrupt):
150
+ print()
151
+ return False
152
+ key = (key or "").strip()
153
+ if not key:
154
+ print(f"Setup skipped — {provider.display} features will be unavailable.")
155
+ return False
156
+
157
+ ok, message = validate_api_key(provider, key)
158
+ if not ok:
159
+ print(f" ✗ {message} The key was NOT saved — check it and try again.")
160
+ continue
161
+
162
+ if store_api_key(provider.name, key):
163
+ print(f" ✓ {message} Saved to your OS credential store.")
164
+ return True
165
+ print(
166
+ " ✗ Key validated, but the OS credential store is unavailable, "
167
+ "so it was NOT saved. Install/unlock your system's keyring backend "
168
+ "(on Linux: gnome-keyring or kwallet) and run `codelith setup`."
169
+ )
170
+ return False
171
+
172
+
173
+ def _missing_providers() -> list[Provider]:
174
+ """Providers whose key cannot be resolved from any existing layer."""
175
+ missing = []
176
+ for name in SETUP_ORDER:
177
+ provider = PROVIDERS[name]
178
+ try:
179
+ if not provider.resolver():
180
+ missing.append(provider)
181
+ except Exception: # noqa: BLE001 - resolution must never crash startup
182
+ missing.append(provider)
183
+ return missing
184
+
185
+
186
+ def ensure_keys_at_startup(stream=None, prompt_func=None, interactive: bool | None = None) -> None:
187
+ """Startup gate: prompt only for providers with no resolvable key.
188
+
189
+ Never raises and never blocks non-interactive sessions — scripts and
190
+ CI get a one-line hint pointing at ``codelith setup`` instead.
191
+ """
192
+ if stream is None:
193
+ stream = sys.stdout
194
+ if interactive is None:
195
+ interactive = bool(sys.stdin and sys.stdin.isatty())
196
+
197
+ missing = _missing_providers()
198
+ if not missing:
199
+ return
200
+
201
+ names = " and ".join(p.display for p in missing)
202
+ if not interactive:
203
+ print(
204
+ f"[codelith] {names} API key(s) not configured — "
205
+ "run `codelith setup` to add them interactively.",
206
+ file=stream,
207
+ )
208
+ return
209
+
210
+ print(f"[codelith] First-run setup: a {names} API key is needed.")
211
+ for provider in missing:
212
+ collect_and_store_key(provider, prompt_func=prompt_func)
213
+
214
+
215
+ def run_setup(provider_name: str | None = None, prompt_func=None) -> int:
216
+ """Explicit ``codelith setup [provider]`` entry point.
217
+
218
+ Validates and (re)stores keys even when one already exists — the
219
+ natural fix for a revoked or mistyped key. Returns a process exit
220
+ code.
221
+ """
222
+ targets = (
223
+ [PROVIDERS[provider_name]]
224
+ if provider_name
225
+ else [PROVIDERS[name] for name in SETUP_ORDER]
226
+ )
227
+ failed = False
228
+ for provider in targets:
229
+ if not collect_and_store_key(provider, prompt_func=prompt_func):
230
+ failed = True
231
+ return 1 if failed else 0
232
+
233
+
234
+ # Sanity: the storage layer's accounts and this registry must not drift.
235
+ assert set(PROVIDERS) == set(KEYRING_ACCOUNTS.values()), (
236
+ "key_setup providers must match backend.llm.client keyring accounts"
237
+ )
backend/main.py ADDED
@@ -0,0 +1,13 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+
5
+
6
+ def main() -> None:
7
+ parser = argparse.ArgumentParser(prog="codelith", description="CodeLith backend")
8
+ parser.parse_args()
9
+ print("CodeLith backend — not implemented yet.")
10
+
11
+
12
+ if __name__ == "__main__":
13
+ main()
@@ -0,0 +1 @@
1
+ """CodeLith orchestrator — LangGraph-based workflow that routes prompts to agents."""
@@ -0,0 +1,52 @@
1
+ """Live event emission for agent activity.
2
+
3
+ Agent nodes call :func:`emit_event` to report what they are doing while
4
+ they are doing it (reading a file, running a command, ...). The
5
+ orchestrator installs a sink via :func:`set_event_sink` before invoking
6
+ the graph; the daemon wires the sink to an SSE queue so clients can
7
+ watch the agent work in real time.
8
+
9
+ A ``ContextVar`` carries the sink so nodes can emit events without
10
+ plumbing a callback through graph state. If no sink is installed the
11
+ events are simply dropped, so emitting is always safe.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import time
17
+ from contextvars import ContextVar, Token
18
+ from typing import Any, Callable
19
+
20
+ EventSink = Callable[[dict[str, Any]], None]
21
+
22
+ _event_sink: ContextVar[EventSink | None] = ContextVar(
23
+ "codelith_event_sink", default=None
24
+ )
25
+
26
+
27
+ def set_event_sink(sink: EventSink) -> Token:
28
+ """Install *sink* for the current execution context.
29
+
30
+ Returns a token to pass to :func:`reset_event_sink` when done.
31
+ """
32
+ return _event_sink.set(sink)
33
+
34
+
35
+ def reset_event_sink(token: Token) -> None:
36
+ """Restore the previous sink for a token from :func:`set_event_sink`."""
37
+ _event_sink.reset(token)
38
+
39
+
40
+ def emit_event(type_: str, **data: Any) -> None:
41
+ """Emit a live activity event to the installed sink, if any.
42
+
43
+ Never raises — a broken sink must never take down the agent.
44
+ """
45
+ sink = _event_sink.get()
46
+ if sink is None:
47
+ return
48
+ event = {"type": type_, "ts": time.time(), **data}
49
+ try:
50
+ sink(event)
51
+ except Exception: # noqa: BLE001 — events are best-effort
52
+ pass
@@ -0,0 +1,316 @@
1
+ """LangGraph orchestrator — wires the user prompt through the agent graph.
2
+
3
+ Flow:
4
+ User Prompt → Coding Agent → (if tests fail) → Debug Agent ┐
5
+ └─ (tests pass) ───────────────────────┤
6
+ ▼
7
+ Detect Concepts
8
+ │
9
+ ┌────────────────┴──────────────┐
10
+ ▼ ▼
11
+ Assessment Agent Teacher Agent → END
12
+ (mode-gated)
13
+
14
+ The coding agent handles file operations and code generation. When one
15
+ of its commands actually fails (non-zero exit code or stderr output,
16
+ read from the structured ``tool_calls_log`` entries), the graph routes
17
+ to the debug agent which diagnoses errors, fixes code, and re-runs
18
+ tests. The shared detect_concepts
19
+ node then runs concept detection ONCE per turn (registry scan + mode-gated
20
+ LLM detection) and writes the result to ``concepts_detected``. The
21
+ assessment agent generates Socratic questions for the dashboard and the
22
+ teacher agent saves teaching content to the dashboard; both read the
23
+ shared detection result instead of re-scanning tool calls.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from typing import Annotated, Any, Callable, TypedDict
29
+
30
+ from langchain_core.messages import AIMessage, BaseMessage, HumanMessage
31
+ from langgraph.graph import END, StateGraph
32
+ from langgraph.graph.message import add_messages
33
+
34
+ from backend.agents.concept_detector import detect_concepts
35
+ from backend.agents.coding_agent import coding_agent_node
36
+
37
+ from backend.agents.debug_agent import debug_agent_node
38
+ from backend.agents.assessment_agent import assessment_agent_node
39
+ from backend.agents.teacher_agent import teacher_agent_node
40
+ from backend.database.concepts import load_concepts, save_concepts_bulk
41
+ from backend.orchestrator.modes import get_mode, DEFAULT_MODE
42
+ from backend.orchestrator.events import emit_event, set_event_sink, reset_event_sink
43
+
44
+
45
+ # ---------------------------------------------------------------------------
46
+ # State
47
+ # ---------------------------------------------------------------------------
48
+
49
+ class AgentState(TypedDict):
50
+ """State passed through the graph."""
51
+
52
+ messages: Annotated[list[BaseMessage], add_messages]
53
+ workspace_root: str
54
+ mode: str
55
+ session: str
56
+ tool_calls_log: list[dict]
57
+ concepts: list[dict]
58
+ # Written by the detect_concepts node each turn: newly detected
59
+ # concepts as dicts with name/category/description/diagram (+
60
+ # source_file/line_range). Read by the assessment and teacher agents.
61
+ concepts_detected: list[dict]
62
+ pending_assessments: list[dict]
63
+ # Set by the coding agent when the turn failed on a provider-side LLM
64
+ # error (not a code problem) so the debug agent can be skipped.
65
+ llm_error: bool
66
+ # Mode-specific state
67
+ current_mode_config: dict | None
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Graph construction
72
+ # ---------------------------------------------------------------------------
73
+
74
+ graph_builder = StateGraph(AgentState)
75
+
76
+ # Nodes — each is wrapped to emit a live "node" event when it starts,
77
+ # so stream consumers can see which agent is running.
78
+
79
+ def _traced(name: str, fn: Callable[[dict], dict]) -> Callable[[dict], dict]:
80
+ """Wrap a graph node so it emits a ``node`` event when it begins."""
81
+
82
+ def wrapped(state: dict) -> dict:
83
+ emit_event("node", node=name)
84
+ return fn(state)
85
+
86
+ return wrapped
87
+
88
+
89
+ graph_builder.add_node("coding_agent", _traced("coding_agent", coding_agent_node))
90
+ graph_builder.add_node("debug_agent", _traced("debug_agent", debug_agent_node))
91
+ graph_builder.add_node(
92
+ "detect_concepts", _traced("detect_concepts", detect_concepts)
93
+ )
94
+ graph_builder.add_node("assessment_agent", _traced("assessment_agent", assessment_agent_node))
95
+ graph_builder.add_node("teacher_agent", _traced("teacher_agent", teacher_agent_node))
96
+
97
+
98
+ # --- Routing logic --------------------------------------------------------
99
+ # After the coding agent runs, inspect the turn's structured tool results.
100
+ # The most recent run_command result decides: non-zero exit code or stderr
101
+ # hands off to the debug agent; either way the next stop is the shared
102
+ # detect_concepts node.
103
+
104
+ def _route_after_coding(state: AgentState) -> str:
105
+ """Route to ``debug_agent`` when the latest command actually failed.
106
+
107
+ Only the most recent ``run_command`` entry in ``tool_calls_log`` is
108
+ consulted: an earlier failing command that the coding agent already
109
+ re-ran cleanly is a self-corrected turn, not a job for the debug
110
+ agent. The latest command failed when it recorded
111
+ ``exit_code != 0`` or ``stderr_present``. The routing decision is
112
+ based purely on these structured execution results — the coding
113
+ agent's reply text is never keyword-scanned, so a clean reply that
114
+ merely mentions a word such as error cannot trigger the debug agent.
115
+ """
116
+ # A provider-side LLM failure is not a code problem — routing to the
117
+ # debug agent would burn another LLM call trying to "fix" a glitch.
118
+ if state.get("llm_error"):
119
+ return "detect_concepts"
120
+ tool_calls_log = state.get("tool_calls_log") or []
121
+ last_command: dict | None = None
122
+ for entry in tool_calls_log:
123
+ if (entry.get("function") or {}).get("name") == "run_command":
124
+ last_command = entry
125
+ if last_command is not None:
126
+ exit_code = last_command.get("exit_code", 0)
127
+ stderr_present = last_command.get("stderr_present", False)
128
+ if exit_code != 0 or stderr_present:
129
+ return "debug_agent"
130
+ return "detect_concepts"
131
+
132
+
133
+ def _route_after_debug(state: AgentState) -> str:
134
+ """After debug agent, go to the shared concept-detection node."""
135
+ return "detect_concepts"
136
+
137
+
138
+ # Entry point → coding_agent
139
+ graph_builder.set_entry_point("coding_agent")
140
+
141
+ # coding_agent → conditional → debug_agent | assessment_agent
142
+ graph_builder.add_conditional_edges(
143
+ "coding_agent",
144
+ _route_after_coding,
145
+ {"debug_agent": "debug_agent", "detect_concepts": "detect_concepts"},
146
+ )
147
+
148
+ # debug_agent → detect_concepts
149
+ graph_builder.add_conditional_edges(
150
+ "debug_agent",
151
+ _route_after_debug,
152
+ {"detect_concepts": "detect_concepts"},
153
+ )
154
+
155
+ def _route_after_detect(state: AgentState) -> str:
156
+ """Skip the assessment agent when the mode disables questions."""
157
+ if (state.get("current_mode_config") or {}).get(
158
+ "assessment_frequency", "high"
159
+ ) == "none":
160
+ return "end"
161
+ return "assessment_agent"
162
+
163
+
164
+ def _route_after_assessment(state: AgentState) -> str:
165
+ """Run the teacher agent in modes where it always runs."""
166
+ if (state.get("current_mode_config") or {}).get("teacher_always_runs", True):
167
+ return "teacher_agent"
168
+ return "end"
169
+
170
+
171
+ # detect_concepts → conditional → assessment_agent | END
172
+ # (assessment agent is skipped when the mode disables questions; the
173
+ # teacher agent always gets the shared concepts_detected result)
174
+ graph_builder.add_conditional_edges(
175
+ "detect_concepts",
176
+ _route_after_detect,
177
+ {"assessment_agent": "assessment_agent", "end": END},
178
+ )
179
+
180
+ # assessment_agent → conditional → teacher_agent | END
181
+ # (teacher agent is skipped in modes where it doesn't always run)
182
+ graph_builder.add_conditional_edges(
183
+ "assessment_agent",
184
+ _route_after_assessment,
185
+ {"teacher_agent": "teacher_agent", "end": END},
186
+ )
187
+
188
+ # teacher_agent → END
189
+ graph_builder.add_edge("teacher_agent", END)
190
+
191
+ # Compile once; reused by the daemon.
192
+ graph = graph_builder.compile()
193
+
194
+
195
+ # ---------------------------------------------------------------------------
196
+ # Convenience runner
197
+ # ---------------------------------------------------------------------------
198
+
199
+ def run_graph(
200
+ user_message: str,
201
+ workspace_root: str | None = None,
202
+ history: list[dict] | None = None,
203
+ mode: str = DEFAULT_MODE,
204
+ session: str = "default",
205
+ event_sink: Callable[[dict], None] | None = None,
206
+ ) -> dict:
207
+ """Run the graph with a user message and return the result dict.
208
+
209
+ Args:
210
+ user_message: The user's prompt.
211
+ workspace_root: Path to the project root for file operations.
212
+ Defaults to the current working directory.
213
+ history: Prior conversation messages as ``[{role, content}, ...]``.
214
+ If provided, they are prepended before the new user message so
215
+ the agent retains context across turns.
216
+ mode: Session mode ("learn", "pair-programming", "autonomous").
217
+ session: Session id for concept storage.
218
+ event_sink: Optional callback invoked with live activity event dicts
219
+ (``{"type": ..., ...}``) while the graph runs — used by the
220
+ daemon to stream progress to the CLI/dashboard. Events are
221
+ emitted as the coding agent reads/writes/executes.
222
+
223
+ Returns:
224
+ A dict with:
225
+ - "reply": the AI's text reply (coding agent output)
226
+ - "concepts": list of newly detected concepts
227
+ - "teaching": the teacher agent's message (if any)
228
+ - "tool_calls_log": tool calls made by the coding agent,
229
+ each ``{"function": {"name", "arguments"}}``
230
+ """
231
+ import os
232
+
233
+ if workspace_root is None:
234
+ workspace_root = os.getcwd()
235
+
236
+ mode_config = get_mode(mode)
237
+
238
+ # Load existing concepts for this session
239
+ existing_concepts = load_concepts(session)
240
+
241
+ # Build the initial message list.
242
+ messages: list[BaseMessage] = []
243
+ if history:
244
+ for entry in history:
245
+ if entry["role"] == "user":
246
+ messages.append(HumanMessage(content=entry["content"]))
247
+ elif entry["role"] == "assistant":
248
+ messages.append(AIMessage(content=entry["content"]))
249
+ messages.append(HumanMessage(content=user_message))
250
+
251
+ initial: AgentState = {
252
+ "messages": messages,
253
+ "workspace_root": workspace_root,
254
+ "mode": mode,
255
+ "session": session,
256
+ "tool_calls_log": [],
257
+ "concepts": existing_concepts,
258
+ "pending_assessments": [],
259
+ "current_mode_config": {
260
+ "name": mode_config.name,
261
+ "teacher_always_runs": mode_config.teacher_always_runs,
262
+ "agent_explains": mode_config.agent_explains,
263
+ "llm_detection": mode_config.llm_detection,
264
+ "surface_concepts": mode_config.surface_concepts,
265
+ "max_tool_rounds": mode_config.max_tool_rounds,
266
+ "prompt_suffix": mode_config.prompt_suffix,
267
+ "assessment_frequency": mode_config.assessment_frequency,
268
+ },
269
+ }
270
+ # Install the event sink for this graph run so nodes can emit live
271
+ # activity events (see backend/orchestrator/events.py).
272
+ if event_sink is None:
273
+ result = graph.invoke(initial)
274
+ else:
275
+ token = set_event_sink(event_sink)
276
+ try:
277
+ result = graph.invoke(initial)
278
+ finally:
279
+ reset_event_sink(token)
280
+
281
+ # Extract results
282
+ all_messages = result.get("messages", [])
283
+ new_concepts = result.get("concepts", [])
284
+ tool_calls_log = result.get("tool_calls_log", [])
285
+
286
+ # Find the coding agent's reply and teaching message
287
+ coding_reply = ""
288
+ teaching_msg = ""
289
+ for msg in all_messages:
290
+ if hasattr(msg, "content"):
291
+ text = msg.content if isinstance(msg.content, str) else str(msg.content)
292
+ # The first AI message after user input is the coding agent's reply
293
+ if isinstance(msg, AIMessage) and not coding_reply:
294
+ coding_reply = text
295
+ elif isinstance(msg, AIMessage) and coding_reply and not teaching_msg:
296
+ teaching_msg = text
297
+ break
298
+
299
+ # Save newly detected concepts (dedup by concept identity slug,
300
+ # falling back to name comparison for legacy entries without one)
301
+ existing_names = {ec["name"] for ec in existing_concepts}
302
+ existing_slugs = {ec.get("slug") for ec in existing_concepts if ec.get("slug")}
303
+ new_only = [
304
+ c for c in new_concepts
305
+ if c["name"] not in existing_names
306
+ and c.get("slug", "") not in existing_slugs
307
+ ]
308
+ if new_only:
309
+ save_concepts_bulk(session, new_only)
310
+
311
+ return {
312
+ "reply": coding_reply,
313
+ "concepts": new_only,
314
+ "teaching": teaching_msg,
315
+ "tool_calls_log": tool_calls_log,
316
+ }