@thanh01.pmt/domain-kit 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +682 -168
  2. package/dist/assembly/index.d.cts +2 -2
  3. package/dist/assembly/index.d.ts +2 -2
  4. package/dist/{chunk-TYXSJUQU.mjs → chunk-2OGNQUXC.mjs} +81 -32
  5. package/dist/chunk-2OGNQUXC.mjs.map +1 -0
  6. package/dist/{chunk-G7HLFMD2.mjs → chunk-5OMSYKQP.mjs} +43 -23
  7. package/dist/chunk-5OMSYKQP.mjs.map +1 -0
  8. package/dist/{chunk-BL5NBKPH.mjs → chunk-67GXK3ZE.mjs} +18 -8
  9. package/dist/chunk-67GXK3ZE.mjs.map +1 -0
  10. package/dist/{chunk-E4GCCWJO.mjs → chunk-CGNXGVHV.mjs} +37 -4
  11. package/dist/chunk-CGNXGVHV.mjs.map +1 -0
  12. package/dist/{chunk-VCPIWJ7R.mjs → chunk-ERDTHXQA.mjs} +21 -3
  13. package/dist/chunk-ERDTHXQA.mjs.map +1 -0
  14. package/dist/chunk-OYSZQVPY.mjs +10 -0
  15. package/dist/chunk-OYSZQVPY.mjs.map +1 -0
  16. package/dist/{chunk-TWKIVUTQ.mjs → chunk-WKTH3JPA.mjs} +197 -36
  17. package/dist/chunk-WKTH3JPA.mjs.map +1 -0
  18. package/dist/{conceptEscalator-YfMKIKCZ.d.ts → conceptEscalator-BCsw0ys_.d.ts} +10 -1
  19. package/dist/{conceptEscalator-ltRLtPhf.d.cts → conceptEscalator-DPrs4PwC.d.cts} +10 -1
  20. package/dist/concepts/index.d.cts +1 -1
  21. package/dist/concepts/index.d.ts +1 -1
  22. package/dist/{curriculumFeedSchema-DUFwFyr2.d.ts → curriculumFeedSchema-DYayelAA.d.cts} +156 -1
  23. package/dist/{curriculumFeedSchema-DUFwFyr2.d.cts → curriculumFeedSchema-DYayelAA.d.ts} +156 -1
  24. package/dist/detector/index.cjs +21 -6
  25. package/dist/detector/index.cjs.map +1 -1
  26. package/dist/detector/index.mjs +2 -1
  27. package/dist/feed/index.cjs +229 -33
  28. package/dist/feed/index.cjs.map +1 -1
  29. package/dist/feed/index.d.cts +32 -2
  30. package/dist/feed/index.d.ts +32 -2
  31. package/dist/feed/index.mjs +2 -2
  32. package/dist/graph/index.cjs +73 -29
  33. package/dist/graph/index.cjs.map +1 -1
  34. package/dist/graph/index.d.cts +14 -5
  35. package/dist/graph/index.d.ts +14 -5
  36. package/dist/graph/index.mjs +2 -1
  37. package/dist/{hybridGraphPipeline-DXQDbBEj.d.ts → hybridGraphPipeline-9dRKJj7r.d.ts} +6 -2
  38. package/dist/{hybridGraphPipeline-dT-bio-3.d.cts → hybridGraphPipeline-CocfYe4d.d.cts} +6 -2
  39. package/dist/{hybridGraphSchema-nrUeSwpU.d.ts → hybridGraphSchema-C37qxdTM.d.cts} +16 -3
  40. package/dist/{hybridGraphSchema-nrUeSwpU.d.cts → hybridGraphSchema-C37qxdTM.d.ts} +16 -3
  41. package/dist/index.cjs +2634 -2359
  42. package/dist/index.cjs.map +1 -1
  43. package/dist/index.d.cts +6 -6
  44. package/dist/index.d.ts +6 -6
  45. package/dist/index.mjs +7 -6
  46. package/dist/llmClient-CX5uUiQ1.d.cts +60 -0
  47. package/dist/llmClient-CX5uUiQ1.d.ts +60 -0
  48. package/dist/pipeline/index.cjs +2207 -1958
  49. package/dist/pipeline/index.cjs.map +1 -1
  50. package/dist/pipeline/index.d.cts +13 -5
  51. package/dist/pipeline/index.d.ts +13 -5
  52. package/dist/pipeline/index.mjs +6 -5
  53. package/dist/schemas/index.cjs +55 -2
  54. package/dist/schemas/index.cjs.map +1 -1
  55. package/dist/schemas/index.d.cts +2 -2
  56. package/dist/schemas/index.d.ts +2 -2
  57. package/dist/schemas/index.mjs +2 -2
  58. package/package.json +10 -9
  59. package/LICENSE +0 -21
  60. package/dist/chunk-BL5NBKPH.mjs.map +0 -1
  61. package/dist/chunk-E4GCCWJO.mjs.map +0 -1
  62. package/dist/chunk-G7HLFMD2.mjs.map +0 -1
  63. package/dist/chunk-TWKIVUTQ.mjs.map +0 -1
  64. package/dist/chunk-TYXSJUQU.mjs.map +0 -1
  65. package/dist/chunk-VCPIWJ7R.mjs.map +0 -1
  66. package/dist/curriculumFeedEmitter-T7NUBLAM.mjs +0 -4
  67. package/dist/curriculumFeedEmitter-T7NUBLAM.mjs.map +0 -1
  68. package/dist/llmClient-ysPhLjcH.d.cts +0 -16
  69. package/dist/llmClient-ysPhLjcH.d.ts +0 -16
package/README.md CHANGED
@@ -11,6 +11,13 @@
11
11
  ## Table of Contents
12
12
 
13
13
  - [Overview](#overview)
14
+ - [End-to-End: From Raw Input to `CurriculumFeed`](#end-to-end-from-raw-input-to-curriculumfeed)
15
+ - [Layer 1 - Detect](#layer-1---detect)
16
+ - [Layer 2 - Generate](#layer-2---generate-one-branch-by-learning-mode)
17
+ - [Layer 3 - Emit](#layer-3---emit-one-shape-for-every-graph)
18
+ - [Layer 4 - Validate](#layer-4---validate-fail-closed-at-the-source)
19
+ - [The three graph types](#the-three-graph-types-at-a-glance)
20
+ - [Depth and minute conventions](#depth-and-minute-conventions)
14
21
  - [Installation](#installation)
15
22
  - [Quick Start](#quick-start)
16
23
  - [5. Emit Curriculum Feed](#5-emit-curriculum-feed-from-any-graph)
@@ -38,13 +45,455 @@
38
45
  - **Auto-detect** input type (repo, syllabus, description) and domain category (software, math, science...) with multi-language support (EN, VI, ZH, JA, KO, FR)
39
46
  - **Context-dependent domain detection** — `.cpp` files default to `software`; `hardware_iot` only when Arduino/ESP32/sensor context is present
40
47
  - **Choose graph type** automatically: `project_graph` (product-driven) or `knowledge_graph` (concept-driven) or `hybrid_graph`
41
- - **AST-level parsing** of Swift, TypeScript, Python, C++ source code
48
+ - **Source scanning** of Swift, TypeScript/JavaScript, Python and C/C++ files (regex-based — the package has no parser dependency)
42
49
  - **LLM-powered decomposition** of projects into features → steps → keywords
43
50
  - **ULO/CIO/SIO depth classification** — each keyword/concept mapped to WHAT (ulo), HOW (cio), or IMPLEMENTATION (sio)
44
51
  - **Concept resolution** with ULO/CIO-aware matching against Master Tree concept codes
45
52
  - **Prerequisite cycle detection** — DFS cycle detection + Kahn's topological sort for knowledge graphs
46
53
  - **Keyword-per-milestone tracking** with new vs prerequisite distinction
47
- - **Time-budget-aware assembly** — configurable session duration, overhead factor, and automatic milestone splitting
54
+ - **Time-budget-aware assembly** — milestones split against a 90-minute session, a 1.15 overhead factor and a 120-minute cap (fixed by the pipeline today; `assembleRoadmap()` takes them as parameters)
55
+
56
+ ---
57
+
58
+ ## End-to-End: From Raw Input to `CurriculumFeed`
59
+
60
+ `domain-kit` never writes a lesson, never schedules a session, never picks an activity.
61
+ Its entire job is to turn *something that already exists* — a repository, a syllabus, a
62
+ short description — into **one JSON object with a fixed shape** (`CurriculumFeed` v2)
63
+ that `curriculum-kit` can schedule without guessing.
64
+
65
+ Read this section as four layers that always run in this order:
66
+
67
+ | Layer | Who | Calls the LLM? | When it fails |
68
+ |---|---|---|---|
69
+ | 1. Detect | `detectDomainProfile()` | no | never throws — reports `confidence` / `ambiguous` |
70
+ | 2. Generate | `runProjectGraphPipeline` / `runKnowledgeGraphPipeline` / `runHybridGraphPipeline` | yes | `throw` (fail-closed) |
71
+ | 3. Emit | `emitCurriculumFeed()` | no | `throw` (fail-closed) |
72
+ | 4. Validate | `validateCurriculumFeed()` | no | `throw` (fail-closed) |
73
+
74
+ The rule that makes the package trustworthy: **layer 2 and 3 never hand back a
75
+ half-built graph.** A missing LLM key, an empty phase plan, a concept with no keywords,
76
+ a backwards prerequisite edge — all of them stop the run at the layer that produced
77
+ them, instead of surfacing as a broken lesson three packages later.
78
+
79
+ > **Shape is fail-closed; content is reported.** Structural violations throw at
80
+ > emission. Things that make a graph *pedagogically* suspect — a feature with no
81
+ > steps, a concept no phase introduces — are **diagnostics** (`source.diagnostics`,
82
+ > each with a `code` and a `severity`) rather than throws, so the host decides whether
83
+ > to block. See [ADR-0002](docs/adr/2026-10-05-ADR-0002-graph-integrity-provenance-and-diagnostics.md).
84
+
85
+ ### The whole picture
86
+
87
+ ```
88
+ +------------------------------------------------------------------------------+
89
+ | INPUT |
90
+ | repoDir / repoUrl - syllabusText / parsedSyllabus - description - goal |
91
+ | techStack - chapters - fileExtensions - gradeBand |
92
+ | forcedLearningMode (router override) - knowledgeTreePath - proposalsPath |
93
+ +------------------------------------+-----------------------------------------+
94
+ v
95
+ +------------------------------------------------------------------------------+
96
+ | LAYER 1 - detectDomainProfile() - pure, deterministic, no LLM |
97
+ | detectInputType() -> detectDomain() -> detectLearningMode() |
98
+ | => DomainProfile { inputType, domainCategory, learningMode, graphType, |
99
+ | confidence, evidence, ambiguous? } |
100
+ +------------------------------------+-----------------------------------------+
101
+ | effectiveMode = forcedLearningMode
102
+ | ?? domainProfile.learningMode
103
+ | (graphType is only its echo)
104
+ +---------------------+---------------------+
105
+ v v v
106
+ 'product' 'concept' 'hybrid'
107
+ | | |
108
+ | | repoDir present? --no--> FALLBACK to the
109
+ | | | concept branch,
110
+ | | | loud if auto (R1)
111
+ v v v
112
+ runProjectGraph runKnowledgeGraph runProjectGraphPipeline + runKnowledgeGraphPipeline
113
+ Pipeline Pipeline (two full graph runs), then:
114
+ STEP_1 scan/ (parseSyllabus generateHybridGraph()
115
+ select/AST first if raw STEP_1 feature-concept links
116
+ STEP_2 keyword syllabusText) STEP_2 decomposePhases
117
+ extract generateKnowledge (auditPhasePlan fail-closed)
118
+ STEP_3 SDK API Graph: STEP_3 totals
119
+ index STEP_1 decompose
120
+ STEP_4 LLM STEP_2 validate & clean
121
+ C0/C1/C2 STEP_2_5 standardize vs
122
+ STEP_5 verify mlo-knowlege-tree.tsv <== knowledgeTreePath enters
123
+ vs source here, fail-closed (R5, R6)
124
+ STEP_6 escalate STEP_3 detect + break cycles (R4)
125
+ keywords STEP_4 topological sort
126
+ STEP_7 resolve to STEP_5 ULO/CIO/SIO authoring
127
+ Master Tree STEP_5b SIO keyword enrichment
128
+ concepts STEP_5c minute calibration
129
+ STEP_8 assemble STEP_6 verify + coverage audit
130
+ roadmap (feed is
131
+ emitted by the
132
+ ROUTER, not this
133
+ pipeline - R3)
134
+ | | |
135
+ +---------------------+---------------------+
136
+ v
137
+ +------------------------------------------------------------------------------+
138
+ | LAYER 3 - emitCurriculumFeed(graph) - pure projection, no LLM |
139
+ | Called from THREE places (R3): router for the project branch |
140
+ | (domainRouter.ts:171), knowledge orchestrator, hybrid orchestrator. |
141
+ | learning_nodes[] - dependency_edges[] - suggested_groupings[] - phases[] |
142
+ | (phases[] is filled ONLY by the hybrid emitter - R1) |
143
+ +------------------------------------+-----------------------------------------+
144
+ v
145
+ +------------------------------------------------------------------------------+
146
+ | LAYER 4 - validateCurriculumFeed(feed) - structural + teaching order |
147
+ | Entry dispatches by graph shape; anything that already looks like a |
148
+ | feed is pass-through validated (R7) |
149
+ +------------------------------------------------------------------------------+
150
+ v
151
+ CurriculumFeed (schema_version: 2)
152
+ ||
153
+ || ====== package boundary ======
154
+ || nothing below this line is
155
+ || domain-kit's business
156
+ v
157
+ curriculum-kit -> units -> sessions -> lessons -> activities
158
+ ```
159
+
160
+ Risk tags (R1-R9) are defined in the Risk map below.
161
+
162
+ Two facts about that diagram are easy to miss and worth stating up front:
163
+
164
+ 1. **Only the hybrid path fills `phases[]`.** `emitFromProject` and `emitFromKnowledge`
165
+ both return `phases: []`. "The order in the file is the order in the classroom"
166
+ is therefore a guarantee that only hybrid consumers can rely on — a project-only or
167
+ knowledge-only feed has no phase authority at all.
168
+ 2. **`runProjectGraphPipeline` stops at STEP_8.** Feed emission for the product branch is
169
+ done by `runDomainPipeline()` (static import of the emitter - the old dynamic
170
+ `import()` was removed as pure inconsistency), not by the project pipeline itself.
171
+ The knowledge and hybrid branches emit from their own thin orchestrators in
172
+ `src/pipeline/`.
173
+
174
+ 3. **Hybrid degrades loudly; explicit requests fail closed.** An AUTO-detected
175
+ `hybrid` without a `repoDir` runs the concept branch instead and returns a
176
+ structured `modeDowngrade: { from, to, reason }` on the result (R1). A
177
+ FORCED `forcedLearningMode: 'hybrid'` without `repoDir` throws. Both hybrid
178
+ knowledge legs pass `knowledgeTreePath`/`proposalsPath` through, so STEP_2_5
179
+ tree standardization runs in every mode (R2). Both fixed 2026-10-07.
180
+
181
+
182
+ ### Risk map (2026-10-07 - re-verified against code)
183
+
184
+ Written while re-drawing the diagram above; every row cites the code it was
185
+ verified against. Full postmortems: deep review 2026-10-05 (app repo,
186
+ `docs/analysis/2026-10-05-graph-curriculum-deep-review.md`) and
187
+ [ADR-0002](docs/adr/2026-10-05-ADR-0002-graph-integrity-provenance-and-diagnostics.md).
188
+
189
+ | # | Risk | Where | Severity |
190
+ |---|---|---|---|
191
+ | R1 | **FIXED 2026-10-07.** Was: `hybrid` without `repoDir` silently degraded to a knowledge-graph feed (only a `warnings[]` entry). Now a FORCED hybrid throws (fail-closed); an auto-detected downgrade returns a structured `modeDowngrade { from, to, reason }`. | `src/pipeline/domainRouter.ts` | High |
192
+ | R2 | **FIXED 2026-10-07.** Was: the hybrid branch's knowledge legs dropped `knowledgeTreePath`/`proposalsPath`, so STEP_2_5 ran in concept mode but not hybrid mode for the same subject. Now both hybrid legs pass them through. | `src/pipeline/domainRouter.ts` | High |
193
+ | R3 | Feed emission is invoked from three different places, and emission options already diverge (`hallucinationCount` is only passed by the router). Drift risk with every emitter change. | `domainRouter.ts:171`, `pipeline/knowledgeGraphPipeline.ts:37`, `pipeline/hybridGraphPipeline.ts:22` | Medium |
194
+ | R4 | Cycle breaking mutates `prerequisites` in place ("weakest edge" heuristic) and only warns. Which edge is dropped decides teaching order - a pedagogy-relevant decision hidden in a warning string. | `src/graph/knowledgeGraphPipeline.ts:383` | Medium |
195
+ | R5 | STEP_2_5 only runs when `knowledgeTreePath` is provided. Without it, every concept stays `proposed_new` and nothing in the output flags the absence of tree anchoring. | `src/graph/knowledgeGraphPipeline.ts:322` | Medium |
196
+ | R6 | Tree prerequisite edges are seeded only when the referenced tree concept is part of THIS analysis - partial graphs silently lose cross-analysis prerequisite edges. Matters for gold-set coverage measurement across runs. | `src/graph/knowledgeGraphPipeline.ts:363-375` | Medium |
197
+ | R7 | `emitCurriculumFeed` dispatches by shape (key presence) and pass-through-validates anything that "looks like a feed" (CF-8 in the deep review). | `src/feed/curriculumFeedEmitter.ts:833-839` | Low |
198
+ | R8 | Non-fatal pedagogy issues funnel into a single lossy `warnings[]` string channel; some drops only hit `console.warn` (e.g. a concept falling out of `learning_path`, KG-3). Structured diagnostics per ADR-0002 should replace string warnings. | deep review KG-3 | Medium |
199
+ | R9 | **FIXED 2026-10-07.** Was: ASCII-only regex assumptions broke on Vietnamese text (KG-1) - empty derived keywords threw, registries filled with fragments. Now a shared `stripDiacritics()` (NFD + combining-mark strip + `đ`→`d`, incl. the precomposed-base-letter case NFD cannot decompose) backs every tokenizer; regression tests in `src/__tests__/vietnamese-tokenization.test.ts`. | `src/utils/diacritics.ts` | Critical |
200
+
201
+ ### Layer 1 - Detect
202
+
203
+ `detectDomainProfile()` runs three detectors in sequence and never calls an LLM:
204
+
205
+ - **Input type** — `repository | syllabus | description | files | mixed`
206
+ - **Domain category** — multi-signal scoring over 340+ keywords in 10 categories,
207
+ with Vietnamese / English / Chinese / Japanese / French keyword lists and
208
+ context-dependent extension mapping (`.cpp` is ambiguous: software by default,
209
+ `hardware_iot` only with Arduino/ESP32/sensor context).
210
+ - **Learning mode** — `product | concept | hybrid`, the signal that actually picks
211
+ the branch in layer 2.
212
+
213
+ `graphType` on the profile is a convenience mirror of `learningMode`
214
+ (`product→project_graph`, `concept→knowledge_graph`, `hybrid→hybrid_graph`); when the
215
+ two would disagree, `forcedLearningMode` overrides the mode and `graphType` is
216
+ recomputed at the branch.
217
+
218
+ ### Layer 2 - Generate (one branch by learning mode)
219
+
220
+ #### Branch A — `project_graph` (product-driven)
221
+
222
+ ```
223
+ STEP_1 scan the file TREE, names only (no content)
224
+ └─▶ LLM picks the relevant files ─▶ read content for those files only
225
+ └─▶ parse .swift / .ts .tsx / .js .jsx / .py / .ino .cpp .c .h .hpp
226
+ └─▶ merge results by PRIMARY language
227
+ STEP_2 extractKeywords() → weighted keywords (app / esp32 tags)
228
+ STEP_3 SDK API index, derived from STEP_2 keywords
229
+ STEP_4_C0 scaffold → feature F0 "FOUNDATION & SETUP" (LLM failure ⇒ F0 is skipped
230
+ silently and the pipeline continues without it)
231
+ STEP_4_C1 overview → product{ goals, users, journeys, development_stages } + features
232
+ STEP_4_C2 steps → per-feature steps, one batched LLM call, falling back to
233
+ per-feature; a feature that still fails is logged and left
234
+ step-less (no throw)
235
+ STEP_5 verifyProjectGraph() → strip files / APIs / keywords not found in code;
236
+ F0 keywords are exempt (pedagogical terms)
237
+ STEP_6 escalateAndMapConcepts() → keyword → technology-neutral concept +
238
+ depth (ulo | cio | sio) + evidence files.
239
+ Depth comes from the LLM; on LLM failure a
240
+ regex fallback reads the usage context
241
+ (import → sio, code body → sio, comment → ulo)
242
+ STEP_7 resolveConcepts() → concept → Master-Tree code, scored ULO/CIO-aware:
243
+ sio keyword → concept.keywords
244
+ cio keyword → concept.description + concept.cio
245
+ ulo keyword → concept.ulo + concept.description
246
+ STEP_8 assembleRoadmap() → milestones grouped into phases, each milestone tagged
247
+ with all / new / prerequisite keywords + a time budget
248
+ ───────────────────────── package boundary ───────────────────────────────
249
+ emitCurriculumFeed(projectGraph) → CurriculumFeed (phases: [])
250
+ ```
251
+
252
+ #### Branch B — `knowledge_graph` (concept-driven)
253
+
254
+ ```
255
+ STEP_1 parseSyllabus(syllabusText) → structured units/topics
256
+ LLM decomposes the subject → concepts + prerequisites + problem_types
257
+ + categories + techniques. The syllabus is truncated at
258
+ KG_MAX_SYLLABUS_CHARS (default 16000) with a warning.
259
+
260
+ STEP_2 validate & clean
261
+ · description shorter than 50 chars → warning
262
+ · KEYWORD REGISTRY IS FAIL-CLOSED: a concept with no keywords gets them
263
+ derived from its name/description; if nothing is derivable → throw
264
+ · a prerequisite pointing at a non-existent concept → edge removed + warning
265
+
266
+ STEP_2_5 only when knowledgeTreePath is supplied: standardize every concept against
267
+ mlo-knowlege-tree.tsv → `standardized` (adopts the tree code + the keywords
268
+ the standardizer chose for this tech stack) or `proposed_new`.
269
+ A concept with no decision → throw. Tree prerequisite codes then seed
270
+ internal edges where both ends are in this analysis.
271
+
272
+ STEP_3 detectPrerequisiteCycles() → breakPrerequisiteCycles() at the weakest edge
273
+ STEP_4 topologicalSort() (Kahn) → learning_path is REBUILT from the sorted order
274
+ STEP_5 hierarchical ULO/CIO/SIO authoring, then deterministic ULO/CIO/SIO codes
275
+ (a tree-standardized code from STEP_2_5 always wins over a generated one)
276
+ STEP_5b mine the SIO text for technology identifiers — backticked tokens,
277
+ @Annotations, a.b() calls, .modifiers, camelCase — and merge them into the
278
+ keyword registry (capped at 8 per concept)
279
+ STEP_5c minute calibration:
280
+ score = 0.2·sio + 0.15·problem_types + 0.15·prerequisites
281
+ + 0.2·(has Analyze/Evaluate/Create)
282
+ minutes = 15 + 45·score ⇒ 15 … 60 minutes
283
+ The calibrated value overrides the LLM estimate only when the LLM returned
284
+ its default 30, or when the two diverge by more than 50%.
285
+
286
+ STEP_6 verifyKnowledgeGraph() → referential integrity + REVERSE COVERAGE audit:
287
+ every syllabus topic must be covered by at least one concept.
288
+ Findings are reported as warnings, not thrown.
289
+ ───────────────────────── package boundary ───────────────────────────────
290
+ emitCurriculumFeed(knowledgeGraph) → CurriculumFeed (phases: [])
291
+ ```
292
+
293
+ #### Branch C — `hybrid_graph` (both, and how they interlock)
294
+
295
+ Needs a `project_graph` **and** a `knowledge_graph` — it is the only branch that runs two
296
+ generators before it starts.
297
+
298
+ ```
299
+ input: project_graph + knowledge_graph
300
+
301
+ STEP_1 LLM emits feature ↔ concept links
302
+ { feature_id, concept_id, relationship: requires|demonstrates|applies|extends,
303
+ depth: ulo|cio|sio, estimated_prereq_minutes }
304
+ A link naming an unknown feature or concept is DROPPED with a warning.
305
+
306
+ STEP_2 decomposePhases() — the LLM turns the build into PROGRESSIVE COMPLETION
307
+ PHASES: each phase states what the product can demonstrably do once it is
308
+ done, owns every feature built in it, and declares which concepts it teaches
309
+ at its start (aspect = new | advanced | preview).
310
+ auditPhasePlan() then checks the contract deterministically:
311
+ · every feature in EXACTLY one phase, none skipped
312
+ · every phase has a non-empty product_completion
313
+ · a concept is introduced no later than the phase that consumes it
314
+ · no concept before its own prerequisites — checked per PHASE, not per array
315
+ position: teaching a concept and its prerequisite in the SAME phase is valid
316
+ whichever order the LLM emitted them, and flagging it would burn all three
317
+ repair attempts and then throw away a good plan
318
+ · a preview is satisfied by an earlier anchor or an earlier preview, and every
319
+ preview has an anchor teach somewhere
320
+ · preview must precede its anchor and precede nothing already introduced
321
+ Violations are fed back verbatim; at most 3 attempts, then throw. A phase
322
+ model that cannot be trusted is worse than no output.
323
+
324
+ STEP_3 totals
325
+ total_project_minutes = Σ step.effort.estimated_minutes (fallback 15)
326
+ total_concept_minutes = Σ concept.estimated_minutes (fallback 30)
327
+ linked_prereq_minutes = total_project_minutes + Σ max-per-concept prereq
328
+ minutes, counting only concepts that HAVE a feature
329
+ link. It is NOT the course total.
330
+ ───────────────────────── package boundary ───────────────────────────────
331
+ emitCurriculumFeed(hybridGraph) → CurriculumFeed — the only path that
332
+ fills phases[], and therefore the only path that defines teaching order.
333
+ ```
334
+
335
+ ### Layer 3 - Emit (one shape for every graph)
336
+
337
+ ```
338
+ emitCurriculumFeed(graph, opts?)
339
+ ├─ graph already has learning_nodes[] → validate and pass through unchanged
340
+ ├─ has project_graph | knowledge_graph | links[] | phases[]
341
+ │ → emitFromHybrid()
342
+ ├─ has concepts[] or type==='knowledge_graph'
343
+ │ → emitFromKnowledge()
344
+ └─ otherwise → emitFromProject()
345
+ ```
346
+
347
+ **`emitFromHybrid` — phase-driven emission.** For each phase, in order, it emits
348
+ `new` concepts first, then `advanced` revisits, then `previews`, and only then the
349
+ phase's features and their steps (project-graph order):
350
+
351
+ | Aspect | Node id | Minutes | Carries |
352
+ |---|---|---|---|
353
+ | `new` | `<cid>` | the concept's own estimate | `keywords.new` — the legitimate vocabulary introducer |
354
+ | `advanced` | `<cid>__ADV<phase>` | `max(15, 50% of anchor)` | a `sio → cio` downgrade candidate saving 30% **of the revisit's own minutes** |
355
+ | `preview` | `<cid>__PREV<phase>` | `max(5, 25% of anchor)` | `keywords.all` — may introduce the vocabulary early, with **no** prerequisite edges and no scaffolding |
356
+
357
+ **A preview in an EARLIER phase satisfies a downstream concept's prerequisite** — that is
358
+ what it is for. A preview in the *same* phase does not, because within a phase the order
359
+ is `new → advanced → preview`, so it would arrive after the dependent concept.
360
+
361
+ ### Knowledge edges
362
+
363
+ Edges are materialized as *concept → first step of the consumer feature* (`kind: 'knowledge'`);
364
+ feature build order and `depends_on` become `kind: 'task'`. When a prerequisite is satisfied
365
+ only by a preview, the edge points at the **preview node** — the vocabulary genuinely comes
366
+ from there.
367
+
368
+ Keyword identity is case- and Unicode-insensitive: `SwiftUI` / `swiftui` and an NFD/NFC
369
+ Vietnamese pair are the same keyword. The ledger and the edge materializer share one
370
+ comparison key, so they cannot disagree about what counts as a first appearance.
371
+
372
+ Edges are materialized as *concept → first step of the consumer feature* (`kind: 'knowledge'`);
373
+ feature build order and `depends_on` become `kind: 'task'`.
374
+
375
+ Where the graph contradicts itself, the emitter **reports instead of inventing**:
376
+
377
+ - a link whose anchor teach sits in a LATER phase than its consumer → `hybrid_link_drift`, no edge;
378
+ - a product step that would become the introducer of a concept's vocabulary → `concept_overtaken_by_step`, no edge;
379
+ - a keyword introducer that sits in a later phase than its consumer → `keyword_order_drift`, no edge.
380
+
381
+ That last class of edge is what once produced the 2026-09-17 "Teaching-order violation"
382
+ incident: the contradiction used to escape emission and explode inside the planner.
383
+
384
+ #### Coverage holes that used to be silent
385
+
386
+ | Diagnostic | Severity | Meaning |
387
+ |---|---|---|
388
+ | `orphan_concept` | `error` | the knowledge graph holds a concept that no phase introduces and no link consumes — it will not be taught |
389
+ | `feature_without_steps` | `warning` | the project graph declares a feature whose steps never materialised (STEP_4_C2 failed for it) — it will not appear in the feed |
390
+ | `step_without_id` | `warning` | a step with no `id` cannot become a node and was dropped (reported as `feature#index`) |
391
+ | `external_prerequisite` | `info` | a concept depends on knowledge outside this analysis — recorded on the node as `external_prerequisites`, not an error |
392
+ | `upstream_warning` | `warning` | a warning string handed in by the graph pipeline, kept for backward compatibility |
393
+
394
+ Both of the first two used to vanish with an empty `warnings: []`. The host can gate on
395
+ `code`/`severity` instead of substring-matching free text.
396
+
397
+ ### Entry requirements
398
+
399
+ A concept can depend on knowledge the analysis does not cover — a course on vector
400
+ calculus legitimately requires linear algebra. Those ids are **not** nodes here, so they
401
+ cannot be edges; dropping them silently would throw away exactly what someone deciding
402
+ whether to enrol needs most.
403
+
404
+ They land on the node instead:
405
+
406
+ ```json
407
+ { "id": "C7", "kind": "concept", "external_prerequisites": ["C_LINALG"] }
408
+ ```
409
+
410
+ and surface once as an `external_prerequisite` **info** diagnostic. Info, not warning —
411
+ a course with entry requirements is normal, and a host that gates on severity should not
412
+ block on it. The three emission paths used to disagree here: the knowledge path dropped
413
+ them silently while **both** hybrid paths hard-failed (one built a dangling edge, the other
414
+ threw an ordering error blaming a concept that was not in the analysis), so any graph
415
+ referencing prior knowledge was un-emittable.
416
+
417
+ ### Provenance
418
+
419
+ `source.provenance` records which providers actually served the run:
420
+
421
+ ```typescript
422
+ provenance: {
423
+ providers: ['openrouter', 'nvidia'], // failover is visible, not silent
424
+ models: ['preset-x', 'nemotron-3-ultra'],
425
+ calls: 7,
426
+ failed_calls: 1, // openrouter fell through
427
+ degraded: true,
428
+ }
429
+ ```
430
+
431
+ Without it, a graph produced half by a paid preset and half by a free tier is
432
+ indistinguishable from a clean one — and not reproducible or auditable. The client is
433
+ created **once per pipeline run** and threaded through every step, so all calls land in
434
+ one log; it is per-instance, so concurrent runs never inherit each other's history.
435
+
436
+ ### Layer 4 - Validate (fail-closed, at the source)
437
+
438
+ `validateCurriculumFeed()` runs on every emission path and rejects:
439
+
440
+ - duplicate node ids;
441
+ - an edge pointing at an unknown node, or a self-loop;
442
+ - a grouping or phase referencing an unknown node;
443
+ - duplicate phase ids;
444
+ - **duplicate or non-ascending phase `order`** — `order` *is* the teaching sequence once
445
+ phases exist, so two phases claiming one slot makes it undefined;
446
+ - an empty feed (`learning_nodes` has a minimum of 1);
447
+ - **the teaching-order invariant** — when `phases[]` is non-empty, a prerequisite must
448
+ live in the same phase or an *earlier* one than its consumer.
449
+
450
+ `emitFromProject` adds one more: a graph with `features[]` but no `features[].steps` is
451
+ rejected by name as a legacy knowledge-tree graph, with the regeneration command in the
452
+ message.
453
+
454
+ ### The three graph types at a glance
455
+
456
+ | | `project_graph` | `knowledge_graph` | `hybrid_graph` |
457
+ |---|---|---|---|
458
+ | Answers | "what is being built?" | "what must be understood?" | "both, and how they interlock" |
459
+ | LLM sees | file tree + selected file contents | syllabus / description | both graphs, rendered as text |
460
+ | Atomic node | `features[].steps[]` | `concepts[]` | both |
461
+ | Ordering authority | array order of features/steps | `learning_path` (topological) | `phases[]` → `feed.phases` |
462
+ | Time source | `step.effort.estimated_minutes` | `concept.estimated_minutes` (calibrated 15–60) | sum of both |
463
+ | Prerequisite evidence | none — build order only | `concept.prerequisites[]`, proven acyclic | both, cross-checked against concept prerequisites |
464
+ | `feed.phases` | always `[]` | always `[]` | filled ⇒ teaching order |
465
+ | Entry point | `runProjectGraphPipeline()` | `runKnowledgeGraphPipeline()` | `runHybridGraphPipeline()` |
466
+
467
+ ### Depth and minute conventions
468
+
469
+ Exported from code as `DEPTH_MINUTE_FRACTIONS` and `calibrateConceptMinutes()` — read
470
+ those rather than copying numbers out of this table.
471
+
472
+ | Convention | Value | Source |
473
+ |---|---|---|
474
+ | preview node minutes | `max(5, 25%)` of the anchor | `DEPTH_MINUTE_FRACTIONS` |
475
+ | ULO / CIO / SIO minutes for a concept node | `max(5, 35%)` / `max(ulo+5, 65%)` / `max(cio+5, 100%)` | `DEPTH_MINUTE_FRACTIONS` |
476
+ | advanced revisit minutes | `max(15, 50%)` of the anchor | `DEPTH_MINUTE_FRACTIONS` |
477
+ | advanced revisit downgrade candidate | `sio → cio`, saves 30% **of the revisit's own minutes** | `DEPTH_MINUTE_FRACTIONS` |
478
+ | step with no `estimated_minutes` | 15 minutes (same constant the hybrid totals use) | `DEFAULT_STEP_MINUTES` |
479
+ | concept minutes after calibration | 15–60 minutes | `calibrateConceptMinutes()` |
480
+ | roadmap session / overhead / milestone split cap | 90 min / ×1.15 / 120 min — hardcoded by the pipeline, not configurable through it | `STEP_8` in `pipeline/projectGraphPipeline.ts` |
481
+ | phase minute balance demanded of the LLM | 120–240 min per phase; rebalance below 60 or above 400 | `STEP_2` in `graph/hybridGraphPipeline.ts` |
482
+ | file budget per run | 70 files / 500 000 chars | `utils/fileUtils.ts` |
483
+
484
+ > The three depth levels are a **single ladder applied everywhere**:
485
+ > `ulo` = WHAT + WHY (technology-agnostic), `cio` = HOW the mechanism works
486
+ > (still no concrete API), `sio` = IMPLEMENTATION (must name a real API, function,
487
+ > module or file). A keyword can only ever move *down* this ladder, never across it.
488
+ >
489
+ > The percentages above live in one exported table, `DEPTH_MINUTE_FRACTIONS`. They were
490
+ > previously hardcoded in three places (25% / 35% / 30%), which meant "what does a ULO
491
+ > pass cost?" had three answers. Read that table rather than hardcoding a fourth.
492
+
493
+ > **Unicode:** Vietnamese input reaches the detector in NFC or NFD form. Text is
494
+ > normalized at the boundary and also matched with diacritics stripped, so the same
495
+ > sentence classifies identically whichever form it arrives in — before this, one
496
+ > sentence measured as `math` (confidence 1.0) in NFC and `other` (confidence 0.2) in NFD.
48
497
 
49
498
  ---
50
499
 
@@ -74,9 +523,34 @@ npm link @thanh01.pmt/domain-kit
74
523
  LLM_API_KEY=sk-... # OpenAI-compatible API key
75
524
  LLM_BASE_URL=https://api.openai.com/v1 # API base URL
76
525
  LLM_MODEL=gpt-4o-mini # Model to use
77
- LLM_MAX_TOKENS=16384 # Max output tokens
526
+ LLM_MAX_TOKENS=65536 # Max output tokens (default)
527
+ LLM_MIN_REQUEST_INTERVAL_MS=2500 # process-wide pacing; 0 disables
528
+
529
+ # Extra knobs the pipelines read directly:
530
+ KG_MAX_SYLLABUS_CHARS=16000 # syllabus truncation before decomposition
531
+ KG_MAX_TOKENS=65536 # knowledge-graph generation budget
532
+ LLM_REQUEST_TIMEOUT_MS=300000 # per-request cap
78
533
  ```
79
534
 
535
+ **Provider chain.** `createLlmClient()` builds an ordered chain and fails over through
536
+ it on transient errors (408/429/500/502/503/504):
537
+
538
+ 1. explicit `llmConfig.apiKey`
539
+ 2. `LLM_*` (or `OPENAI_*`) — generic OpenAI-compatible
540
+ 3. `DASHSCOPE_*` / `ALIBABA_*`
541
+ 4. `NVIDIA_API_KEY` → NVIDIA NIM
542
+ 5. `OPENROUTER_API_KEY` → OpenRouter
543
+
544
+ If `OPENROUTER_MODEL` is set, OpenRouter is moved to the front. Requests are paced
545
+ process-wide so free tiers are not rate-limited by a burst; the interval is configurable
546
+ via `LLM_MIN_REQUEST_INTERVAL_MS` or `llmConfig.minRequestIntervalMs` (set `0` when the
547
+ caller owns the rate limit — several concurrent pipelines in one process, or serverless).
548
+
549
+ Each client keeps a `callLog` of every provider attempt it made. One client is created
550
+ per pipeline run and threaded through every step, so `PipelineResult.provenance` and
551
+ `CurriculumFeed.source.provenance` reflect that whole run — and because the log is
552
+ per-instance, concurrent runs never inherit each other's history.
553
+
80
554
  ---
81
555
 
82
556
  ## Quick Start
@@ -138,7 +612,7 @@ const result = await generateHybridGraph({
138
612
  knowledgeGraph: existingKnowledgeGraph,
139
613
  });
140
614
 
141
- console.log(result.hybridGraph); // HybridGraph with links[]
615
+ console.log(result.hybridGraph); // HybridGraph with links[] + phases[]
142
616
  ```
143
617
 
144
618
  ### 5. Emit Curriculum Feed (from any graph)
@@ -146,7 +620,11 @@ console.log(result.hybridGraph); // HybridGraph with links[]
146
620
  ```typescript
147
621
  import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
148
622
 
149
- const feed = emitCurriculumFeed(result.projectGraph); // or knowledgeGraph or hybridGraph
623
+ // `graph` is whichever graph object you just produced — note that step 4 above
624
+ // rebinds `result` to the HYBRID result, so name the graph explicitly:
625
+ const feed = emitCurriculumFeed(projectGraph); // project_graph
626
+ // const feed = emitCurriculumFeed(knowledgeGraph); // knowledge_graph
627
+ // const feed = emitCurriculumFeed(hybridGraph); // hybrid_graph (phases filled)
150
628
 
151
629
  console.log(feed.learning_nodes.length); // unified nodes ready for planner
152
630
  console.log(feed.dependency_edges.length); // knowledge + task edges
@@ -159,48 +637,22 @@ console.log(feed.phases.length); // non-empty ⇒ learning_nodes are
159
637
 
160
638
  ## Architecture
161
639
 
162
- ```
163
- ┌─────────────────────────────────────────────────────────────┐
164
- │ domain-kit │
165
- ├─────────────────────────────────────────────────────────────┤
166
- │ │
167
- │ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ │
168
- │ │ schemas/ │ │ detector/ │ │ parsers/ │ │
169
- │ │ │ │ │ │ │ │
170
- │ │ ProjectGraph │ │ inputType │ │ Swift │ │
171
- │ │ KnowledgeGr. │ │ domain │ │ TypeScript │ │
172
- │ │ HybridGraph │ │ learningMode │ │ Python │ │
173
- │ │ DomainProfile│ │ orchestrator │ │ C++/Arduino │ │
174
- │ │ Extensions │ │ │ │ │ │
175
- │ └──────────────┘ └──────────────┘ └──────────────┘ │
176
- │ │
177
- │ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ │
178
- │ │ extractors/ │ │ graph/ │ │ concepts/ │ │
179
- │ │ │ │ │ │ │ │
180
- │ │ keywords │ │ scaffold │ │ concept │ │
181
- │ │ │ │ overview │ │ resolver │ │
182
- │ │ │ │ steps │ │ │ │
183
- │ │ │ │ verify │ │ │ │
184
- │ │ │ │ escalate │ │ │ │
185
- │ │ │ │ knowledgeKG │ │ │ │
186
- │ │ │ │ hybridKG │ │ │ │
187
- │ └──────────────┘ └──────────────┘ └──────────────┘ │
188
- │ │
189
- │ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ │
190
- │ │ assembly/ │ │ pipeline/ │ │ feed/ │ │
191
- │ │ │ │ │ │ │ │
192
- │ │ roadmap │ │ projectGraph │ │ emitCurricu..│ │
193
- │ │ assembler │ │ pipeline │ │ lumFeed │ │
194
- │ └──────────────┘ └──────────────┘ └──────────────┘ │
195
- │ │
196
- │ ┌──────────────┐ │
197
- │ │ utils/ │ │
198
- │ │ llmClient │ │
199
- │ │ fileUtils │ │
200
- │ └──────────────┘ │
201
- │ │
202
- └─────────────────────────────────────────────────────────────┘
203
- ```
640
+ | Directory | Files | Owns |
641
+ |---|---|---|
642
+ | `schemas/` | `projectGraphSchema`, `knowledgeGraphSchema`, `hybridGraphSchema`, `domainProfileSchema`, `curriculumFeedSchema`, `domainExtensions` | every Zod schema; `validateCurriculumFeed()` lives in `curriculumFeedSchema` |
643
+ | `detector/` | `inputTypeDetector`, `domainDetector`, `learningModeDetector`, `domainProfileDetector` | layer 1 — pure, no LLM |
644
+ | `parsers/` | `swift`, `typescript`, `python`, `cpp`, `syllabusParser` | source + syllabus readers |
645
+ | `extractors/` | `keywordExtractor`, `fileSelector`, `fileRelevanceScorer` | keyword weights, LLM file relevance |
646
+ | `graph/` | `scaffoldExtractor`, `overviewExtractor`, `stepExtractor`, `graphVerifier`, `conceptEscalator`, `hierarchicalLoAuthoring`, `loCodeStandard`, `loExemplars`, `depthAudit`, `knowledgeTreeCatalog`, `conceptStandardizer`, `knowledgeGraphPipeline`, `knowledgeGraphVerifier`, `hybridGraphPipeline` | layer 2 — the three graph generators |
647
+ | `concepts/` | `conceptResolver` | keyword → Master-Tree code, ULO/CIO-aware |
648
+ | `assembly/` | `roadmapAssembler` | milestones, phases, keyword ledger, time budget |
649
+ | `pipeline/` | `domainRouter`, `projectGraphPipeline`, `knowledgeGraphPipeline`, `hybridGraphPipeline` | the three **orchestrators** (thin) + the router; only `projectGraphPipeline` holds the 8 real steps |
650
+ | `feed/` | `curriculumFeedEmitter` | layer 3 — projects any graph onto `CurriculumFeed` v2 |
651
+ | `utils/` | `llmClient`, `fileUtils` | provider chain + pacing, file budget |
652
+
653
+ > `graph/` and `pipeline/` both contain files named `knowledgeGraphPipeline.ts` and
654
+ > `hybridGraphPipeline.ts`. They are not duplicates: the `graph/` ones do the LLM work,
655
+ > the `pipeline/` ones parse the syllabus, call the `graph/` one, and emit the feed.
204
656
 
205
657
  ---
206
658
 
@@ -242,7 +694,12 @@ Step 2: ROUTE (by graphType)
242
694
  └── hybrid_graph → project_graph + knowledge_graph + LLM link features↔concepts
243
695
  │
244
696
  Step 3: OUTPUT
245
- └── domain_graph.json
697
+ ├── the graph object (project_graph | knowledge_graph | hybrid_graph)
698
+ └── CurriculumFeed v2 (learning_nodes / dependency_edges / groupings / phases)
699
+ ↑ always produced by emitCurriculumFeed(); hybrid is the only branch that fills phases[]
700
+
701
+ Note: domain-kit returns objects in memory — it never writes *.json itself.
702
+ Serialising them is the host's job.
246
703
  ```
247
704
 
248
705
  ### Output
@@ -264,7 +721,7 @@ Always **one of three schemas**:
264
721
  `detectDomainProfile()` runs three detectors in sequence:
265
722
 
266
723
  1. **Input Type Detector** — Checks for repo URLs, syllabus text, file extensions, descriptions
267
- 2. **Domain Detector** — Multi-signal scoring with 200+ keywords across 10 domains, **multi-language support** (EN, VI, ZH, JA, KO, FR), and **context-dependent extension mapping** (e.g., `.cpp` → software by default, hardware only when Arduino/ESP32/sensor context present)
724
+ 2. **Domain Detector** — Multi-signal scoring with 340+ keywords across 10 domains, **multi-language support** (EN, VI, ZH, JA, KO, FR), and **context-dependent extension mapping** (e.g., `.cpp` → software by default, hardware only when Arduino/ESP32/sensor context present)
268
725
  3. **Learning Mode Detector** — Combines input signals + domain heuristics + verb patterns
269
726
 
270
727
  ### Ambiguous Domain Handling
@@ -274,7 +731,7 @@ Some file extensions map to multiple domains (e.g., `.cpp` → software or hardw
274
731
  1. Default-maps ambiguous extensions to `software` (most common use)
275
732
  2. Checks context keywords (e.g., `arduino`, `ESP32`, `GPIO`, `sensor`) to upgrade to `hardware_iot`
276
733
  3. Sets `ambiguous` field in output when detection is borderline
277
- 4. Agent should **ask user to confirm** when `ambiguous` is set or confidence < 0.6
734
+ 4. Agent should **ask user to confirm** when `ambiguous` is set or `confidence.domainCategory < 0.6`
278
735
 
279
736
  ### Domain Categories
280
737
 
@@ -287,8 +744,8 @@ Some file extensions map to multiple domains (e.g., `.cpp` → software or hardw
287
744
  | `science` | Lý, hóa, sinh | thí nghiệm, năng lượng, cell |
288
745
  | `language` | Tiếng Anh, writing | grammar, vocabulary, essay |
289
746
  | `arts` | Âm nhạc, thiết kế | piano, color theory, typography |
290
- | `research` | Nghiên cứu, thống kê | methodology, hypothesis, SPSS |
291
- | `business` | Kinh doanh, tài chính | revenue, agile, business plan |
747
+ | `research` | Nghiên cứu, thống kê | methodology, hypothesis, regression |
748
+ | `business` | Kinh doanh, tài chính | revenue, agile, kế hoạch kinh doanh |
292
749
  | `other` | Fallback | — |
293
750
 
294
751
  ### Confidence & Fallback
@@ -380,9 +837,27 @@ For projects that require **both building and understanding**.
380
837
  "concept_id": "G3",
381
838
  "relationship": "requires",
382
839
  "depth": "cio",
383
- "estimated_prereq_minutes": 30
840
+ "estimated_prereq_minutes": 30,
841
+ "phase_id": "PH2"
384
842
  }
385
- ]
843
+ ],
844
+ // phases[] is the teaching order — and the only source of CurriculumFeed.phases
845
+ "phases": [
846
+ {
847
+ "id": "PH1",
848
+ "order": 1,
849
+ "name": "Runnable slice",
850
+ "product_completion": "App launches and shows an empty conversation list",
851
+ "feature_ids": ["F0", "F1"],
852
+ "introduces": [
853
+ { "concept_id": "G1", "aspect": "new", "note": "..." },
854
+ { "concept_id": "G3", "aspect": "preview", "note": "vocabulary only" }
855
+ ]
856
+ }
857
+ ],
858
+ "total_project_minutes": 480,
859
+ "total_concept_minutes": 210,
860
+ "linked_prereq_minutes": 540
386
861
  }
387
862
  ```
388
863
 
@@ -398,23 +873,34 @@ All schemas are Zod-validated TypeScript types.
398
873
  | `KnowledgeGraphSchema` | `knowledgeGraphSchema.ts` | Concept-driven subject structure |
399
874
  | `HybridGraphSchema` | `hybridGraphSchema.ts` | Combined project + knowledge |
400
875
  | `DomainProfileSchema` | `domainProfileSchema.ts` | Domain detection output |
401
- | `HardwareFeatureExtension` | `domainExtensions.ts` | Hardware overlay (wiring, materials) |
402
- | `ThreeDesignFeatureExtension` | `domainExtensions.ts` | 3D Design overlay (shapes, dimensions) |
876
+ | `HardwareFeatureExtensionSchema` | `domainExtensions.ts` | Hardware overlay (wiring, materials) |
877
+ | `ThreeDesignFeatureExtensionSchema` | `domainExtensions.ts` | 3D Design overlay (shapes, dimensions) |
878
+ | `MathConceptExtensionSchema` | `domainExtensions.ts` | Math overlay (theorems) |
879
+ | `ScienceConceptExtensionSchema` | `domainExtensions.ts` | Science overlay (experiment procedures) |
880
+ | `LanguageConceptExtensionSchema` | `domainExtensions.ts` | Language overlay (grammar rules) |
881
+ | `ConceptSchema` | `knowledgeGraphSchema.ts` | One knowledge concept |
882
+ | `LearningPathStepSchema` | `knowledgeGraphSchema.ts` | One topological-order step |
403
883
  | `CurriculumFeedSchema` | `curriculumFeedSchema.ts` | Planner-ready normalized feed (nodes, edges, groupings, phases) |
404
884
  | `FeedNodeSchema` | `curriculumFeedSchema.ts` | Individual learning node (concept, skill, or product step) |
405
885
  | `FeedEdgeSchema` | `curriculumFeedSchema.ts` | Dependency edge (knowledge or task) |
406
886
  | `FeedGroupingSchema` | `curriculumFeedSchema.ts` | Suggested grouping of nodes (feature or category) |
407
887
  | `FeedPhaseSchema` | `curriculumFeedSchema.ts` | Development phase — when non-empty, defines teaching sequence and unit structure |
888
+ | `FeedDiagnosticSchema` | `curriculumFeedSchema.ts` | Typed finding (code + severity + node ids) |
889
+ | `FeedProvenanceSchema` | `curriculumFeedSchema.ts` | Which LLM providers served the run |
408
890
 
409
891
  ### Key Types
410
892
 
411
893
  ```typescript
412
894
  // Project Graph
413
895
  type ProjectGraph = {
896
+ schema_version: 3;
414
897
  project: ProjectInfo;
415
898
  product: Product;
416
899
  features: Feature[];
900
+ capabilities: Capability[];
417
901
  implementation: { tasks: Task[] };
902
+ missing_gaps: MissingGap[];
903
+ tech_debt: TechDebt[];
418
904
  }
419
905
 
420
906
  type Feature = {
@@ -430,11 +916,16 @@ type Step = {
430
916
  keywords: string[];
431
917
  api_usage: string[];
432
918
  completion_level: 'base' | 'mvp' | 'extend' | 'polish';
433
- effort: { estimated_minutes: number; complexity: 'low' | 'medium' | 'high' };
919
+ outcome?: { user_visible: string; technical: string }; // → FeedNode.user_visible_deliverable
920
+ // all effort fields are OPTIONAL — downstream totals fall back to 15 minutes
921
+ effort?: { estimated_minutes?: number; complexity?: 'low' | 'medium' | 'high';
922
+ concepts_count?: number; files_touched?: number };
434
923
  }
435
924
 
436
925
  // Knowledge Graph
437
926
  type KnowledgeGraph = {
927
+ schema_version: 1;
928
+ type: 'knowledge_graph';
438
929
  subject: SubjectInfo;
439
930
  concepts: Concept[];
440
931
  categories: ConceptCategory[];
@@ -444,11 +935,26 @@ type KnowledgeGraph = {
444
935
  type Concept = {
445
936
  id: string;
446
937
  name: string;
938
+ description: string;
939
+ keywords: string[]; // SIO-level tech terms — drives the keyword ledger, STEP_5b enrichment
940
+ // and the SIO-binding audit; NEVER silently empty (fail-closed)
447
941
  ulo: string; // WHAT + WHY
448
942
  cio: string; // HOW
449
943
  sio: string[]; // Specific implementations
450
944
  prerequisites: string[];
945
+ estimated_minutes: number; // calibrated 15–60 (STEP_5c)
451
946
  problem_types: ProblemType[];
947
+ techniques: string[];
948
+ common_mistakes: string[];
949
+ materials: string[]; // physical lab equipment
950
+ visual_aids: string[]; // diagrams / images / animations
951
+ // codes — a knowledge-tree code from STEP_2_5 always wins over a generated one
952
+ code?: string;
953
+ standardization?: 'standardized' | 'proposed_new';
954
+ tree_code?: string;
955
+ ulo_code?: string;
956
+ cio_code?: string;
957
+ sio_codes?: string[];
452
958
  }
453
959
 
454
960
  // Domain Detection
@@ -467,27 +973,51 @@ type ConceptMapping = {
467
973
  concept_code: string;
468
974
  concept_name: string;
469
975
  depth: 'ulo' | 'cio' | 'sio'; // WHAT (ulo), HOW (cio), IMPLEMENTATION (sio)
470
- rationale: string;
976
+ depth_source: 'llm' | 'heuristic' | 'override'; // who decided it — the regex
977
+ // fallback (comment→ulo, import→sio) has no
978
+ // pedagogical basis and must stay distinguishable
979
+ depth_rationale: string;
471
980
  evidence_files: string[];
472
981
  }
473
982
 
474
- // Assembled Roadmap (with time budget)
983
+ // Assembled Roadmap (with time budget) — TypeScript interface, not a Zod schema
475
984
  type AssembledRoadmap = {
476
- phases: Phase[];
477
- all_keywords: string[];
478
- new_keywords: string[];
479
- prerequisite_keywords: string[];
985
+ project: ProjectGraph['project'];
986
+ product: ProjectGraph['product'];
987
+ phases: AssembledPhase[]; // { phase_id, phase_name, milestones, total_minutes,
988
+ // concept_progression }
989
+ feature_concepts: Map<string, ConceptMapping[]>;
480
990
  time_budget: {
481
- total_estimated_minutes: number;
482
- overhead_minutes: number;
483
- session_minutes: number;
991
+ session_minutes: number; // 90
992
+ overhead_factor: number; // 1.15
993
+ total_estimated_minutes: number; // Σ milestone.estimated_minutes (raw)
994
+ total_adjusted_minutes: number; // Σ milestone.adjusted_minutes (after overhead)
484
995
  };
485
- }
996
+ };
997
+
998
+ type AssembledMilestone = {
999
+ id: string; name: string; description: string;
1000
+ feature_id: string; feature_name: string;
1001
+ concept_code: string; depth: 'ulo' | 'cio' | 'sio';
1002
+ all_keywords: string[]; // keyword ledger lives HERE, per milestone
1003
+ new_keywords: string[];
1004
+ prerequisite_keywords: string[];
1005
+ steps: Step[];
1006
+ estimated_minutes: number;
1007
+ adjusted_minutes: number; // estimated_minutes × overhead_factor
1008
+ completion_level: string;
1009
+ };
486
1010
 
487
1011
  // Curriculum Feed (planner-ready projection)
488
1012
  type CurriculumFeed = {
489
1013
  schema_version: 2;
490
- source: { graph_type: 'project_graph' | 'knowledge_graph' | 'hybrid_graph'; warnings: string[]; hallucination_count: number };
1014
+ source: {
1015
+ graph_type: 'project_graph' | 'knowledge_graph' | 'hybrid_graph';
1016
+ warnings: string[]; // DERIVED from diagnostics (severity ≠ 'info')
1017
+ hallucination_count: number;
1018
+ diagnostics: FeedDiagnostic[]; // { code, severity, message, node_ids }
1019
+ provenance?: FeedProvenance; // providers / models / calls / degraded
1020
+ };
491
1021
  learning_nodes: FeedNode[]; // in TEACHING ORDER when phases[] is non-empty
492
1022
  dependency_edges: FeedEdge[];
493
1023
  suggested_groupings: FeedGrouping[];
@@ -514,6 +1044,7 @@ type FeedNode = {
514
1044
  depth_variants?: { ulo: number; cio: number; sio: number }; // minutes per depth level — reconciler price list (ULO ≤ CIO ≤ SIO)
515
1045
  depth_scaffold_candidates?: { from_depth: 'ulo' | 'cio' | 'sio'; to_depth: 'ulo' | 'cio' | 'sio'; minutes_saved: number; reason: string }[]; // DEPTH downgrades (SIO→CIO, CIO→ULO) — parallel to scaffold_candidates
516
1046
  is_core: boolean; // Master-Tree core concept — never depth-downgraded; escalate instead
1047
+ external_prerequisites: string[]; // what the learner must already know that this feed does not teach
517
1048
  }
518
1049
 
519
1050
  type FeedPhase = {
@@ -560,6 +1091,7 @@ const profile = detectDomainProfile({
560
1091
  fileExtensions?: string[];
561
1092
  techStack?: string;
562
1093
  gradeBand?: [number, number];
1094
+ parsedSyllabus?: ParsedSyllabus; // pre-parsed syllabus; skips parseSyllabus()
563
1095
  });
564
1096
  ```
565
1097
 
@@ -576,12 +1108,13 @@ const result = await runProjectGraphPipeline({
576
1108
  techStack: string; // Required: comma-separated technologies
577
1109
  llmConfig?: LlmClientConfig;
578
1110
  embeddings?: Record<string, { embedding?: number[] }>;
579
- sessionMinutes?: number; // Lesson/session duration (default: 90)
580
- overheadFactor?: number; // Setup/transition overhead multiplier (default: 1.15)
581
- maxMilestoneMinutes?: number; // Max minutes before splitting milestone (default: 120)
582
1111
  onProgress?: (step: string, message: string) => void;
583
1112
  });
584
1113
 
1114
+ // NOTE: sessionMinutes / overheadFactor / maxMilestoneMinutes are NOT options here.
1115
+ // STEP_8 hardcodes 90 / 1.15 / 120 when it calls assembleRoadmap(). Call
1116
+ // assembleRoadmap() yourself if you need different values.
1117
+
585
1118
  // Returns:
586
1119
  {
587
1120
  projectGraph: ProjectGraph;
@@ -590,6 +1123,7 @@ const result = await runProjectGraphPipeline({
590
1123
  hallucinations: Hallucination[];
591
1124
  keywords: Keyword[];
592
1125
  featureConcepts: Map<string, ConceptMapping[]>;
1126
+ provenance: LlmCallRecord[]; // every provider attempt this run made
593
1127
  }
594
1128
  ```
595
1129
 
@@ -606,17 +1140,26 @@ const result = await generateKnowledgeGraph({
606
1140
  syllabusText?: string; // Full syllabus text (multi-language supported)
607
1141
  gradeBand?: [number, number];
608
1142
  domain?: string;
1143
+ techStack?: string; // target tech for SIO keyword binding (e.g. "python")
1144
+ knowledgeTreePath?: string; // path to mlo-knowlege-tree.tsv → enables STEP_2_5 standardization
1145
+ proposalsPath?: string; // where proposed-new concepts are persisted for the later merge
609
1146
  llmConfig?: LlmClientConfig;
1147
+ onProgress?: (step: string, message: string) => void;
610
1148
  });
611
- // Internally runs: LLM decomposition → cycle detection → topological sort → depth estimation
1149
+ // Internally runs STEP_1 … STEP_6 — see [Pipeline Details](#pipeline-details).
612
1150
 
613
1151
  // Returns:
614
1152
  {
615
1153
  knowledgeGraph: KnowledgeGraph;
616
1154
  warnings: string[];
1155
+ verificationReport?: KnowledgeGraphVerificationReport; // uncovered topics, missing ULO/CIO/SIO…
617
1156
  }
618
1157
  ```
619
1158
 
1159
+ > Prefer `runKnowledgeGraphPipeline()` over `generateKnowledgeGraph()`: it parses the
1160
+ > syllabus, applies syllabus-derived defaults (subject, domain, grade band) and emits the
1161
+ > `CurriculumFeed` for you.
1162
+
620
1163
  ### `generateHybridGraph(options) → HybridGraphPipelineResult`
621
1164
 
622
1165
  Links project graph with knowledge graph.
@@ -626,13 +1169,15 @@ import { generateHybridGraph } from '@thanh01.pmt/domain-kit';
626
1169
 
627
1170
  const result = await generateHybridGraph({
628
1171
  projectGraph: ProjectGraph;
629
- knowledgeGraph: KnowledgeGraph;
1172
+ knowledgeGraph: KnowledgeGraph; // both are required
630
1173
  llmConfig?: LlmClientConfig;
1174
+ onProgress?: (step: string, message: string) => void;
631
1175
  });
632
1176
 
633
1177
  // Returns:
634
1178
  {
635
- hybridGraph: HybridGraph;
1179
+ hybridGraph: HybridGraph; // links[] + phases[] + total_project_minutes /
1180
+ // total_concept_minutes / linked_prereq_minutes
636
1181
  warnings: string[];
637
1182
  }
638
1183
  ```
@@ -654,8 +1199,13 @@ const feed = emitCurriculumFeed(existingFeed); // pass-through (already
654
1199
  const feed = emitCurriculumFeed(graph, {
655
1200
  hallucinationCount: 3, // metadata for source info
656
1201
  warnings: ['Feature F5 had low confidence'],
1202
+ provenance: { providers: ['nvidia'], models: ['nemotron'], calls: 7, failed_calls: 0, degraded: false },
657
1203
  });
658
1204
 
1205
+ // Gate on diagnostics rather than substring-matching warning text:
1206
+ const blockers = feed.source.diagnostics.filter((d) => d.severity === 'error');
1207
+ if (blockers.length) throw new Error(blockers.map((d) => d.code).join(', '));
1208
+
659
1209
  // Returns CurriculumFeed with:
660
1210
  // - learning_nodes[] — unified node format (concept | skill | product_step), in teaching order when phases[] is non-empty
661
1211
  // - dependency_edges[] — knowledge (prerequisite) or task (build order) edges
@@ -700,93 +1250,52 @@ const keywords = extractKeywords(parsedSourceContext);
700
1250
 
701
1251
  ## Pipeline Details
702
1252
 
703
- ### Project Graph Pipeline (Product-Driven)
1253
+ ### Project Graph Pipeline (product-driven)
704
1254
 
705
- ```
706
- STEP 1: AST Analysis
707
- → Parse all source files (Swift, TS, Python, C++)
708
- → Merge by language, extract imports/types/functions/wrappers
709
-
710
- STEP 2: Keyword Extraction
711
- → Filter stdlib, assign weights, tag platform (app/esp32)
712
-
713
- STEP 3: SDK API Index
714
- → Build symbol index from AST results
715
-
716
- STEP 4: LLM Graph Generation
717
- C0: Scaffold F0 (foundation features: tools, setup, minimal knowledge)
718
- C1: Overview (features meta, journeys, architecture, development stages)
719
- C2: Steps per feature (batched or per-feature LLM calls)
720
-
721
- STEP 5: Verification
722
- → Remove hallucinated files, APIs, keywords not found in code
723
- → F0 keywords exempt (pedagogical terms)
724
-
725
- STEP 6: Concept Escalation
726
- → Map keywords → neutral concepts via LLM
727
- → Classify each keyword as ULO (WHAT/WHY), CIO (HOW), or SIO (IMPLEMENTATION)
728
- → Infer depth from code context (function bodies→sio, types→cio, imports→sio, docs→ulo)
729
- → Join evidence files
730
-
731
- STEP 7: Concept Resolution
732
- → Map concepts → Master Tree codes with ULO/CIO-aware scoring:
733
- - SIO keywords matched against concept.keywords
734
- - CIO keywords matched against concept.description + concept.cio
735
- - ULO keywords matched against concept.ulo + concept.description
736
- → Depth field propagated to resolved concepts
737
-
738
- STEP 8: Assembly
739
- → Group features into phases (session-time-aware)
740
- → Respect sessionMinutes, overheadFactor, maxMilestoneMinutes constraints
741
- → Split large features across multiple milestones when exceeding time budget
742
- → Compute all_keywords / new_keywords / prerequisite_keywords per milestone
743
- → Output time_budget with total_estimated_minutes + overhead_minutes
744
-
745
- STEP 9: Feed Emission
746
- → emitCurriculumFeed(graph) — pure transformation, no LLM
747
- → Convert project_graph → FeedNode[] (kind: product_step) + FeatureGroupings
748
- → Convert knowledge_graph → FeedNode[] (kind: concept) + CategoryGroupings
749
- → Convert hybrid_graph → combined nodes + knowledge/task edges + hybrid link depth hints
750
- → Project steps: project outcome.user_visible onto user_visible_deliverable (checkpoint candidate)
751
- → Concept nodes: emit depth_variants (ULO ≤ CIO ≤ SIO minutes) + depth_scaffold_candidates + is_core
752
- → Teaching order: when phases[] exist, nodes are emitted in phase sequence; new/prerequisite
753
- keywords computed in TEACHING order (first appearance = new), never array position
754
- → Preview priming: phase-declared __PREVxx awareness nodes carry keywords.all (legitimate
755
- vocabulary introducer) without fabricating prerequisite edges
756
- → Validate: fail-closed (duplicate IDs, edge resolution, self-loops, non-empty, duplicate
757
- phase ids, and the teaching-order invariant: a prerequisite must live in the same or an
758
- EARLIER phase than its consumer — a backwards edge throws at emission, never in the planner)
759
- ```
1255
+ `runProjectGraphPipeline()` — the only pipeline whose steps live inside its orchestrator.
1256
+ Step ids below are the ones it actually logs.
760
1257
 
761
- ### Knowledge Graph Pipeline (Concept-Driven)
1258
+ | Step | What happens | Fail mode |
1259
+ |---|---|---|
1260
+ | `STEP_1` | scan the file **tree** (names only) → LLM selects relevant files → read only those → parse → merge by primary language | parse failure is per-file, non-fatal |
1261
+ | `STEP_2` | `extractKeywords()` — filter stdlib, assign weights, tag platform (`app` / `esp32`) | — |
1262
+ | `STEP_3` | SDK API index, derived from STEP_2 keywords | — |
1263
+ | `STEP_4_C0` | scaffold → feature `F0` "FOUNDATION & SETUP" | scaffold extractor returns `null`; the pipeline continues **without** F0 |
1264
+ | `STEP_4_C1` | overview → product meta (goals, users, journeys, development stages) + the feature list | — |
1265
+ | `STEP_4_C2` | steps per feature — one batched LLM call, falling back to per-feature | a feature that still fails is logged and left step-less; **no throw** |
1266
+ | `STEP_5` | `verifyProjectGraph()` — strip files / APIs / keywords not found in code; F0 keywords exempt | findings returned as `hallucinations[]`, graph is repaired |
1267
+ | `STEP_6` | `escalateAndMapConcepts()` — keyword → neutral concept + `ulo`/`cio`/`sio` + evidence files. Depth from the LLM; regex fallback on LLM failure (import → sio, code body → sio, comment → ulo) | LLM failure degrades to the regex fallback |
1268
+ | `STEP_7` | `resolveConcepts()` — concept → Master-Tree code, scored ULO/CIO-aware | unmatched keywords become `proposed` concepts |
1269
+ | `STEP_8` | `assembleRoadmap()` — milestones grouped into phases; each milestone gets `all` / `new` / `prerequisite` keywords and a time budget (90 min sessions, ×1.15 overhead, 120 min split cap — hardcoded) | — |
1270
+
1271
+ There is **no STEP 9 here.** Feed emission happens one layer up: `runDomainPipeline()`
1272
+ calls `emitCurriculumFeed()` for the product branch, and the knowledge / hybrid
1273
+ orchestrators call it for theirs.
1274
+
1275
+ ### Knowledge Graph Pipeline (concept-driven)
1276
+
1277
+ `generateKnowledgeGraph()` — the LLM work. Step ids as logged:
1278
+
1279
+ | Step | What happens | Fail mode |
1280
+ |---|---|---|
1281
+ | `STEP_1` | parse the syllabus into structured units/topics, then LLM-decompose the subject into concepts, prerequisites, `problem_types`, categories | syllabus truncated at `KG_MAX_SYLLABUS_CHARS` (default 16000) + warning |
1282
+ | `STEP_2` | validate & clean — description ≥ 50 chars; **keyword registry fail-closed** (empty keywords are derived from name/description; nothing derivable → `throw`); dangling prerequisite edges removed | `throw` when keywords cannot be derived |
1283
+ | `STEP_2_5` | *only when `knowledgeTreePath` is supplied* — standardize every concept against `mlo-knowlege-tree.tsv` into `standardized` or `proposed_new`; adopted tech-stack keywords merge into the registry; tree prerequisite codes seed internal edges | missing decision for a concept → `throw` |
1284
+ | `STEP_3` | `detectPrerequisiteCycles()` → `breakPrerequisiteCycles()` at the weakest edge | warnings; edges are removed |
1285
+ | `STEP_4` | `topologicalSort()` (Kahn) — `learning_path` is **rebuilt** from the sorted order | incomplete sort → warning |
1286
+ | `STEP_5` | hierarchical ULO/CIO/SIO authoring + deterministic `ulo_code` / `cio_code` / `sio_codes` (a STEP_2_5 tree code always wins) | — |
1287
+ | `STEP_5b` | mine SIO text for tech identifiers (backticked tokens, `@Annotations`, `a.b()`, `.modifiers`, camelCase, cap 8) → merge into keywords | — |
1288
+ | `STEP_5c` | minute calibration — each signal normalised against its own cap before weighting (`0.30·sio + 0.25·problem_types + 0.20·prereqs + 0.25·highBloom`, all over `min(count/4, 1)`), `minutes = 15 + 45·score` ⇒ 15–60 min. Overrides the LLM only when the two diverge by more than 50% | — |
1289
+ | `STEP_6` | `verifyKnowledgeGraph()` — referential integrity + **reverse coverage** (every syllabus topic covered by ≥ 1 concept) | warnings, never throws |
762
1290
 
763
- ```
764
- STEP 1: LLM Decomposition
765
- → Decompose subject into concepts with ULO/CIO/SIO
766
- → Assign prerequisites
767
- → Generate problem_types per concept (bloom levels)
768
- → Group into categories
769
-
770
- STEP 2: Prerequisite Cycle Detection
771
- → DFS to detect cycles in prerequisite graph
772
- → Break cycles by removing weakest back-edges
773
- → Generate warnings for each broken cycle
774
-
775
- STEP 3: Topological Sort (Kahn's Algorithm)
776
- → Sort concepts in valid learning order
777
- → Assign sequential order numbers
778
- → Estimate concept depth per milestone (ULO→CIO→SIO progression)
779
-
780
- STEP 4: Validation
781
- → Check prerequisite references exist
782
- → Check learning_path references exist
783
- → Generate warnings for invalid references
784
-
785
- STEP 5: Feed Emission
786
- → emitCurriculumFeed(knowledgeGraph) → CurriculumFeed
787
- → Concept nodes with bloom_hint (from problem_types) + depth_hint (from sio count)
788
- → Category groupings for suggested lesson grouping
789
- ```
1291
+ Feed emission is **not** a step of `generateKnowledgeGraph()` — it is the next line of
1292
+ `runKnowledgeGraphPipeline()` in `src/pipeline/`.
1293
+
1294
+ ### Hybrid Graph Pipeline (both)
1295
+
1296
+ `generateHybridGraph()` — see [Layer 2 · Branch C](#layer-2---generate-one-branch-by-learning-mode)
1297
+ for the annotated diagram. `STEP_1` links, `STEP_2` phases (up to 3 repair attempts, then
1298
+ `throw`), `STEP_3` totals.
790
1299
 
791
1300
  ---
792
1301
 
@@ -839,7 +1348,7 @@ profile.graphType === 'hybrid_graph'
839
1348
  → You need BOTH source code AND subject description.
840
1349
  → Run projectGraphPipeline() + generateKnowledgeGraph() + generateHybridGraph().
841
1350
 
842
- profile.confidence < 0.6
1351
+ profile.confidence.domainCategory < 0.6 // confidence is an OBJECT — always compare the field
843
1352
  → Low confidence. Ask the user to confirm domain category.
844
1353
 
845
1354
  profile.ambiguous !== undefined
@@ -955,10 +1464,10 @@ try {
955
1464
  2. **Missing syllabus**: `generateKnowledgeGraph()` needs `syllabusText` to decompose concepts
956
1465
  3. **Low confidence detection**: If `profile.confidence.domainCategory < 0.6`, the detection may be wrong — ask user to confirm
957
1466
  4. **Ambiguous domain**: If `profile.ambiguous` is set, present both options to user (e.g., `.cpp` with ESP32 context could be software or hardware)
958
- 5. **LLM not configured**: Graph generation (graph/, concepts/, pipeline/) requires `LLM_API_KEY` env var
1467
+ 5. **LLM not configured**: graph generation needs at least one provider key — `LLM_API_KEY` / `OPENAI_API_KEY`, `DASHSCOPE_API_KEY`, `NVIDIA_API_KEY` or `OPENROUTER_API_KEY` (see [Environment Variables](#environment-variables-for-llm-features))
959
1468
  6. **Large repos**: The pipeline reads up to 70 files / 500K chars — very large repos may be truncated
960
1469
  7. **ULO/CIO/SIO depth**: Concept escalation now classifies keywords by depth level — ULO (intro), CIO (mechanism), SIO (implementation). This affects how `curriculum-kit` structures exposition depth per lesson
961
- 8. **Time budget**: `assembleRoadmap()` respects `sessionMinutes` and `maxMilestoneMinutes` — large features are automatically split across milestones. Set these to match your curriculum constraints
1470
+ 8. **Time budget**: `assembleRoadmap()` respects `sessionMinutes` and `maxMilestoneMinutes` — large features are automatically split across milestones. `runProjectGraphPipeline()` does **not** forward these; it hardcodes 90 / 1.15 / 120. Call `assembleRoadmap()` yourself to change them
962
1471
 
963
1472
  ### Available Exports
964
1473
 
@@ -977,24 +1486,28 @@ import { extractKeywords } from '@thanh01.pmt/domain-kit';
977
1486
 
978
1487
  // Schema validation
979
1488
  import {
980
- ProjectGraphSchema, KnowledgeGraphSchema, HybridGraphSchema,
981
- DomainProfileSchema, ConceptMappingSchema, AssembledRoadmapSchema,
982
- KnowledgeConceptSchema, KnowledgeLearningPathStepSchema,
1489
+ ProjectGraphSchema, KnowledgeGraphSchema, HybridGraphSchema, DomainProfileSchema,
1490
+ ConceptSchema, LearningPathStepSchema, ProblemTypeSchema,
983
1491
  CurriculumFeedSchema, FeedNodeSchema, FeedEdgeSchema, FeedGroupingSchema, FeedPhaseSchema,
1492
+ validateCurriculumFeed,
984
1493
  } from '@thanh01.pmt/domain-kit';
985
1494
 
1495
+ // Graph verification
1496
+ import { verifyKnowledgeGraph, detectPrerequisiteCycles, breakPrerequisiteCycles, auditSyllabusCoverage }
1497
+ from '@thanh01.pmt/domain-kit';
1498
+
986
1499
  // Roadmap assembly
987
1500
  import { assembleRoadmap } from '@thanh01.pmt/domain-kit';
988
1501
 
989
1502
  // Concept resolution (ULO/CIO-aware)
990
1503
  import { resolveConcepts } from '@thanh01.pmt/domain-kit';
991
1504
 
992
- // Graph verification
993
- import { verifyProjectGraph } from '@thanh01.pmt/domain-kit';
994
-
995
1505
  // Concept escalation (with ULO/CIO/SIO depth classification)
996
1506
  import { escalateAndMapConcepts } from '@thanh01.pmt/domain-kit';
997
1507
 
1508
+ // Router — detect + dispatch + emit in one call
1509
+ import { runDomainPipeline } from '@thanh01.pmt/domain-kit';
1510
+
998
1511
  // Curriculum Feed emission (any graph → planner-ready feed)
999
1512
  import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
1000
1513
 
@@ -1002,7 +1515,7 @@ import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
1002
1515
  import { validateCurriculumFeed } from '@thanh01.pmt/domain-kit';
1003
1516
  ```
1004
1517
 
1005
- > **Note:** Internal utilities like `detectCycles`, `topologicalSort`, `inferDepthFromContext` are used within the pipelines but not exported directly. They are invoked automatically during `generateKnowledgeGraph()` and concept escalation.
1518
+ > **Note:** `topologicalSort()` (Kahn) and the depth fallback (`inferDepthFromSource`) are module-private — they run inside the pipelines. `detectPrerequisiteCycles`, `breakPrerequisiteCycles` and `auditSyllabusCoverage` **are** exported, from `graph/knowledgeGraphVerifier`.
1006
1519
 
1007
1520
  ---
1008
1521
 
@@ -1090,6 +1603,7 @@ const knowledgeResult = await generateKnowledgeGraph({
1090
1603
  subject: 'Lý thuyết điện tử',
1091
1604
  syllabusText: profile.syllabusText,
1092
1605
  domain: 'science',
1606
+ techStack: 'Arduino,ESP32', // ← binds SIO keywords to the project's real APIs
1093
1607
  });
1094
1608
 
1095
1609
  // Step C