@thanh01.pmt/domain-kit 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +682 -168
- package/dist/assembly/index.d.cts +2 -2
- package/dist/assembly/index.d.ts +2 -2
- package/dist/{chunk-TYXSJUQU.mjs → chunk-2OGNQUXC.mjs} +81 -32
- package/dist/chunk-2OGNQUXC.mjs.map +1 -0
- package/dist/{chunk-G7HLFMD2.mjs → chunk-5OMSYKQP.mjs} +43 -23
- package/dist/chunk-5OMSYKQP.mjs.map +1 -0
- package/dist/{chunk-BL5NBKPH.mjs → chunk-67GXK3ZE.mjs} +18 -8
- package/dist/chunk-67GXK3ZE.mjs.map +1 -0
- package/dist/{chunk-E4GCCWJO.mjs → chunk-CGNXGVHV.mjs} +37 -4
- package/dist/chunk-CGNXGVHV.mjs.map +1 -0
- package/dist/{chunk-VCPIWJ7R.mjs → chunk-ERDTHXQA.mjs} +21 -3
- package/dist/chunk-ERDTHXQA.mjs.map +1 -0
- package/dist/chunk-OYSZQVPY.mjs +10 -0
- package/dist/chunk-OYSZQVPY.mjs.map +1 -0
- package/dist/{chunk-TWKIVUTQ.mjs → chunk-WKTH3JPA.mjs} +197 -36
- package/dist/chunk-WKTH3JPA.mjs.map +1 -0
- package/dist/{conceptEscalator-YfMKIKCZ.d.ts → conceptEscalator-BCsw0ys_.d.ts} +10 -1
- package/dist/{conceptEscalator-ltRLtPhf.d.cts → conceptEscalator-DPrs4PwC.d.cts} +10 -1
- package/dist/concepts/index.d.cts +1 -1
- package/dist/concepts/index.d.ts +1 -1
- package/dist/{curriculumFeedSchema-DUFwFyr2.d.ts → curriculumFeedSchema-DYayelAA.d.cts} +156 -1
- package/dist/{curriculumFeedSchema-DUFwFyr2.d.cts → curriculumFeedSchema-DYayelAA.d.ts} +156 -1
- package/dist/detector/index.cjs +21 -6
- package/dist/detector/index.cjs.map +1 -1
- package/dist/detector/index.mjs +2 -1
- package/dist/feed/index.cjs +229 -33
- package/dist/feed/index.cjs.map +1 -1
- package/dist/feed/index.d.cts +32 -2
- package/dist/feed/index.d.ts +32 -2
- package/dist/feed/index.mjs +2 -2
- package/dist/graph/index.cjs +73 -29
- package/dist/graph/index.cjs.map +1 -1
- package/dist/graph/index.d.cts +14 -5
- package/dist/graph/index.d.ts +14 -5
- package/dist/graph/index.mjs +2 -1
- package/dist/{hybridGraphPipeline-DXQDbBEj.d.ts → hybridGraphPipeline-9dRKJj7r.d.ts} +6 -2
- package/dist/{hybridGraphPipeline-dT-bio-3.d.cts → hybridGraphPipeline-CocfYe4d.d.cts} +6 -2
- package/dist/{hybridGraphSchema-nrUeSwpU.d.ts → hybridGraphSchema-C37qxdTM.d.cts} +16 -3
- package/dist/{hybridGraphSchema-nrUeSwpU.d.cts → hybridGraphSchema-C37qxdTM.d.ts} +16 -3
- package/dist/index.cjs +2634 -2359
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +6 -6
- package/dist/index.d.ts +6 -6
- package/dist/index.mjs +7 -6
- package/dist/llmClient-CX5uUiQ1.d.cts +60 -0
- package/dist/llmClient-CX5uUiQ1.d.ts +60 -0
- package/dist/pipeline/index.cjs +2207 -1958
- package/dist/pipeline/index.cjs.map +1 -1
- package/dist/pipeline/index.d.cts +13 -5
- package/dist/pipeline/index.d.ts +13 -5
- package/dist/pipeline/index.mjs +6 -5
- package/dist/schemas/index.cjs +55 -2
- package/dist/schemas/index.cjs.map +1 -1
- package/dist/schemas/index.d.cts +2 -2
- package/dist/schemas/index.d.ts +2 -2
- package/dist/schemas/index.mjs +2 -2
- package/package.json +10 -9
- package/LICENSE +0 -21
- package/dist/chunk-BL5NBKPH.mjs.map +0 -1
- package/dist/chunk-E4GCCWJO.mjs.map +0 -1
- package/dist/chunk-G7HLFMD2.mjs.map +0 -1
- package/dist/chunk-TWKIVUTQ.mjs.map +0 -1
- package/dist/chunk-TYXSJUQU.mjs.map +0 -1
- package/dist/chunk-VCPIWJ7R.mjs.map +0 -1
- package/dist/curriculumFeedEmitter-T7NUBLAM.mjs +0 -4
- package/dist/curriculumFeedEmitter-T7NUBLAM.mjs.map +0 -1
- package/dist/llmClient-ysPhLjcH.d.cts +0 -16
- package/dist/llmClient-ysPhLjcH.d.ts +0 -16
package/README.md
CHANGED
|
@@ -11,6 +11,13 @@
|
|
|
11
11
|
## Table of Contents
|
|
12
12
|
|
|
13
13
|
- [Overview](#overview)
|
|
14
|
+
- [End-to-End: From Raw Input to `CurriculumFeed`](#end-to-end-from-raw-input-to-curriculumfeed)
|
|
15
|
+
- [Layer 1 - Detect](#layer-1---detect)
|
|
16
|
+
- [Layer 2 - Generate](#layer-2---generate-one-branch-by-learning-mode)
|
|
17
|
+
- [Layer 3 - Emit](#layer-3---emit-one-shape-for-every-graph)
|
|
18
|
+
- [Layer 4 - Validate](#layer-4---validate-fail-closed-at-the-source)
|
|
19
|
+
- [The three graph types](#the-three-graph-types-at-a-glance)
|
|
20
|
+
- [Depth and minute conventions](#depth-and-minute-conventions)
|
|
14
21
|
- [Installation](#installation)
|
|
15
22
|
- [Quick Start](#quick-start)
|
|
16
23
|
- [5. Emit Curriculum Feed](#5-emit-curriculum-feed-from-any-graph)
|
|
@@ -38,13 +45,455 @@
|
|
|
38
45
|
- **Auto-detect** input type (repo, syllabus, description) and domain category (software, math, science...) with multi-language support (EN, VI, ZH, JA, KO, FR)
|
|
39
46
|
- **Context-dependent domain detection** — `.cpp` files default to `software`; `hardware_iot` only when Arduino/ESP32/sensor context is present
|
|
40
47
|
- **Choose graph type** automatically: `project_graph` (product-driven) or `knowledge_graph` (concept-driven) or `hybrid_graph`
|
|
41
|
-
- **
|
|
48
|
+
- **Source scanning** of Swift, TypeScript/JavaScript, Python and C/C++ files (regex-based — the package has no parser dependency)
|
|
42
49
|
- **LLM-powered decomposition** of projects into features → steps → keywords
|
|
43
50
|
- **ULO/CIO/SIO depth classification** — each keyword/concept mapped to WHAT (ulo), HOW (cio), or IMPLEMENTATION (sio)
|
|
44
51
|
- **Concept resolution** with ULO/CIO-aware matching against Master Tree concept codes
|
|
45
52
|
- **Prerequisite cycle detection** — DFS cycle detection + Kahn's topological sort for knowledge graphs
|
|
46
53
|
- **Keyword-per-milestone tracking** with new vs prerequisite distinction
|
|
47
|
-
- **Time-budget-aware assembly** —
|
|
54
|
+
- **Time-budget-aware assembly** — milestones split against a 90-minute session, a 1.15 overhead factor and a 120-minute cap (fixed by the pipeline today; `assembleRoadmap()` takes them as parameters)
|
|
55
|
+
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
## End-to-End: From Raw Input to `CurriculumFeed`
|
|
59
|
+
|
|
60
|
+
`domain-kit` never writes a lesson, never schedules a session, never picks an activity.
|
|
61
|
+
Its entire job is to turn *something that already exists* — a repository, a syllabus, a
|
|
62
|
+
short description — into **one JSON object with a fixed shape** (`CurriculumFeed` v2)
|
|
63
|
+
that `curriculum-kit` can schedule without guessing.
|
|
64
|
+
|
|
65
|
+
Read this section as four layers that always run in this order:
|
|
66
|
+
|
|
67
|
+
| Layer | Who | Calls the LLM? | When it fails |
|
|
68
|
+
|---|---|---|---|
|
|
69
|
+
| 1. Detect | `detectDomainProfile()` | no | never throws — reports `confidence` / `ambiguous` |
|
|
70
|
+
| 2. Generate | `runProjectGraphPipeline` / `runKnowledgeGraphPipeline` / `runHybridGraphPipeline` | yes | `throw` (fail-closed) |
|
|
71
|
+
| 3. Emit | `emitCurriculumFeed()` | no | `throw` (fail-closed) |
|
|
72
|
+
| 4. Validate | `validateCurriculumFeed()` | no | `throw` (fail-closed) |
|
|
73
|
+
|
|
74
|
+
The rule that makes the package trustworthy: **layer 2 and 3 never hand back a
|
|
75
|
+
half-built graph.** A missing LLM key, an empty phase plan, a concept with no keywords,
|
|
76
|
+
a backwards prerequisite edge — all of them stop the run at the layer that produced
|
|
77
|
+
them, instead of surfacing as a broken lesson three packages later.
|
|
78
|
+
|
|
79
|
+
> **Shape is fail-closed; content is reported.** Structural violations throw at
|
|
80
|
+
> emission. Things that make a graph *pedagogically* suspect — a feature with no
|
|
81
|
+
> steps, a concept no phase introduces — are **diagnostics** (`source.diagnostics`,
|
|
82
|
+
> each with a `code` and a `severity`) rather than throws, so the host decides whether
|
|
83
|
+
> to block. See [ADR-0002](docs/adr/2026-10-05-ADR-0002-graph-integrity-provenance-and-diagnostics.md).
|
|
84
|
+
|
|
85
|
+
### The whole picture
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
+------------------------------------------------------------------------------+
|
|
89
|
+
| INPUT |
|
|
90
|
+
| repoDir / repoUrl - syllabusText / parsedSyllabus - description - goal |
|
|
91
|
+
| techStack - chapters - fileExtensions - gradeBand |
|
|
92
|
+
| forcedLearningMode (router override) - knowledgeTreePath - proposalsPath |
|
|
93
|
+
+------------------------------------+-----------------------------------------+
|
|
94
|
+
v
|
|
95
|
+
+------------------------------------------------------------------------------+
|
|
96
|
+
| LAYER 1 - detectDomainProfile() - pure, deterministic, no LLM |
|
|
97
|
+
| detectInputType() -> detectDomain() -> detectLearningMode() |
|
|
98
|
+
| => DomainProfile { inputType, domainCategory, learningMode, graphType, |
|
|
99
|
+
| confidence, evidence, ambiguous? } |
|
|
100
|
+
+------------------------------------+-----------------------------------------+
|
|
101
|
+
| effectiveMode = forcedLearningMode
|
|
102
|
+
| ?? domainProfile.learningMode
|
|
103
|
+
| (graphType is only its echo)
|
|
104
|
+
+---------------------+---------------------+
|
|
105
|
+
v v v
|
|
106
|
+
'product' 'concept' 'hybrid'
|
|
107
|
+
| | |
|
|
108
|
+
| | repoDir present? --no--> FALLBACK to the
|
|
109
|
+
| | | concept branch,
|
|
110
|
+
| | | loud if auto (R1)
|
|
111
|
+
v v v
|
|
112
|
+
runProjectGraph runKnowledgeGraph runProjectGraphPipeline + runKnowledgeGraphPipeline
|
|
113
|
+
Pipeline Pipeline (two full graph runs), then:
|
|
114
|
+
STEP_1 scan/ (parseSyllabus generateHybridGraph()
|
|
115
|
+
select/AST first if raw STEP_1 feature-concept links
|
|
116
|
+
STEP_2 keyword syllabusText) STEP_2 decomposePhases
|
|
117
|
+
extract generateKnowledge (auditPhasePlan fail-closed)
|
|
118
|
+
STEP_3 SDK API Graph: STEP_3 totals
|
|
119
|
+
index STEP_1 decompose
|
|
120
|
+
STEP_4 LLM STEP_2 validate & clean
|
|
121
|
+
C0/C1/C2 STEP_2_5 standardize vs
|
|
122
|
+
STEP_5 verify mlo-knowlege-tree.tsv <== knowledgeTreePath enters
|
|
123
|
+
vs source here, fail-closed (R5, R6)
|
|
124
|
+
STEP_6 escalate STEP_3 detect + break cycles (R4)
|
|
125
|
+
keywords STEP_4 topological sort
|
|
126
|
+
STEP_7 resolve to STEP_5 ULO/CIO/SIO authoring
|
|
127
|
+
Master Tree STEP_5b SIO keyword enrichment
|
|
128
|
+
concepts STEP_5c minute calibration
|
|
129
|
+
STEP_8 assemble STEP_6 verify + coverage audit
|
|
130
|
+
roadmap (feed is
|
|
131
|
+
emitted by the
|
|
132
|
+
ROUTER, not this
|
|
133
|
+
pipeline - R3)
|
|
134
|
+
| | |
|
|
135
|
+
+---------------------+---------------------+
|
|
136
|
+
v
|
|
137
|
+
+------------------------------------------------------------------------------+
|
|
138
|
+
| LAYER 3 - emitCurriculumFeed(graph) - pure projection, no LLM |
|
|
139
|
+
| Called from THREE places (R3): router for the project branch |
|
|
140
|
+
| (domainRouter.ts:171), knowledge orchestrator, hybrid orchestrator. |
|
|
141
|
+
| learning_nodes[] - dependency_edges[] - suggested_groupings[] - phases[] |
|
|
142
|
+
| (phases[] is filled ONLY by the hybrid emitter - R1) |
|
|
143
|
+
+------------------------------------+-----------------------------------------+
|
|
144
|
+
v
|
|
145
|
+
+------------------------------------------------------------------------------+
|
|
146
|
+
| LAYER 4 - validateCurriculumFeed(feed) - structural + teaching order |
|
|
147
|
+
| Entry dispatches by graph shape; anything that already looks like a |
|
|
148
|
+
| feed is pass-through validated (R7) |
|
|
149
|
+
+------------------------------------------------------------------------------+
|
|
150
|
+
v
|
|
151
|
+
CurriculumFeed (schema_version: 2)
|
|
152
|
+
||
|
|
153
|
+
|| ====== package boundary ======
|
|
154
|
+
|| nothing below this line is
|
|
155
|
+
|| domain-kit's business
|
|
156
|
+
v
|
|
157
|
+
curriculum-kit -> units -> sessions -> lessons -> activities
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Risk tags (R1-R9) are defined in the Risk map below.
|
|
161
|
+
|
|
162
|
+
Two facts about that diagram are easy to miss and worth stating up front:
|
|
163
|
+
|
|
164
|
+
1. **Only the hybrid path fills `phases[]`.** `emitFromProject` and `emitFromKnowledge`
|
|
165
|
+
both return `phases: []`. "The order in the file is the order in the classroom"
|
|
166
|
+
is therefore a guarantee that only hybrid consumers can rely on — a project-only or
|
|
167
|
+
knowledge-only feed has no phase authority at all.
|
|
168
|
+
2. **`runProjectGraphPipeline` stops at STEP_8.** Feed emission for the product branch is
|
|
169
|
+
done by `runDomainPipeline()` (static import of the emitter - the old dynamic
|
|
170
|
+
`import()` was removed as pure inconsistency), not by the project pipeline itself.
|
|
171
|
+
The knowledge and hybrid branches emit from their own thin orchestrators in
|
|
172
|
+
`src/pipeline/`.
|
|
173
|
+
|
|
174
|
+
3. **Hybrid degrades loudly; explicit requests fail closed.** An AUTO-detected
|
|
175
|
+
`hybrid` without a `repoDir` runs the concept branch instead and returns a
|
|
176
|
+
structured `modeDowngrade: { from, to, reason }` on the result (R1). A
|
|
177
|
+
FORCED `forcedLearningMode: 'hybrid'` without `repoDir` throws. Both hybrid
|
|
178
|
+
knowledge legs pass `knowledgeTreePath`/`proposalsPath` through, so STEP_2_5
|
|
179
|
+
tree standardization runs in every mode (R2). Both fixed 2026-10-07.
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
### Risk map (2026-10-07 - re-verified against code)
|
|
183
|
+
|
|
184
|
+
Written while re-drawing the diagram above; every row cites the code it was
|
|
185
|
+
verified against. Full postmortems: deep review 2026-10-05 (app repo,
|
|
186
|
+
`docs/analysis/2026-10-05-graph-curriculum-deep-review.md`) and
|
|
187
|
+
[ADR-0002](docs/adr/2026-10-05-ADR-0002-graph-integrity-provenance-and-diagnostics.md).
|
|
188
|
+
|
|
189
|
+
| # | Risk | Where | Severity |
|
|
190
|
+
|---|---|---|---|
|
|
191
|
+
| R1 | **FIXED 2026-10-07.** Was: `hybrid` without `repoDir` silently degraded to a knowledge-graph feed (only a `warnings[]` entry). Now a FORCED hybrid throws (fail-closed); an auto-detected downgrade returns a structured `modeDowngrade { from, to, reason }`. | `src/pipeline/domainRouter.ts` | High |
|
|
192
|
+
| R2 | **FIXED 2026-10-07.** Was: the hybrid branch's knowledge legs dropped `knowledgeTreePath`/`proposalsPath`, so STEP_2_5 ran in concept mode but not hybrid mode for the same subject. Now both hybrid legs pass them through. | `src/pipeline/domainRouter.ts` | High |
|
|
193
|
+
| R3 | Feed emission is invoked from three different places, and emission options already diverge (`hallucinationCount` is only passed by the router). Drift risk with every emitter change. | `domainRouter.ts:171`, `pipeline/knowledgeGraphPipeline.ts:37`, `pipeline/hybridGraphPipeline.ts:22` | Medium |
|
|
194
|
+
| R4 | Cycle breaking mutates `prerequisites` in place ("weakest edge" heuristic) and only warns. Which edge is dropped decides teaching order - a pedagogy-relevant decision hidden in a warning string. | `src/graph/knowledgeGraphPipeline.ts:383` | Medium |
|
|
195
|
+
| R5 | STEP_2_5 only runs when `knowledgeTreePath` is provided. Without it, every concept stays `proposed_new` and nothing in the output flags the absence of tree anchoring. | `src/graph/knowledgeGraphPipeline.ts:322` | Medium |
|
|
196
|
+
| R6 | Tree prerequisite edges are seeded only when the referenced tree concept is part of THIS analysis - partial graphs silently lose cross-analysis prerequisite edges. Matters for gold-set coverage measurement across runs. | `src/graph/knowledgeGraphPipeline.ts:363-375` | Medium |
|
|
197
|
+
| R7 | `emitCurriculumFeed` dispatches by shape (key presence) and pass-through-validates anything that "looks like a feed" (CF-8 in the deep review). | `src/feed/curriculumFeedEmitter.ts:833-839` | Low |
|
|
198
|
+
| R8 | Non-fatal pedagogy issues funnel into a single lossy `warnings[]` string channel; some drops only hit `console.warn` (e.g. a concept falling out of `learning_path`, KG-3). Structured diagnostics per ADR-0002 should replace string warnings. | deep review KG-3 | Medium |
|
|
199
|
+
| R9 | **FIXED 2026-10-07.** Was: ASCII-only regex assumptions broke on Vietnamese text (KG-1) - empty derived keywords threw, registries filled with fragments. Now a shared `stripDiacritics()` (NFD + combining-mark strip + `đ`→`d`, incl. the precomposed-base-letter case NFD cannot decompose) backs every tokenizer; regression tests in `src/__tests__/vietnamese-tokenization.test.ts`. | `src/utils/diacritics.ts` | Critical |
|
|
200
|
+
|
|
201
|
+
### Layer 1 - Detect
|
|
202
|
+
|
|
203
|
+
`detectDomainProfile()` runs three detectors in sequence and never calls an LLM:
|
|
204
|
+
|
|
205
|
+
- **Input type** — `repository | syllabus | description | files | mixed`
|
|
206
|
+
- **Domain category** — multi-signal scoring over 340+ keywords in 10 categories,
|
|
207
|
+
with Vietnamese / English / Chinese / Japanese / French keyword lists and
|
|
208
|
+
context-dependent extension mapping (`.cpp` is ambiguous: software by default,
|
|
209
|
+
`hardware_iot` only with Arduino/ESP32/sensor context).
|
|
210
|
+
- **Learning mode** — `product | concept | hybrid`, the signal that actually picks
|
|
211
|
+
the branch in layer 2.
|
|
212
|
+
|
|
213
|
+
`graphType` on the profile is a convenience mirror of `learningMode`
|
|
214
|
+
(`product→project_graph`, `concept→knowledge_graph`, `hybrid→hybrid_graph`); when the
|
|
215
|
+
two would disagree, `forcedLearningMode` overrides the mode and `graphType` is
|
|
216
|
+
recomputed at the branch.
|
|
217
|
+
|
|
218
|
+
### Layer 2 - Generate (one branch by learning mode)
|
|
219
|
+
|
|
220
|
+
#### Branch A — `project_graph` (product-driven)
|
|
221
|
+
|
|
222
|
+
```
|
|
223
|
+
STEP_1 scan the file TREE, names only (no content)
|
|
224
|
+
└─▶ LLM picks the relevant files ─▶ read content for those files only
|
|
225
|
+
└─▶ parse .swift / .ts .tsx / .js .jsx / .py / .ino .cpp .c .h .hpp
|
|
226
|
+
└─▶ merge results by PRIMARY language
|
|
227
|
+
STEP_2 extractKeywords() → weighted keywords (app / esp32 tags)
|
|
228
|
+
STEP_3 SDK API index, derived from STEP_2 keywords
|
|
229
|
+
STEP_4_C0 scaffold → feature F0 "FOUNDATION & SETUP" (LLM failure ⇒ F0 is skipped
|
|
230
|
+
silently and the pipeline continues without it)
|
|
231
|
+
STEP_4_C1 overview → product{ goals, users, journeys, development_stages } + features
|
|
232
|
+
STEP_4_C2 steps → per-feature steps, one batched LLM call, falling back to
|
|
233
|
+
per-feature; a feature that still fails is logged and left
|
|
234
|
+
step-less (no throw)
|
|
235
|
+
STEP_5 verifyProjectGraph() → strip files / APIs / keywords not found in code;
|
|
236
|
+
F0 keywords are exempt (pedagogical terms)
|
|
237
|
+
STEP_6 escalateAndMapConcepts() → keyword → technology-neutral concept +
|
|
238
|
+
depth (ulo | cio | sio) + evidence files.
|
|
239
|
+
Depth comes from the LLM; on LLM failure a
|
|
240
|
+
regex fallback reads the usage context
|
|
241
|
+
(import → sio, code body → sio, comment → ulo)
|
|
242
|
+
STEP_7 resolveConcepts() → concept → Master-Tree code, scored ULO/CIO-aware:
|
|
243
|
+
sio keyword → concept.keywords
|
|
244
|
+
cio keyword → concept.description + concept.cio
|
|
245
|
+
ulo keyword → concept.ulo + concept.description
|
|
246
|
+
STEP_8 assembleRoadmap() → milestones grouped into phases, each milestone tagged
|
|
247
|
+
with all / new / prerequisite keywords + a time budget
|
|
248
|
+
───────────────────────── package boundary ───────────────────────────────
|
|
249
|
+
emitCurriculumFeed(projectGraph) → CurriculumFeed (phases: [])
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
#### Branch B — `knowledge_graph` (concept-driven)
|
|
253
|
+
|
|
254
|
+
```
|
|
255
|
+
STEP_1 parseSyllabus(syllabusText) → structured units/topics
|
|
256
|
+
LLM decomposes the subject → concepts + prerequisites + problem_types
|
|
257
|
+
+ categories + techniques. The syllabus is truncated at
|
|
258
|
+
KG_MAX_SYLLABUS_CHARS (default 16000) with a warning.
|
|
259
|
+
|
|
260
|
+
STEP_2 validate & clean
|
|
261
|
+
· description shorter than 50 chars → warning
|
|
262
|
+
· KEYWORD REGISTRY IS FAIL-CLOSED: a concept with no keywords gets them
|
|
263
|
+
derived from its name/description; if nothing is derivable → throw
|
|
264
|
+
· a prerequisite pointing at a non-existent concept → edge removed + warning
|
|
265
|
+
|
|
266
|
+
STEP_2_5 only when knowledgeTreePath is supplied: standardize every concept against
|
|
267
|
+
mlo-knowlege-tree.tsv → `standardized` (adopts the tree code + the keywords
|
|
268
|
+
the standardizer chose for this tech stack) or `proposed_new`.
|
|
269
|
+
A concept with no decision → throw. Tree prerequisite codes then seed
|
|
270
|
+
internal edges where both ends are in this analysis.
|
|
271
|
+
|
|
272
|
+
STEP_3 detectPrerequisiteCycles() → breakPrerequisiteCycles() at the weakest edge
|
|
273
|
+
STEP_4 topologicalSort() (Kahn) → learning_path is REBUILT from the sorted order
|
|
274
|
+
STEP_5 hierarchical ULO/CIO/SIO authoring, then deterministic ULO/CIO/SIO codes
|
|
275
|
+
(a tree-standardized code from STEP_2_5 always wins over a generated one)
|
|
276
|
+
STEP_5b mine the SIO text for technology identifiers — backticked tokens,
|
|
277
|
+
@Annotations, a.b() calls, .modifiers, camelCase — and merge them into the
|
|
278
|
+
keyword registry (capped at 8 per concept)
|
|
279
|
+
STEP_5c minute calibration:
|
|
280
|
+
score = 0.2·sio + 0.15·problem_types + 0.15·prerequisites
|
|
281
|
+
+ 0.2·(has Analyze/Evaluate/Create)
|
|
282
|
+
minutes = 15 + 45·score ⇒ 15 … 60 minutes
|
|
283
|
+
The calibrated value overrides the LLM estimate only when the LLM returned
|
|
284
|
+
its default 30, or when the two diverge by more than 50%.
|
|
285
|
+
|
|
286
|
+
STEP_6 verifyKnowledgeGraph() → referential integrity + REVERSE COVERAGE audit:
|
|
287
|
+
every syllabus topic must be covered by at least one concept.
|
|
288
|
+
Findings are reported as warnings, not thrown.
|
|
289
|
+
───────────────────────── package boundary ───────────────────────────────
|
|
290
|
+
emitCurriculumFeed(knowledgeGraph) → CurriculumFeed (phases: [])
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
#### Branch C — `hybrid_graph` (both, and how they interlock)
|
|
294
|
+
|
|
295
|
+
Needs a `project_graph` **and** a `knowledge_graph` — it is the only branch that runs two
|
|
296
|
+
generators before it starts.
|
|
297
|
+
|
|
298
|
+
```
|
|
299
|
+
input: project_graph + knowledge_graph
|
|
300
|
+
|
|
301
|
+
STEP_1 LLM emits feature ↔ concept links
|
|
302
|
+
{ feature_id, concept_id, relationship: requires|demonstrates|applies|extends,
|
|
303
|
+
depth: ulo|cio|sio, estimated_prereq_minutes }
|
|
304
|
+
A link naming an unknown feature or concept is DROPPED with a warning.
|
|
305
|
+
|
|
306
|
+
STEP_2 decomposePhases() — the LLM turns the build into PROGRESSIVE COMPLETION
|
|
307
|
+
PHASES: each phase states what the product can demonstrably do once it is
|
|
308
|
+
done, owns every feature built in it, and declares which concepts it teaches
|
|
309
|
+
at its start (aspect = new | advanced | preview).
|
|
310
|
+
auditPhasePlan() then checks the contract deterministically:
|
|
311
|
+
· every feature in EXACTLY one phase, none skipped
|
|
312
|
+
· every phase has a non-empty product_completion
|
|
313
|
+
· a concept is introduced no later than the phase that consumes it
|
|
314
|
+
· no concept before its own prerequisites — checked per PHASE, not per array
|
|
315
|
+
position: teaching a concept and its prerequisite in the SAME phase is valid
|
|
316
|
+
whichever order the LLM emitted them, and flagging it would burn all three
|
|
317
|
+
repair attempts and then throw away a good plan
|
|
318
|
+
· a preview is satisfied by an earlier anchor or an earlier preview, and every
|
|
319
|
+
preview has an anchor teach somewhere
|
|
320
|
+
· preview must precede its anchor and precede nothing already introduced
|
|
321
|
+
Violations are fed back verbatim; at most 3 attempts, then throw. A phase
|
|
322
|
+
model that cannot be trusted is worse than no output.
|
|
323
|
+
|
|
324
|
+
STEP_3 totals
|
|
325
|
+
total_project_minutes = Σ step.effort.estimated_minutes (fallback 15)
|
|
326
|
+
total_concept_minutes = Σ concept.estimated_minutes (fallback 30)
|
|
327
|
+
linked_prereq_minutes = total_project_minutes + Σ max-per-concept prereq
|
|
328
|
+
minutes, counting only concepts that HAVE a feature
|
|
329
|
+
link. It is NOT the course total.
|
|
330
|
+
───────────────────────── package boundary ───────────────────────────────
|
|
331
|
+
emitCurriculumFeed(hybridGraph) → CurriculumFeed — the only path that
|
|
332
|
+
fills phases[], and therefore the only path that defines teaching order.
|
|
333
|
+
```
|
|
334
|
+
|
|
335
|
+
### Layer 3 - Emit (one shape for every graph)
|
|
336
|
+
|
|
337
|
+
```
|
|
338
|
+
emitCurriculumFeed(graph, opts?)
|
|
339
|
+
├─ graph already has learning_nodes[] → validate and pass through unchanged
|
|
340
|
+
├─ has project_graph | knowledge_graph | links[] | phases[]
|
|
341
|
+
│ → emitFromHybrid()
|
|
342
|
+
├─ has concepts[] or type==='knowledge_graph'
|
|
343
|
+
│ → emitFromKnowledge()
|
|
344
|
+
└─ otherwise → emitFromProject()
|
|
345
|
+
```
|
|
346
|
+
|
|
347
|
+
**`emitFromHybrid` — phase-driven emission.** For each phase, in order, it emits
|
|
348
|
+
`new` concepts first, then `advanced` revisits, then `previews`, and only then the
|
|
349
|
+
phase's features and their steps (project-graph order):
|
|
350
|
+
|
|
351
|
+
| Aspect | Node id | Minutes | Carries |
|
|
352
|
+
|---|---|---|---|
|
|
353
|
+
| `new` | `<cid>` | the concept's own estimate | `keywords.new` — the legitimate vocabulary introducer |
|
|
354
|
+
| `advanced` | `<cid>__ADV<phase>` | `max(15, 50% of anchor)` | a `sio → cio` downgrade candidate saving 30% **of the revisit's own minutes** |
|
|
355
|
+
| `preview` | `<cid>__PREV<phase>` | `max(5, 25% of anchor)` | `keywords.all` — may introduce the vocabulary early, with **no** prerequisite edges and no scaffolding |
|
|
356
|
+
|
|
357
|
+
**A preview in an EARLIER phase satisfies a downstream concept's prerequisite** — that is
|
|
358
|
+
what it is for. A preview in the *same* phase does not, because within a phase the order
|
|
359
|
+
is `new → advanced → preview`, so it would arrive after the dependent concept.
|
|
360
|
+
|
|
361
|
+
### Knowledge edges
|
|
362
|
+
|
|
363
|
+
Edges are materialized as *concept → first step of the consumer feature* (`kind: 'knowledge'`);
|
|
364
|
+
feature build order and `depends_on` become `kind: 'task'`. When a prerequisite is satisfied
|
|
365
|
+
only by a preview, the edge points at the **preview node** — the vocabulary genuinely comes
|
|
366
|
+
from there.
|
|
367
|
+
|
|
368
|
+
Keyword identity is case- and Unicode-insensitive: `SwiftUI` / `swiftui` and an NFD/NFC
|
|
369
|
+
Vietnamese pair are the same keyword. The ledger and the edge materializer share one
|
|
370
|
+
comparison key, so they cannot disagree about what counts as a first appearance.
|
|
371
|
+
|
|
372
|
+
Edges are materialized as *concept → first step of the consumer feature* (`kind: 'knowledge'`);
|
|
373
|
+
feature build order and `depends_on` become `kind: 'task'`.
|
|
374
|
+
|
|
375
|
+
Where the graph contradicts itself, the emitter **reports instead of inventing**:
|
|
376
|
+
|
|
377
|
+
- a link whose anchor teach sits in a LATER phase than its consumer → `hybrid_link_drift`, no edge;
|
|
378
|
+
- a product step that would become the introducer of a concept's vocabulary → `concept_overtaken_by_step`, no edge;
|
|
379
|
+
- a keyword introducer that sits in a later phase than its consumer → `keyword_order_drift`, no edge.
|
|
380
|
+
|
|
381
|
+
That last class of edge is what once produced the 2026-09-17 "Teaching-order violation"
|
|
382
|
+
incident: the contradiction used to escape emission and explode inside the planner.
|
|
383
|
+
|
|
384
|
+
#### Coverage holes that used to be silent
|
|
385
|
+
|
|
386
|
+
| Diagnostic | Severity | Meaning |
|
|
387
|
+
|---|---|---|
|
|
388
|
+
| `orphan_concept` | `error` | the knowledge graph holds a concept that no phase introduces and no link consumes — it will not be taught |
|
|
389
|
+
| `feature_without_steps` | `warning` | the project graph declares a feature whose steps never materialised (STEP_4_C2 failed for it) — it will not appear in the feed |
|
|
390
|
+
| `step_without_id` | `warning` | a step with no `id` cannot become a node and was dropped (reported as `feature#index`) |
|
|
391
|
+
| `external_prerequisite` | `info` | a concept depends on knowledge outside this analysis — recorded on the node as `external_prerequisites`, not an error |
|
|
392
|
+
| `upstream_warning` | `warning` | a warning string handed in by the graph pipeline, kept for backward compatibility |
|
|
393
|
+
|
|
394
|
+
Both of the first two used to vanish with an empty `warnings: []`. The host can gate on
|
|
395
|
+
`code`/`severity` instead of substring-matching free text.
|
|
396
|
+
|
|
397
|
+
### Entry requirements
|
|
398
|
+
|
|
399
|
+
A concept can depend on knowledge the analysis does not cover — a course on vector
|
|
400
|
+
calculus legitimately requires linear algebra. Those ids are **not** nodes here, so they
|
|
401
|
+
cannot be edges; dropping them silently would throw away exactly what someone deciding
|
|
402
|
+
whether to enrol needs most.
|
|
403
|
+
|
|
404
|
+
They land on the node instead:
|
|
405
|
+
|
|
406
|
+
```json
|
|
407
|
+
{ "id": "C7", "kind": "concept", "external_prerequisites": ["C_LINALG"] }
|
|
408
|
+
```
|
|
409
|
+
|
|
410
|
+
and surface once as an `external_prerequisite` **info** diagnostic. Info, not warning —
|
|
411
|
+
a course with entry requirements is normal, and a host that gates on severity should not
|
|
412
|
+
block on it. The three emission paths used to disagree here: the knowledge path dropped
|
|
413
|
+
them silently while **both** hybrid paths hard-failed (one built a dangling edge, the other
|
|
414
|
+
threw an ordering error blaming a concept that was not in the analysis), so any graph
|
|
415
|
+
referencing prior knowledge was un-emittable.
|
|
416
|
+
|
|
417
|
+
### Provenance
|
|
418
|
+
|
|
419
|
+
`source.provenance` records which providers actually served the run:
|
|
420
|
+
|
|
421
|
+
```typescript
|
|
422
|
+
provenance: {
|
|
423
|
+
providers: ['openrouter', 'nvidia'], // failover is visible, not silent
|
|
424
|
+
models: ['preset-x', 'nemotron-3-ultra'],
|
|
425
|
+
calls: 7,
|
|
426
|
+
failed_calls: 1, // openrouter fell through
|
|
427
|
+
degraded: true,
|
|
428
|
+
}
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
Without it, a graph produced half by a paid preset and half by a free tier is
|
|
432
|
+
indistinguishable from a clean one — and not reproducible or auditable. The client is
|
|
433
|
+
created **once per pipeline run** and threaded through every step, so all calls land in
|
|
434
|
+
one log; it is per-instance, so concurrent runs never inherit each other's history.
|
|
435
|
+
|
|
436
|
+
### Layer 4 - Validate (fail-closed, at the source)
|
|
437
|
+
|
|
438
|
+
`validateCurriculumFeed()` runs on every emission path and rejects:
|
|
439
|
+
|
|
440
|
+
- duplicate node ids;
|
|
441
|
+
- an edge pointing at an unknown node, or a self-loop;
|
|
442
|
+
- a grouping or phase referencing an unknown node;
|
|
443
|
+
- duplicate phase ids;
|
|
444
|
+
- **duplicate or non-ascending phase `order`** — `order` *is* the teaching sequence once
|
|
445
|
+
phases exist, so two phases claiming one slot makes it undefined;
|
|
446
|
+
- an empty feed (`learning_nodes` has a minimum of 1);
|
|
447
|
+
- **the teaching-order invariant** — when `phases[]` is non-empty, a prerequisite must
|
|
448
|
+
live in the same phase or an *earlier* one than its consumer.
|
|
449
|
+
|
|
450
|
+
`emitFromProject` adds one more: a graph with `features[]` but no `features[].steps` is
|
|
451
|
+
rejected by name as a legacy knowledge-tree graph, with the regeneration command in the
|
|
452
|
+
message.
|
|
453
|
+
|
|
454
|
+
### The three graph types at a glance
|
|
455
|
+
|
|
456
|
+
| | `project_graph` | `knowledge_graph` | `hybrid_graph` |
|
|
457
|
+
|---|---|---|---|
|
|
458
|
+
| Answers | "what is being built?" | "what must be understood?" | "both, and how they interlock" |
|
|
459
|
+
| LLM sees | file tree + selected file contents | syllabus / description | both graphs, rendered as text |
|
|
460
|
+
| Atomic node | `features[].steps[]` | `concepts[]` | both |
|
|
461
|
+
| Ordering authority | array order of features/steps | `learning_path` (topological) | `phases[]` → `feed.phases` |
|
|
462
|
+
| Time source | `step.effort.estimated_minutes` | `concept.estimated_minutes` (calibrated 15–60) | sum of both |
|
|
463
|
+
| Prerequisite evidence | none — build order only | `concept.prerequisites[]`, proven acyclic | both, cross-checked against concept prerequisites |
|
|
464
|
+
| `feed.phases` | always `[]` | always `[]` | filled ⇒ teaching order |
|
|
465
|
+
| Entry point | `runProjectGraphPipeline()` | `runKnowledgeGraphPipeline()` | `runHybridGraphPipeline()` |
|
|
466
|
+
|
|
467
|
+
### Depth and minute conventions
|
|
468
|
+
|
|
469
|
+
Exported from code as `DEPTH_MINUTE_FRACTIONS` and `calibrateConceptMinutes()` — read
|
|
470
|
+
those rather than copying numbers out of this table.
|
|
471
|
+
|
|
472
|
+
| Convention | Value | Source |
|
|
473
|
+
|---|---|---|
|
|
474
|
+
| preview node minutes | `max(5, 25%)` of the anchor | `DEPTH_MINUTE_FRACTIONS` |
|
|
475
|
+
| ULO / CIO / SIO minutes for a concept node | `max(5, 35%)` / `max(ulo+5, 65%)` / `max(cio+5, 100%)` | `DEPTH_MINUTE_FRACTIONS` |
|
|
476
|
+
| advanced revisit minutes | `max(15, 50%)` of the anchor | `DEPTH_MINUTE_FRACTIONS` |
|
|
477
|
+
| advanced revisit downgrade candidate | `sio → cio`, saves 30% **of the revisit's own minutes** | `DEPTH_MINUTE_FRACTIONS` |
|
|
478
|
+
| step with no `estimated_minutes` | 15 minutes (same constant the hybrid totals use) | `DEFAULT_STEP_MINUTES` |
|
|
479
|
+
| concept minutes after calibration | 15–60 minutes | `calibrateConceptMinutes()` |
|
|
480
|
+
| roadmap session / overhead / milestone split cap | 90 min / ×1.15 / 120 min — hardcoded by the pipeline, not configurable through it | `STEP_8` in `pipeline/projectGraphPipeline.ts` |
|
|
481
|
+
| phase minute balance demanded of the LLM | 120–240 min per phase; rebalance below 60 or above 400 | `STEP_2` in `graph/hybridGraphPipeline.ts` |
|
|
482
|
+
| file budget per run | 70 files / 500 000 chars | `utils/fileUtils.ts` |
|
|
483
|
+
|
|
484
|
+
> The three depth levels are a **single ladder applied everywhere**:
|
|
485
|
+
> `ulo` = WHAT + WHY (technology-agnostic), `cio` = HOW the mechanism works
|
|
486
|
+
> (still no concrete API), `sio` = IMPLEMENTATION (must name a real API, function,
|
|
487
|
+
> module or file). A keyword can only ever move *down* this ladder, never across it.
|
|
488
|
+
>
|
|
489
|
+
> The percentages above live in one exported table, `DEPTH_MINUTE_FRACTIONS`. They were
|
|
490
|
+
> previously hardcoded in three places (25% / 35% / 30%), which meant "what does a ULO
|
|
491
|
+
> pass cost?" had three answers. Read that table rather than hardcoding a fourth.
|
|
492
|
+
|
|
493
|
+
> **Unicode:** Vietnamese input reaches the detector in NFC or NFD form. Text is
|
|
494
|
+
> normalized at the boundary and also matched with diacritics stripped, so the same
|
|
495
|
+
> sentence classifies identically whichever form it arrives in — before this, one
|
|
496
|
+
> sentence measured as `math` (confidence 1.0) in NFC and `other` (confidence 0.2) in NFD.
|
|
48
497
|
|
|
49
498
|
---
|
|
50
499
|
|
|
@@ -74,9 +523,34 @@ npm link @thanh01.pmt/domain-kit
|
|
|
74
523
|
LLM_API_KEY=sk-... # OpenAI-compatible API key
|
|
75
524
|
LLM_BASE_URL=https://api.openai.com/v1 # API base URL
|
|
76
525
|
LLM_MODEL=gpt-4o-mini # Model to use
|
|
77
|
-
LLM_MAX_TOKENS=
|
|
526
|
+
LLM_MAX_TOKENS=65536 # Max output tokens (default)
|
|
527
|
+
LLM_MIN_REQUEST_INTERVAL_MS=2500 # process-wide pacing; 0 disables
|
|
528
|
+
|
|
529
|
+
# Extra knobs the pipelines read directly:
|
|
530
|
+
KG_MAX_SYLLABUS_CHARS=16000 # syllabus truncation before decomposition
|
|
531
|
+
KG_MAX_TOKENS=65536 # knowledge-graph generation budget
|
|
532
|
+
LLM_REQUEST_TIMEOUT_MS=300000 # per-request cap
|
|
78
533
|
```
|
|
79
534
|
|
|
535
|
+
**Provider chain.** `createLlmClient()` builds an ordered chain and fails over through
|
|
536
|
+
it on transient errors (408/429/500/502/503/504):
|
|
537
|
+
|
|
538
|
+
1. explicit `llmConfig.apiKey`
|
|
539
|
+
2. `LLM_*` (or `OPENAI_*`) — generic OpenAI-compatible
|
|
540
|
+
3. `DASHSCOPE_*` / `ALIBABA_*`
|
|
541
|
+
4. `NVIDIA_API_KEY` → NVIDIA NIM
|
|
542
|
+
5. `OPENROUTER_API_KEY` → OpenRouter
|
|
543
|
+
|
|
544
|
+
If `OPENROUTER_MODEL` is set, OpenRouter is moved to the front. Requests are paced
|
|
545
|
+
process-wide so free tiers are not rate-limited by a burst; the interval is configurable
|
|
546
|
+
via `LLM_MIN_REQUEST_INTERVAL_MS` or `llmConfig.minRequestIntervalMs` (set `0` when the
|
|
547
|
+
caller owns the rate limit — several concurrent pipelines in one process, or serverless).
|
|
548
|
+
|
|
549
|
+
Each client keeps a `callLog` of every provider attempt it made. One client is created
|
|
550
|
+
per pipeline run and threaded through every step, so `PipelineResult.provenance` and
|
|
551
|
+
`CurriculumFeed.source.provenance` reflect that whole run — and because the log is
|
|
552
|
+
per-instance, concurrent runs never inherit each other's history.
|
|
553
|
+
|
|
80
554
|
---
|
|
81
555
|
|
|
82
556
|
## Quick Start
|
|
@@ -138,7 +612,7 @@ const result = await generateHybridGraph({
|
|
|
138
612
|
knowledgeGraph: existingKnowledgeGraph,
|
|
139
613
|
});
|
|
140
614
|
|
|
141
|
-
console.log(result.hybridGraph); // HybridGraph with links[]
|
|
615
|
+
console.log(result.hybridGraph); // HybridGraph with links[] + phases[]
|
|
142
616
|
```
|
|
143
617
|
|
|
144
618
|
### 5. Emit Curriculum Feed (from any graph)
|
|
@@ -146,7 +620,11 @@ console.log(result.hybridGraph); // HybridGraph with links[]
|
|
|
146
620
|
```typescript
|
|
147
621
|
import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
|
|
148
622
|
|
|
149
|
-
|
|
623
|
+
// `graph` is whichever graph object you just produced — note that step 4 above
|
|
624
|
+
// rebinds `result` to the HYBRID result, so name the graph explicitly:
|
|
625
|
+
const feed = emitCurriculumFeed(projectGraph); // project_graph
|
|
626
|
+
// const feed = emitCurriculumFeed(knowledgeGraph); // knowledge_graph
|
|
627
|
+
// const feed = emitCurriculumFeed(hybridGraph); // hybrid_graph (phases filled)
|
|
150
628
|
|
|
151
629
|
console.log(feed.learning_nodes.length); // unified nodes ready for planner
|
|
152
630
|
console.log(feed.dependency_edges.length); // knowledge + task edges
|
|
@@ -159,48 +637,22 @@ console.log(feed.phases.length); // non-empty ⇒ learning_nodes are
|
|
|
159
637
|
|
|
160
638
|
## Architecture
|
|
161
639
|
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
│ │ extractors/ │ │ graph/ │ │ concepts/ │ │
|
|
179
|
-
│ │ │ │ │ │ │ │
|
|
180
|
-
│ │ keywords │ │ scaffold │ │ concept │ │
|
|
181
|
-
│ │ │ │ overview │ │ resolver │ │
|
|
182
|
-
│ │ │ │ steps │ │ │ │
|
|
183
|
-
│ │ │ │ verify │ │ │ │
|
|
184
|
-
│ │ │ │ escalate │ │ │ │
|
|
185
|
-
│ │ │ │ knowledgeKG │ │ │ │
|
|
186
|
-
│ │ │ │ hybridKG │ │ │ │
|
|
187
|
-
│ └──────────────┘ └──────────────┘ └──────────────┘ │
|
|
188
|
-
│ │
|
|
189
|
-
│ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ │
|
|
190
|
-
│ │ assembly/ │ │ pipeline/ │ │ feed/ │ │
|
|
191
|
-
│ │ │ │ │ │ │ │
|
|
192
|
-
│ │ roadmap │ │ projectGraph │ │ emitCurricu..│ │
|
|
193
|
-
│ │ assembler │ │ pipeline │ │ lumFeed │ │
|
|
194
|
-
│ └──────────────┘ └──────────────┘ └──────────────┘ │
|
|
195
|
-
│ │
|
|
196
|
-
│ ┌──────────────┐ │
|
|
197
|
-
│ │ utils/ │ │
|
|
198
|
-
│ │ llmClient │ │
|
|
199
|
-
│ │ fileUtils │ │
|
|
200
|
-
│ └──────────────┘ │
|
|
201
|
-
│ │
|
|
202
|
-
└─────────────────────────────────────────────────────────────┘
|
|
203
|
-
```
|
|
640
|
+
| Directory | Files | Owns |
|
|
641
|
+
|---|---|---|
|
|
642
|
+
| `schemas/` | `projectGraphSchema`, `knowledgeGraphSchema`, `hybridGraphSchema`, `domainProfileSchema`, `curriculumFeedSchema`, `domainExtensions` | every Zod schema; `validateCurriculumFeed()` lives in `curriculumFeedSchema` |
|
|
643
|
+
| `detector/` | `inputTypeDetector`, `domainDetector`, `learningModeDetector`, `domainProfileDetector` | layer 1 — pure, no LLM |
|
|
644
|
+
| `parsers/` | `swift`, `typescript`, `python`, `cpp`, `syllabusParser` | source + syllabus readers |
|
|
645
|
+
| `extractors/` | `keywordExtractor`, `fileSelector`, `fileRelevanceScorer` | keyword weights, LLM file relevance |
|
|
646
|
+
| `graph/` | `scaffoldExtractor`, `overviewExtractor`, `stepExtractor`, `graphVerifier`, `conceptEscalator`, `hierarchicalLoAuthoring`, `loCodeStandard`, `loExemplars`, `depthAudit`, `knowledgeTreeCatalog`, `conceptStandardizer`, `knowledgeGraphPipeline`, `knowledgeGraphVerifier`, `hybridGraphPipeline` | layer 2 — the three graph generators |
|
|
647
|
+
| `concepts/` | `conceptResolver` | keyword → Master-Tree code, ULO/CIO-aware |
|
|
648
|
+
| `assembly/` | `roadmapAssembler` | milestones, phases, keyword ledger, time budget |
|
|
649
|
+
| `pipeline/` | `domainRouter`, `projectGraphPipeline`, `knowledgeGraphPipeline`, `hybridGraphPipeline` | the three **orchestrators** (thin) + the router; only `projectGraphPipeline` holds the 8 real steps |
|
|
650
|
+
| `feed/` | `curriculumFeedEmitter` | layer 3 — projects any graph onto `CurriculumFeed` v2 |
|
|
651
|
+
| `utils/` | `llmClient`, `fileUtils` | provider chain + pacing, file budget |
|
|
652
|
+
|
|
653
|
+
> `graph/` and `pipeline/` both contain files named `knowledgeGraphPipeline.ts` and
|
|
654
|
+
> `hybridGraphPipeline.ts`. They are not duplicates: the `graph/` ones do the LLM work,
|
|
655
|
+
> the `pipeline/` ones parse the syllabus, call the `graph/` one, and emit the feed.
|
|
204
656
|
|
|
205
657
|
---
|
|
206
658
|
|
|
@@ -242,7 +694,12 @@ Step 2: ROUTE (by graphType)
|
|
|
242
694
|
└── hybrid_graph → project_graph + knowledge_graph + LLM link features↔concepts
|
|
243
695
|
│
|
|
244
696
|
Step 3: OUTPUT
|
|
245
|
-
|
|
697
|
+
├── the graph object (project_graph | knowledge_graph | hybrid_graph)
|
|
698
|
+
└── CurriculumFeed v2 (learning_nodes / dependency_edges / groupings / phases)
|
|
699
|
+
↑ always produced by emitCurriculumFeed(); hybrid is the only branch that fills phases[]
|
|
700
|
+
|
|
701
|
+
Note: domain-kit returns objects in memory — it never writes *.json itself.
|
|
702
|
+
Serialising them is the host's job.
|
|
246
703
|
```
|
|
247
704
|
|
|
248
705
|
### Output
|
|
@@ -264,7 +721,7 @@ Always **one of three schemas**:
|
|
|
264
721
|
`detectDomainProfile()` runs three detectors in sequence:
|
|
265
722
|
|
|
266
723
|
1. **Input Type Detector** — Checks for repo URLs, syllabus text, file extensions, descriptions
|
|
267
|
-
2. **Domain Detector** — Multi-signal scoring with
|
|
724
|
+
2. **Domain Detector** — Multi-signal scoring with 340+ keywords across 10 domains, **multi-language support** (EN, VI, ZH, JA, KO, FR), and **context-dependent extension mapping** (e.g., `.cpp` → software by default, hardware only when Arduino/ESP32/sensor context present)
|
|
268
725
|
3. **Learning Mode Detector** — Combines input signals + domain heuristics + verb patterns
|
|
269
726
|
|
|
270
727
|
### Ambiguous Domain Handling
|
|
@@ -274,7 +731,7 @@ Some file extensions map to multiple domains (e.g., `.cpp` → software or hardw
|
|
|
274
731
|
1. Default-maps ambiguous extensions to `software` (most common use)
|
|
275
732
|
2. Checks context keywords (e.g., `arduino`, `ESP32`, `GPIO`, `sensor`) to upgrade to `hardware_iot`
|
|
276
733
|
3. Sets `ambiguous` field in output when detection is borderline
|
|
277
|
-
4. Agent should **ask user to confirm** when `ambiguous` is set or confidence < 0.6
|
|
734
|
+
4. Agent should **ask user to confirm** when `ambiguous` is set or `confidence.domainCategory < 0.6`
|
|
278
735
|
|
|
279
736
|
### Domain Categories
|
|
280
737
|
|
|
@@ -287,8 +744,8 @@ Some file extensions map to multiple domains (e.g., `.cpp` → software or hardw
|
|
|
287
744
|
| `science` | Lý, hóa, sinh | thí nghiệm, năng lượng, cell |
|
|
288
745
|
| `language` | Tiếng Anh, writing | grammar, vocabulary, essay |
|
|
289
746
|
| `arts` | Âm nhạc, thiết kế | piano, color theory, typography |
|
|
290
|
-
| `research` | Nghiên cứu, thống kê | methodology, hypothesis,
|
|
291
|
-
| `business` | Kinh doanh, tài chính | revenue, agile,
|
|
747
|
+
| `research` | Nghiên cứu, thống kê | methodology, hypothesis, regression |
|
|
748
|
+
| `business` | Kinh doanh, tài chính | revenue, agile, kế hoạch kinh doanh |
|
|
292
749
|
| `other` | Fallback | — |
|
|
293
750
|
|
|
294
751
|
### Confidence & Fallback
|
|
@@ -380,9 +837,27 @@ For projects that require **both building and understanding**.
|
|
|
380
837
|
"concept_id": "G3",
|
|
381
838
|
"relationship": "requires",
|
|
382
839
|
"depth": "cio",
|
|
383
|
-
"estimated_prereq_minutes": 30
|
|
840
|
+
"estimated_prereq_minutes": 30,
|
|
841
|
+
"phase_id": "PH2"
|
|
384
842
|
}
|
|
385
|
-
]
|
|
843
|
+
],
|
|
844
|
+
// phases[] is the teaching order — and the only source of CurriculumFeed.phases
|
|
845
|
+
"phases": [
|
|
846
|
+
{
|
|
847
|
+
"id": "PH1",
|
|
848
|
+
"order": 1,
|
|
849
|
+
"name": "Runnable slice",
|
|
850
|
+
"product_completion": "App launches and shows an empty conversation list",
|
|
851
|
+
"feature_ids": ["F0", "F1"],
|
|
852
|
+
"introduces": [
|
|
853
|
+
{ "concept_id": "G1", "aspect": "new", "note": "..." },
|
|
854
|
+
{ "concept_id": "G3", "aspect": "preview", "note": "vocabulary only" }
|
|
855
|
+
]
|
|
856
|
+
}
|
|
857
|
+
],
|
|
858
|
+
"total_project_minutes": 480,
|
|
859
|
+
"total_concept_minutes": 210,
|
|
860
|
+
"linked_prereq_minutes": 540
|
|
386
861
|
}
|
|
387
862
|
```
|
|
388
863
|
|
|
@@ -398,23 +873,34 @@ All schemas are Zod-validated TypeScript types.
|
|
|
398
873
|
| `KnowledgeGraphSchema` | `knowledgeGraphSchema.ts` | Concept-driven subject structure |
|
|
399
874
|
| `HybridGraphSchema` | `hybridGraphSchema.ts` | Combined project + knowledge |
|
|
400
875
|
| `DomainProfileSchema` | `domainProfileSchema.ts` | Domain detection output |
|
|
401
|
-
| `
|
|
402
|
-
| `
|
|
876
|
+
| `HardwareFeatureExtensionSchema` | `domainExtensions.ts` | Hardware overlay (wiring, materials) |
|
|
877
|
+
| `ThreeDesignFeatureExtensionSchema` | `domainExtensions.ts` | 3D Design overlay (shapes, dimensions) |
|
|
878
|
+
| `MathConceptExtensionSchema` | `domainExtensions.ts` | Math overlay (theorems) |
|
|
879
|
+
| `ScienceConceptExtensionSchema` | `domainExtensions.ts` | Science overlay (experiment procedures) |
|
|
880
|
+
| `LanguageConceptExtensionSchema` | `domainExtensions.ts` | Language overlay (grammar rules) |
|
|
881
|
+
| `ConceptSchema` | `knowledgeGraphSchema.ts` | One knowledge concept |
|
|
882
|
+
| `LearningPathStepSchema` | `knowledgeGraphSchema.ts` | One topological-order step |
|
|
403
883
|
| `CurriculumFeedSchema` | `curriculumFeedSchema.ts` | Planner-ready normalized feed (nodes, edges, groupings, phases) |
|
|
404
884
|
| `FeedNodeSchema` | `curriculumFeedSchema.ts` | Individual learning node (concept, skill, or product step) |
|
|
405
885
|
| `FeedEdgeSchema` | `curriculumFeedSchema.ts` | Dependency edge (knowledge or task) |
|
|
406
886
|
| `FeedGroupingSchema` | `curriculumFeedSchema.ts` | Suggested grouping of nodes (feature or category) |
|
|
407
887
|
| `FeedPhaseSchema` | `curriculumFeedSchema.ts` | Development phase — when non-empty, defines teaching sequence and unit structure |
|
|
888
|
+
| `FeedDiagnosticSchema` | `curriculumFeedSchema.ts` | Typed finding (code + severity + node ids) |
|
|
889
|
+
| `FeedProvenanceSchema` | `curriculumFeedSchema.ts` | Which LLM providers served the run |
|
|
408
890
|
|
|
409
891
|
### Key Types
|
|
410
892
|
|
|
411
893
|
```typescript
|
|
412
894
|
// Project Graph
|
|
413
895
|
type ProjectGraph = {
|
|
896
|
+
schema_version: 3;
|
|
414
897
|
project: ProjectInfo;
|
|
415
898
|
product: Product;
|
|
416
899
|
features: Feature[];
|
|
900
|
+
capabilities: Capability[];
|
|
417
901
|
implementation: { tasks: Task[] };
|
|
902
|
+
missing_gaps: MissingGap[];
|
|
903
|
+
tech_debt: TechDebt[];
|
|
418
904
|
}
|
|
419
905
|
|
|
420
906
|
type Feature = {
|
|
@@ -430,11 +916,16 @@ type Step = {
|
|
|
430
916
|
keywords: string[];
|
|
431
917
|
api_usage: string[];
|
|
432
918
|
completion_level: 'base' | 'mvp' | 'extend' | 'polish';
|
|
433
|
-
|
|
919
|
+
outcome?: { user_visible: string; technical: string }; // → FeedNode.user_visible_deliverable
|
|
920
|
+
// all effort fields are OPTIONAL — downstream totals fall back to 15 minutes
|
|
921
|
+
effort?: { estimated_minutes?: number; complexity?: 'low' | 'medium' | 'high';
|
|
922
|
+
concepts_count?: number; files_touched?: number };
|
|
434
923
|
}
|
|
435
924
|
|
|
436
925
|
// Knowledge Graph
|
|
437
926
|
type KnowledgeGraph = {
|
|
927
|
+
schema_version: 1;
|
|
928
|
+
type: 'knowledge_graph';
|
|
438
929
|
subject: SubjectInfo;
|
|
439
930
|
concepts: Concept[];
|
|
440
931
|
categories: ConceptCategory[];
|
|
@@ -444,11 +935,26 @@ type KnowledgeGraph = {
|
|
|
444
935
|
type Concept = {
|
|
445
936
|
id: string;
|
|
446
937
|
name: string;
|
|
938
|
+
description: string;
|
|
939
|
+
keywords: string[]; // SIO-level tech terms — drives the keyword ledger, STEP_5b enrichment
|
|
940
|
+
// and the SIO-binding audit; NEVER silently empty (fail-closed)
|
|
447
941
|
ulo: string; // WHAT + WHY
|
|
448
942
|
cio: string; // HOW
|
|
449
943
|
sio: string[]; // Specific implementations
|
|
450
944
|
prerequisites: string[];
|
|
945
|
+
estimated_minutes: number; // calibrated 15–60 (STEP_5c)
|
|
451
946
|
problem_types: ProblemType[];
|
|
947
|
+
techniques: string[];
|
|
948
|
+
common_mistakes: string[];
|
|
949
|
+
materials: string[]; // physical lab equipment
|
|
950
|
+
visual_aids: string[]; // diagrams / images / animations
|
|
951
|
+
// codes — a knowledge-tree code from STEP_2_5 always wins over a generated one
|
|
952
|
+
code?: string;
|
|
953
|
+
standardization?: 'standardized' | 'proposed_new';
|
|
954
|
+
tree_code?: string;
|
|
955
|
+
ulo_code?: string;
|
|
956
|
+
cio_code?: string;
|
|
957
|
+
sio_codes?: string[];
|
|
452
958
|
}
|
|
453
959
|
|
|
454
960
|
// Domain Detection
|
|
@@ -467,27 +973,51 @@ type ConceptMapping = {
|
|
|
467
973
|
concept_code: string;
|
|
468
974
|
concept_name: string;
|
|
469
975
|
depth: 'ulo' | 'cio' | 'sio'; // WHAT (ulo), HOW (cio), IMPLEMENTATION (sio)
|
|
470
|
-
|
|
976
|
+
depth_source: 'llm' | 'heuristic' | 'override'; // who decided it — the regex
|
|
977
|
+
// fallback (comment→ulo, import→sio) has no
|
|
978
|
+
// pedagogical basis and must stay distinguishable
|
|
979
|
+
depth_rationale: string;
|
|
471
980
|
evidence_files: string[];
|
|
472
981
|
}
|
|
473
982
|
|
|
474
|
-
// Assembled Roadmap (with time budget)
|
|
983
|
+
// Assembled Roadmap (with time budget) — TypeScript interface, not a Zod schema
|
|
475
984
|
type AssembledRoadmap = {
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
985
|
+
project: ProjectGraph['project'];
|
|
986
|
+
product: ProjectGraph['product'];
|
|
987
|
+
phases: AssembledPhase[]; // { phase_id, phase_name, milestones, total_minutes,
|
|
988
|
+
// concept_progression }
|
|
989
|
+
feature_concepts: Map<string, ConceptMapping[]>;
|
|
480
990
|
time_budget: {
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
991
|
+
session_minutes: number; // 90
|
|
992
|
+
overhead_factor: number; // 1.15
|
|
993
|
+
total_estimated_minutes: number; // Σ milestone.estimated_minutes (raw)
|
|
994
|
+
total_adjusted_minutes: number; // Σ milestone.adjusted_minutes (after overhead)
|
|
484
995
|
};
|
|
485
|
-
}
|
|
996
|
+
};
|
|
997
|
+
|
|
998
|
+
type AssembledMilestone = {
|
|
999
|
+
id: string; name: string; description: string;
|
|
1000
|
+
feature_id: string; feature_name: string;
|
|
1001
|
+
concept_code: string; depth: 'ulo' | 'cio' | 'sio';
|
|
1002
|
+
all_keywords: string[]; // keyword ledger lives HERE, per milestone
|
|
1003
|
+
new_keywords: string[];
|
|
1004
|
+
prerequisite_keywords: string[];
|
|
1005
|
+
steps: Step[];
|
|
1006
|
+
estimated_minutes: number;
|
|
1007
|
+
adjusted_minutes: number; // estimated_minutes × overhead_factor
|
|
1008
|
+
completion_level: string;
|
|
1009
|
+
};
|
|
486
1010
|
|
|
487
1011
|
// Curriculum Feed (planner-ready projection)
|
|
488
1012
|
type CurriculumFeed = {
|
|
489
1013
|
schema_version: 2;
|
|
490
|
-
source: {
|
|
1014
|
+
source: {
|
|
1015
|
+
graph_type: 'project_graph' | 'knowledge_graph' | 'hybrid_graph';
|
|
1016
|
+
warnings: string[]; // DERIVED from diagnostics (severity ≠ 'info')
|
|
1017
|
+
hallucination_count: number;
|
|
1018
|
+
diagnostics: FeedDiagnostic[]; // { code, severity, message, node_ids }
|
|
1019
|
+
provenance?: FeedProvenance; // providers / models / calls / degraded
|
|
1020
|
+
};
|
|
491
1021
|
learning_nodes: FeedNode[]; // in TEACHING ORDER when phases[] is non-empty
|
|
492
1022
|
dependency_edges: FeedEdge[];
|
|
493
1023
|
suggested_groupings: FeedGrouping[];
|
|
@@ -514,6 +1044,7 @@ type FeedNode = {
|
|
|
514
1044
|
depth_variants?: { ulo: number; cio: number; sio: number }; // minutes per depth level — reconciler price list (ULO ≤ CIO ≤ SIO)
|
|
515
1045
|
depth_scaffold_candidates?: { from_depth: 'ulo' | 'cio' | 'sio'; to_depth: 'ulo' | 'cio' | 'sio'; minutes_saved: number; reason: string }[]; // DEPTH downgrades (SIO→CIO, CIO→ULO) — parallel to scaffold_candidates
|
|
516
1046
|
is_core: boolean; // Master-Tree core concept — never depth-downgraded; escalate instead
|
|
1047
|
+
external_prerequisites: string[]; // what the learner must already know that this feed does not teach
|
|
517
1048
|
}
|
|
518
1049
|
|
|
519
1050
|
type FeedPhase = {
|
|
@@ -560,6 +1091,7 @@ const profile = detectDomainProfile({
|
|
|
560
1091
|
fileExtensions?: string[];
|
|
561
1092
|
techStack?: string;
|
|
562
1093
|
gradeBand?: [number, number];
|
|
1094
|
+
parsedSyllabus?: ParsedSyllabus; // pre-parsed syllabus; skips parseSyllabus()
|
|
563
1095
|
});
|
|
564
1096
|
```
|
|
565
1097
|
|
|
@@ -576,12 +1108,13 @@ const result = await runProjectGraphPipeline({
|
|
|
576
1108
|
techStack: string; // Required: comma-separated technologies
|
|
577
1109
|
llmConfig?: LlmClientConfig;
|
|
578
1110
|
embeddings?: Record<string, { embedding?: number[] }>;
|
|
579
|
-
sessionMinutes?: number; // Lesson/session duration (default: 90)
|
|
580
|
-
overheadFactor?: number; // Setup/transition overhead multiplier (default: 1.15)
|
|
581
|
-
maxMilestoneMinutes?: number; // Max minutes before splitting milestone (default: 120)
|
|
582
1111
|
onProgress?: (step: string, message: string) => void;
|
|
583
1112
|
});
|
|
584
1113
|
|
|
1114
|
+
// NOTE: sessionMinutes / overheadFactor / maxMilestoneMinutes are NOT options here.
|
|
1115
|
+
// STEP_8 hardcodes 90 / 1.15 / 120 when it calls assembleRoadmap(). Call
|
|
1116
|
+
// assembleRoadmap() yourself if you need different values.
|
|
1117
|
+
|
|
585
1118
|
// Returns:
|
|
586
1119
|
{
|
|
587
1120
|
projectGraph: ProjectGraph;
|
|
@@ -590,6 +1123,7 @@ const result = await runProjectGraphPipeline({
|
|
|
590
1123
|
hallucinations: Hallucination[];
|
|
591
1124
|
keywords: Keyword[];
|
|
592
1125
|
featureConcepts: Map<string, ConceptMapping[]>;
|
|
1126
|
+
provenance: LlmCallRecord[]; // every provider attempt this run made
|
|
593
1127
|
}
|
|
594
1128
|
```
|
|
595
1129
|
|
|
@@ -606,17 +1140,26 @@ const result = await generateKnowledgeGraph({
|
|
|
606
1140
|
syllabusText?: string; // Full syllabus text (multi-language supported)
|
|
607
1141
|
gradeBand?: [number, number];
|
|
608
1142
|
domain?: string;
|
|
1143
|
+
techStack?: string; // target tech for SIO keyword binding (e.g. "python")
|
|
1144
|
+
knowledgeTreePath?: string; // path to mlo-knowlege-tree.tsv → enables STEP_2_5 standardization
|
|
1145
|
+
proposalsPath?: string; // where proposed-new concepts are persisted for the later merge
|
|
609
1146
|
llmConfig?: LlmClientConfig;
|
|
1147
|
+
onProgress?: (step: string, message: string) => void;
|
|
610
1148
|
});
|
|
611
|
-
// Internally runs
|
|
1149
|
+
// Internally runs STEP_1 … STEP_6 — see [Pipeline Details](#pipeline-details).
|
|
612
1150
|
|
|
613
1151
|
// Returns:
|
|
614
1152
|
{
|
|
615
1153
|
knowledgeGraph: KnowledgeGraph;
|
|
616
1154
|
warnings: string[];
|
|
1155
|
+
verificationReport?: KnowledgeGraphVerificationReport; // uncovered topics, missing ULO/CIO/SIO…
|
|
617
1156
|
}
|
|
618
1157
|
```
|
|
619
1158
|
|
|
1159
|
+
> Prefer `runKnowledgeGraphPipeline()` over `generateKnowledgeGraph()`: it parses the
|
|
1160
|
+
> syllabus, applies syllabus-derived defaults (subject, domain, grade band) and emits the
|
|
1161
|
+
> `CurriculumFeed` for you.
|
|
1162
|
+
|
|
620
1163
|
### `generateHybridGraph(options) → HybridGraphPipelineResult`
|
|
621
1164
|
|
|
622
1165
|
Links project graph with knowledge graph.
|
|
@@ -626,13 +1169,15 @@ import { generateHybridGraph } from '@thanh01.pmt/domain-kit';
|
|
|
626
1169
|
|
|
627
1170
|
const result = await generateHybridGraph({
|
|
628
1171
|
projectGraph: ProjectGraph;
|
|
629
|
-
knowledgeGraph: KnowledgeGraph;
|
|
1172
|
+
knowledgeGraph: KnowledgeGraph; // both are required
|
|
630
1173
|
llmConfig?: LlmClientConfig;
|
|
1174
|
+
onProgress?: (step: string, message: string) => void;
|
|
631
1175
|
});
|
|
632
1176
|
|
|
633
1177
|
// Returns:
|
|
634
1178
|
{
|
|
635
|
-
hybridGraph: HybridGraph;
|
|
1179
|
+
hybridGraph: HybridGraph; // links[] + phases[] + total_project_minutes /
|
|
1180
|
+
// total_concept_minutes / linked_prereq_minutes
|
|
636
1181
|
warnings: string[];
|
|
637
1182
|
}
|
|
638
1183
|
```
|
|
@@ -654,8 +1199,13 @@ const feed = emitCurriculumFeed(existingFeed); // pass-through (already
|
|
|
654
1199
|
const feed = emitCurriculumFeed(graph, {
|
|
655
1200
|
hallucinationCount: 3, // metadata for source info
|
|
656
1201
|
warnings: ['Feature F5 had low confidence'],
|
|
1202
|
+
provenance: { providers: ['nvidia'], models: ['nemotron'], calls: 7, failed_calls: 0, degraded: false },
|
|
657
1203
|
});
|
|
658
1204
|
|
|
1205
|
+
// Gate on diagnostics rather than substring-matching warning text:
|
|
1206
|
+
const blockers = feed.source.diagnostics.filter((d) => d.severity === 'error');
|
|
1207
|
+
if (blockers.length) throw new Error(blockers.map((d) => d.code).join(', '));
|
|
1208
|
+
|
|
659
1209
|
// Returns CurriculumFeed with:
|
|
660
1210
|
// - learning_nodes[] — unified node format (concept | skill | product_step), in teaching order when phases[] is non-empty
|
|
661
1211
|
// - dependency_edges[] — knowledge (prerequisite) or task (build order) edges
|
|
@@ -700,93 +1250,52 @@ const keywords = extractKeywords(parsedSourceContext);
|
|
|
700
1250
|
|
|
701
1251
|
## Pipeline Details
|
|
702
1252
|
|
|
703
|
-
### Project Graph Pipeline (
|
|
1253
|
+
### Project Graph Pipeline (product-driven)
|
|
704
1254
|
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
→ Parse all source files (Swift, TS, Python, C++)
|
|
708
|
-
→ Merge by language, extract imports/types/functions/wrappers
|
|
709
|
-
|
|
710
|
-
STEP 2: Keyword Extraction
|
|
711
|
-
→ Filter stdlib, assign weights, tag platform (app/esp32)
|
|
712
|
-
|
|
713
|
-
STEP 3: SDK API Index
|
|
714
|
-
→ Build symbol index from AST results
|
|
715
|
-
|
|
716
|
-
STEP 4: LLM Graph Generation
|
|
717
|
-
C0: Scaffold F0 (foundation features: tools, setup, minimal knowledge)
|
|
718
|
-
C1: Overview (features meta, journeys, architecture, development stages)
|
|
719
|
-
C2: Steps per feature (batched or per-feature LLM calls)
|
|
720
|
-
|
|
721
|
-
STEP 5: Verification
|
|
722
|
-
→ Remove hallucinated files, APIs, keywords not found in code
|
|
723
|
-
→ F0 keywords exempt (pedagogical terms)
|
|
724
|
-
|
|
725
|
-
STEP 6: Concept Escalation
|
|
726
|
-
→ Map keywords → neutral concepts via LLM
|
|
727
|
-
→ Classify each keyword as ULO (WHAT/WHY), CIO (HOW), or SIO (IMPLEMENTATION)
|
|
728
|
-
→ Infer depth from code context (function bodies→sio, types→cio, imports→sio, docs→ulo)
|
|
729
|
-
→ Join evidence files
|
|
730
|
-
|
|
731
|
-
STEP 7: Concept Resolution
|
|
732
|
-
→ Map concepts → Master Tree codes with ULO/CIO-aware scoring:
|
|
733
|
-
- SIO keywords matched against concept.keywords
|
|
734
|
-
- CIO keywords matched against concept.description + concept.cio
|
|
735
|
-
- ULO keywords matched against concept.ulo + concept.description
|
|
736
|
-
→ Depth field propagated to resolved concepts
|
|
737
|
-
|
|
738
|
-
STEP 8: Assembly
|
|
739
|
-
→ Group features into phases (session-time-aware)
|
|
740
|
-
→ Respect sessionMinutes, overheadFactor, maxMilestoneMinutes constraints
|
|
741
|
-
→ Split large features across multiple milestones when exceeding time budget
|
|
742
|
-
→ Compute all_keywords / new_keywords / prerequisite_keywords per milestone
|
|
743
|
-
→ Output time_budget with total_estimated_minutes + overhead_minutes
|
|
744
|
-
|
|
745
|
-
STEP 9: Feed Emission
|
|
746
|
-
→ emitCurriculumFeed(graph) — pure transformation, no LLM
|
|
747
|
-
→ Convert project_graph → FeedNode[] (kind: product_step) + FeatureGroupings
|
|
748
|
-
→ Convert knowledge_graph → FeedNode[] (kind: concept) + CategoryGroupings
|
|
749
|
-
→ Convert hybrid_graph → combined nodes + knowledge/task edges + hybrid link depth hints
|
|
750
|
-
→ Project steps: project outcome.user_visible onto user_visible_deliverable (checkpoint candidate)
|
|
751
|
-
→ Concept nodes: emit depth_variants (ULO ≤ CIO ≤ SIO minutes) + depth_scaffold_candidates + is_core
|
|
752
|
-
→ Teaching order: when phases[] exist, nodes are emitted in phase sequence; new/prerequisite
|
|
753
|
-
keywords computed in TEACHING order (first appearance = new), never array position
|
|
754
|
-
→ Preview priming: phase-declared __PREVxx awareness nodes carry keywords.all (legitimate
|
|
755
|
-
vocabulary introducer) without fabricating prerequisite edges
|
|
756
|
-
→ Validate: fail-closed (duplicate IDs, edge resolution, self-loops, non-empty, duplicate
|
|
757
|
-
phase ids, and the teaching-order invariant: a prerequisite must live in the same or an
|
|
758
|
-
EARLIER phase than its consumer — a backwards edge throws at emission, never in the planner)
|
|
759
|
-
```
|
|
1255
|
+
`runProjectGraphPipeline()` — the only pipeline whose steps live inside its orchestrator.
|
|
1256
|
+
Step ids below are the ones it actually logs.
|
|
760
1257
|
|
|
761
|
-
|
|
1258
|
+
| Step | What happens | Fail mode |
|
|
1259
|
+
|---|---|---|
|
|
1260
|
+
| `STEP_1` | scan the file **tree** (names only) → LLM selects relevant files → read only those → parse → merge by primary language | parse failure is per-file, non-fatal |
|
|
1261
|
+
| `STEP_2` | `extractKeywords()` — filter stdlib, assign weights, tag platform (`app` / `esp32`) | — |
|
|
1262
|
+
| `STEP_3` | SDK API index, derived from STEP_2 keywords | — |
|
|
1263
|
+
| `STEP_4_C0` | scaffold → feature `F0` "FOUNDATION & SETUP" | scaffold extractor returns `null`; the pipeline continues **without** F0 |
|
|
1264
|
+
| `STEP_4_C1` | overview → product meta (goals, users, journeys, development stages) + the feature list | — |
|
|
1265
|
+
| `STEP_4_C2` | steps per feature — one batched LLM call, falling back to per-feature | a feature that still fails is logged and left step-less; **no throw** |
|
|
1266
|
+
| `STEP_5` | `verifyProjectGraph()` — strip files / APIs / keywords not found in code; F0 keywords exempt | findings returned as `hallucinations[]`, graph is repaired |
|
|
1267
|
+
| `STEP_6` | `escalateAndMapConcepts()` — keyword → neutral concept + `ulo`/`cio`/`sio` + evidence files. Depth from the LLM; regex fallback on LLM failure (import → sio, code body → sio, comment → ulo) | LLM failure degrades to the regex fallback |
|
|
1268
|
+
| `STEP_7` | `resolveConcepts()` — concept → Master-Tree code, scored ULO/CIO-aware | unmatched keywords become `proposed` concepts |
|
|
1269
|
+
| `STEP_8` | `assembleRoadmap()` — milestones grouped into phases; each milestone gets `all` / `new` / `prerequisite` keywords and a time budget (90 min sessions, ×1.15 overhead, 120 min split cap — hardcoded) | — |
|
|
1270
|
+
|
|
1271
|
+
There is **no STEP 9 here.** Feed emission happens one layer up: `runDomainPipeline()`
|
|
1272
|
+
calls `emitCurriculumFeed()` for the product branch, and the knowledge / hybrid
|
|
1273
|
+
orchestrators call it for theirs.
|
|
1274
|
+
|
|
1275
|
+
### Knowledge Graph Pipeline (concept-driven)
|
|
1276
|
+
|
|
1277
|
+
`generateKnowledgeGraph()` — the LLM work. Step ids as logged:
|
|
1278
|
+
|
|
1279
|
+
| Step | What happens | Fail mode |
|
|
1280
|
+
|---|---|---|
|
|
1281
|
+
| `STEP_1` | parse the syllabus into structured units/topics, then LLM-decompose the subject into concepts, prerequisites, `problem_types`, categories | syllabus truncated at `KG_MAX_SYLLABUS_CHARS` (default 16000) + warning |
|
|
1282
|
+
| `STEP_2` | validate & clean — description ≥ 50 chars; **keyword registry fail-closed** (empty keywords are derived from name/description; nothing derivable → `throw`); dangling prerequisite edges removed | `throw` when keywords cannot be derived |
|
|
1283
|
+
| `STEP_2_5` | *only when `knowledgeTreePath` is supplied* — standardize every concept against `mlo-knowlege-tree.tsv` into `standardized` or `proposed_new`; adopted tech-stack keywords merge into the registry; tree prerequisite codes seed internal edges | missing decision for a concept → `throw` |
|
|
1284
|
+
| `STEP_3` | `detectPrerequisiteCycles()` → `breakPrerequisiteCycles()` at the weakest edge | warnings; edges are removed |
|
|
1285
|
+
| `STEP_4` | `topologicalSort()` (Kahn) — `learning_path` is **rebuilt** from the sorted order | incomplete sort → warning |
|
|
1286
|
+
| `STEP_5` | hierarchical ULO/CIO/SIO authoring + deterministic `ulo_code` / `cio_code` / `sio_codes` (a STEP_2_5 tree code always wins) | — |
|
|
1287
|
+
| `STEP_5b` | mine SIO text for tech identifiers (backticked tokens, `@Annotations`, `a.b()`, `.modifiers`, camelCase, cap 8) → merge into keywords | — |
|
|
1288
|
+
| `STEP_5c` | minute calibration — each signal normalised against its own cap before weighting (`0.30·sio + 0.25·problem_types + 0.20·prereqs + 0.25·highBloom`, all over `min(count/4, 1)`), `minutes = 15 + 45·score` ⇒ 15–60 min. Overrides the LLM only when the two diverge by more than 50% | — |
|
|
1289
|
+
| `STEP_6` | `verifyKnowledgeGraph()` — referential integrity + **reverse coverage** (every syllabus topic covered by ≥ 1 concept) | warnings, never throws |
|
|
762
1290
|
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
→ DFS to detect cycles in prerequisite graph
|
|
772
|
-
→ Break cycles by removing weakest back-edges
|
|
773
|
-
→ Generate warnings for each broken cycle
|
|
774
|
-
|
|
775
|
-
STEP 3: Topological Sort (Kahn's Algorithm)
|
|
776
|
-
→ Sort concepts in valid learning order
|
|
777
|
-
→ Assign sequential order numbers
|
|
778
|
-
→ Estimate concept depth per milestone (ULO→CIO→SIO progression)
|
|
779
|
-
|
|
780
|
-
STEP 4: Validation
|
|
781
|
-
→ Check prerequisite references exist
|
|
782
|
-
→ Check learning_path references exist
|
|
783
|
-
→ Generate warnings for invalid references
|
|
784
|
-
|
|
785
|
-
STEP 5: Feed Emission
|
|
786
|
-
→ emitCurriculumFeed(knowledgeGraph) → CurriculumFeed
|
|
787
|
-
→ Concept nodes with bloom_hint (from problem_types) + depth_hint (from sio count)
|
|
788
|
-
→ Category groupings for suggested lesson grouping
|
|
789
|
-
```
|
|
1291
|
+
Feed emission is **not** a step of `generateKnowledgeGraph()` — it is the next line of
|
|
1292
|
+
`runKnowledgeGraphPipeline()` in `src/pipeline/`.
|
|
1293
|
+
|
|
1294
|
+
### Hybrid Graph Pipeline (both)
|
|
1295
|
+
|
|
1296
|
+
`generateHybridGraph()` — see [Layer 2 · Branch C](#layer-2---generate-one-branch-by-learning-mode)
|
|
1297
|
+
for the annotated diagram. `STEP_1` links, `STEP_2` phases (up to 3 repair attempts, then
|
|
1298
|
+
`throw`), `STEP_3` totals.
|
|
790
1299
|
|
|
791
1300
|
---
|
|
792
1301
|
|
|
@@ -839,7 +1348,7 @@ profile.graphType === 'hybrid_graph'
|
|
|
839
1348
|
→ You need BOTH source code AND subject description.
|
|
840
1349
|
→ Run projectGraphPipeline() + generateKnowledgeGraph() + generateHybridGraph().
|
|
841
1350
|
|
|
842
|
-
profile.confidence < 0.6
|
|
1351
|
+
profile.confidence.domainCategory < 0.6 // confidence is an OBJECT — always compare the field
|
|
843
1352
|
→ Low confidence. Ask the user to confirm domain category.
|
|
844
1353
|
|
|
845
1354
|
profile.ambiguous !== undefined
|
|
@@ -955,10 +1464,10 @@ try {
|
|
|
955
1464
|
2. **Missing syllabus**: `generateKnowledgeGraph()` needs `syllabusText` to decompose concepts
|
|
956
1465
|
3. **Low confidence detection**: If `profile.confidence.domainCategory < 0.6`, the detection may be wrong — ask user to confirm
|
|
957
1466
|
4. **Ambiguous domain**: If `profile.ambiguous` is set, present both options to user (e.g., `.cpp` with ESP32 context could be software or hardware)
|
|
958
|
-
5. **LLM not configured**:
|
|
1467
|
+
5. **LLM not configured**: graph generation needs at least one provider key — `LLM_API_KEY` / `OPENAI_API_KEY`, `DASHSCOPE_API_KEY`, `NVIDIA_API_KEY` or `OPENROUTER_API_KEY` (see [Environment Variables](#environment-variables-for-llm-features))
|
|
959
1468
|
6. **Large repos**: The pipeline reads up to 70 files / 500K chars — very large repos may be truncated
|
|
960
1469
|
7. **ULO/CIO/SIO depth**: Concept escalation now classifies keywords by depth level — ULO (intro), CIO (mechanism), SIO (implementation). This affects how `curriculum-kit` structures exposition depth per lesson
|
|
961
|
-
8. **Time budget**: `assembleRoadmap()` respects `sessionMinutes` and `maxMilestoneMinutes` — large features are automatically split across milestones.
|
|
1470
|
+
8. **Time budget**: `assembleRoadmap()` respects `sessionMinutes` and `maxMilestoneMinutes` — large features are automatically split across milestones. `runProjectGraphPipeline()` does **not** forward these; it hardcodes 90 / 1.15 / 120. Call `assembleRoadmap()` yourself to change them
|
|
962
1471
|
|
|
963
1472
|
### Available Exports
|
|
964
1473
|
|
|
@@ -977,24 +1486,28 @@ import { extractKeywords } from '@thanh01.pmt/domain-kit';
|
|
|
977
1486
|
|
|
978
1487
|
// Schema validation
|
|
979
1488
|
import {
|
|
980
|
-
ProjectGraphSchema, KnowledgeGraphSchema, HybridGraphSchema,
|
|
981
|
-
|
|
982
|
-
KnowledgeConceptSchema, KnowledgeLearningPathStepSchema,
|
|
1489
|
+
ProjectGraphSchema, KnowledgeGraphSchema, HybridGraphSchema, DomainProfileSchema,
|
|
1490
|
+
ConceptSchema, LearningPathStepSchema, ProblemTypeSchema,
|
|
983
1491
|
CurriculumFeedSchema, FeedNodeSchema, FeedEdgeSchema, FeedGroupingSchema, FeedPhaseSchema,
|
|
1492
|
+
validateCurriculumFeed,
|
|
984
1493
|
} from '@thanh01.pmt/domain-kit';
|
|
985
1494
|
|
|
1495
|
+
// Graph verification
|
|
1496
|
+
import { verifyKnowledgeGraph, detectPrerequisiteCycles, breakPrerequisiteCycles, auditSyllabusCoverage }
|
|
1497
|
+
from '@thanh01.pmt/domain-kit';
|
|
1498
|
+
|
|
986
1499
|
// Roadmap assembly
|
|
987
1500
|
import { assembleRoadmap } from '@thanh01.pmt/domain-kit';
|
|
988
1501
|
|
|
989
1502
|
// Concept resolution (ULO/CIO-aware)
|
|
990
1503
|
import { resolveConcepts } from '@thanh01.pmt/domain-kit';
|
|
991
1504
|
|
|
992
|
-
// Graph verification
|
|
993
|
-
import { verifyProjectGraph } from '@thanh01.pmt/domain-kit';
|
|
994
|
-
|
|
995
1505
|
// Concept escalation (with ULO/CIO/SIO depth classification)
|
|
996
1506
|
import { escalateAndMapConcepts } from '@thanh01.pmt/domain-kit';
|
|
997
1507
|
|
|
1508
|
+
// Router — detect + dispatch + emit in one call
|
|
1509
|
+
import { runDomainPipeline } from '@thanh01.pmt/domain-kit';
|
|
1510
|
+
|
|
998
1511
|
// Curriculum Feed emission (any graph → planner-ready feed)
|
|
999
1512
|
import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
|
|
1000
1513
|
|
|
@@ -1002,7 +1515,7 @@ import { emitCurriculumFeed } from '@thanh01.pmt/domain-kit';
|
|
|
1002
1515
|
import { validateCurriculumFeed } from '@thanh01.pmt/domain-kit';
|
|
1003
1516
|
```
|
|
1004
1517
|
|
|
1005
|
-
> **Note:**
|
|
1518
|
+
> **Note:** `topologicalSort()` (Kahn) and the depth fallback (`inferDepthFromSource`) are module-private — they run inside the pipelines. `detectPrerequisiteCycles`, `breakPrerequisiteCycles` and `auditSyllabusCoverage` **are** exported, from `graph/knowledgeGraphVerifier`.
|
|
1006
1519
|
|
|
1007
1520
|
---
|
|
1008
1521
|
|
|
@@ -1090,6 +1603,7 @@ const knowledgeResult = await generateKnowledgeGraph({
|
|
|
1090
1603
|
subject: 'Lý thuyết điện tử',
|
|
1091
1604
|
syllabusText: profile.syllabusText,
|
|
1092
1605
|
domain: 'science',
|
|
1606
|
+
techStack: 'Arduino,ESP32', // ← binds SIO keywords to the project's real APIs
|
|
1093
1607
|
});
|
|
1094
1608
|
|
|
1095
1609
|
// Step C
|