dexin-content 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,289 +1,297 @@
1
- // ─────────────────────────────────────────────────────────────
2
- // dexin-content/collection.ts
3
- //
4
- // Collection declaration + source adaptation + discovery +
5
- // batch compile orchestration.
6
- //
7
- // The batch compileCollections() wraps the formal per-file
8
- // core/compiler.ts compile() — it discovers files, builds
9
- // CompileInput per file, calls compile, and writes PositiveArtifact
10
- // docs + ContentIndex to the store.
11
- //
12
- // grep-zero: no business vocabulary appears anywhere in this
13
- // module — including error messages.
14
- // ─────────────────────────────────────────────────────────────
15
-
16
- import { readdir, readFile } from 'node:fs/promises'
17
- import { join } from 'node:path'
18
- import { createHash } from 'node:crypto'
19
- import type {
20
- PositiveArtifact,
21
- DomainName
22
- } from './core/types'
23
- import type { DomainParserRegistry, CompileInput } from './core/compiler'
24
- import { compile } from './core/compiler'
25
- import { buildDocumentIdentity, sanitizeSourceText } from './core/discovery'
26
- import { splitFrontmatter } from './core/frontmatter'
27
- import type { ArtifactStore, ContentIndex, IndexEntry } from './store'
28
-
29
- // ── Collection types ──
30
-
31
- export interface CollectionDefinition {
32
- /** Collection name (appears in identity.collection + IndexEntry.collection). */
33
- name: string
34
- /** Directory under content root; .md files collected recursively. */
35
- source: string
36
- /** Domain tag; defaults to 'document' when omitted. */
37
- domain?: DomainName | string
38
- /** Frontmatter schema (reuses core Schema; validator-only, no meta projection). */
39
- schema?: { required?: string[]; types?: Record<string, 'string' | 'number' | 'boolean'> }
40
- }
41
-
42
- export interface ResolvedCollection extends CollectionDefinition {
43
- domain: string
44
- sourceRoot: string
45
- }
46
-
47
- /** Declaration passthrough (declarative API semantic); resolution in resolveCollections. */
48
- export function defineCollection (def: CollectionDefinition): CollectionDefinition {
49
- return def
50
- }
51
-
52
- /** Resolve definitions: fill domain default + bind absolute sourceRoot. */
53
- export function resolveCollections (
54
- defs: CollectionDefinition[],
55
- contentRoot: string
56
- ): ResolvedCollection[] {
57
- return defs.map(d => ({
58
- ...d,
59
- domain: d.domain ?? 'document',
60
- sourceRoot: join(contentRoot, d.source)
61
- }))
62
- }
63
-
64
- // ── SourceAdapter ──
65
-
66
- export interface SourceAdapter {
67
- list(): Promise<string[]>
68
- read(relPath: string): Promise<string>
69
- }
70
-
71
- /**
72
- * Local filesystem source. Reads .md files recursively from root.
73
- * CRLF protection (R-d): read() throws LINE_ENDING_CONTAMINATION
74
- * on CRLF so the batch path enjoys the same guard as the synchronous
75
- * single-file read path.
76
- *
77
- * Sanitisation (BOM strip + CRLF guard) is delegated to the shared
78
- * `sanitizeSourceText` pure helper in core/discovery — byte-identical
79
- * to the synchronous readSourceFile() path.
80
- */
81
- export function createLocalSource (root: string): SourceAdapter {
82
- async function walk (dir: string, base: string): Promise<string[]> {
83
- const entries = await readdir(dir, { withFileTypes: true })
84
- const out: string[] = []
85
- for (const e of entries) {
86
- const rel = base ? `${base}/${e.name}` : e.name
87
- if (e.isDirectory()) out.push(...await walk(join(dir, e.name), rel))
88
- else if (e.isFile() && e.name.endsWith('.md')) out.push(rel)
89
- }
90
- return out.sort()
91
- }
92
- return {
93
- list: () => walk(root, ''),
94
- read: async (rel) => {
95
- const raw = await readFile(join(root, rel), 'utf8')
96
- return sanitizeSourceText(raw, rel)
97
- }
98
- }
99
- }
100
-
101
- /** Virtual in-memory source (test use). */
102
- export function createMemorySource (files: Record<string, string>): SourceAdapter {
103
- return {
104
- list: async () => Object.keys(files).sort(),
105
- read: async (rel) => {
106
- if (!(rel in files)) throw new Error(`[MemorySource] Not found: ${rel}`)
107
- return files[rel]!
108
- }
109
- }
110
- }
111
-
112
- // ── Discovery ──
113
-
114
- export interface ContentFile {
115
- id: string
116
- path: string
117
- collection: string
118
- file: string
119
- /**
120
- * Raw source (YAML frontmatter + Markdown body), already BOM/CRLF sanitised.
121
- * Cached at discovery phase so compileCollections() avoids a second IO read
122
- * per file; the split into frontmatter/body happens only when needed below.
123
- */
124
- raw: string
125
- }
126
-
127
- /**
128
- * Scan all collections via source. Files not belonging to any
129
- * collection → throw (fail-fast, no silent drop).
130
- *
131
- * Identity derivation: document path convention via
132
- * buildDocumentIdentity (id = rel minus .md minus trailing /index,
133
- * path = '/' + id, file = rel, collection = name). Non-document
134
- * domains are not supported in the batch path; hosts requiring
135
- * custom identity shapes must construct CompileInput per file
136
- * themselves — a thrown error makes the gap explicit.
137
- */
138
- export async function discover (
139
- source: SourceAdapter,
140
- collections: ResolvedCollection[]
141
- ): Promise<ContentFile[]> {
142
- const byDir = new Map(collections.map(c => [c.source, c]))
143
- const all = await source.list()
144
- const files: ContentFile[] = []
145
- for (const rel of all) {
146
- const top = rel.split('/')[0]!
147
- const col = byDir.get(top)
148
- if (!col) {
149
- throw new Error(
150
- `[discover] File '${rel}' belongs to no collection (top dir '${top}' undefined).`
151
- )
152
- }
153
- if (col.domain !== 'document') {
154
- throw new Error(
155
- `[discover] Domain '${col.domain}' discovery not yet implemented in batch path; ` +
156
- `only 'document' domain is supported here. Non-document domains require ` +
157
- `explicit identity construction.`
158
- )
159
- }
160
- const identity = buildDocumentIdentity(rel, col.name)
161
- const raw = await source.read(rel)
162
- files.push({
163
- id: identity.id,
164
- path: identity.path,
165
- collection: col.name,
166
- file: identity.file,
167
- raw
168
- })
169
- }
170
- return files
171
- }
172
-
173
- // ── Batch compile orchestration ──
174
-
175
- function checksum (s: string): string {
176
- return createHash('md5').update(s, 'utf-8').digest('hex').slice(0, 12)
177
- }
178
-
179
- /**
180
- * Batch compile: discover → per-file compile → write docs + index to store.
181
- *
182
- * For each ContentFile:
183
- * 1. Build CompileInput (id = file.id, identity = buildDocumentIdentity,
184
- * source = raw, schema = col.schema)
185
- * 2. Call formal compile(input, registry)
186
- * 3. On positive → collect PositiveArtifact + IndexEntry
187
- * On error → collect error message (fail-fast: throw after all files processed)
188
- *
189
- * Store receives only positive artifacts. compile failures are aggregated
190
- * and thrown as a single Error.
191
- */
192
- export async function compileCollections (
193
- source: SourceAdapter,
194
- collections: ResolvedCollection[],
195
- registry: DomainParserRegistry,
196
- store: ArtifactStore
197
- ): Promise<{ docs: PositiveArtifact[]; index: ContentIndex }> {
198
- const files = await discover(source, collections)
199
- const docs: PositiveArtifact[] = []
200
- const entries: IndexEntry[] = []
201
- const errors: string[] = []
202
-
203
- for (const f of files) {
204
- const col = collections.find(c => c.name === f.collection)!
205
- // Reuse discovery-cached source; no second per-file IO read.
206
- // `body` (post-frontmatter) is derived once solely for the
207
- // change-detection checksum — compile() takes the full raw
208
- // source per its stable CompileInput contract.
209
- const { body } = splitFrontmatter(f.raw)
210
- const input: CompileInput = {
211
- id: f.id,
212
- domain: col.domain,
213
- identity: buildDocumentIdentity(f.file, col.name),
214
- source: f.raw,
215
- file: f.file,
216
- schema: col.schema
217
- }
218
- const result = compile(input, registry)
219
- if (result.kind !== 'positive') {
220
- errors.push(` ✗ ${f.file}: ${result.error?.message ?? 'unknown'}`)
221
- continue
222
- }
223
- const artifact = result.artifact as PositiveArtifact
224
- docs.push(artifact)
225
- entries.push({
226
- id: f.id,
227
- path: f.path,
228
- collection: f.collection,
229
- domain: col.domain,
230
- file: f.file,
231
- meta: result.meta,
232
- checksum: checksum(body)
233
- })
234
- }
235
-
236
- if (errors.length > 0) {
237
- throw new Error(`Compile failed (${errors.length} error(s)):\n${errors.join('\n')}`)
238
- }
239
-
240
- const index: ContentIndex = {
241
- generator: 'dexin-content',
242
- builtAt: new Date().toISOString(),
243
- entries
244
- }
245
-
246
- await store.writeIndex(index)
247
- for (const doc of docs) {
248
- await store.writeDoc(doc)
249
- }
250
-
251
- return { docs, index }
252
- }
253
-
254
- // ── Incremental recompile ──
255
-
256
- /**
257
- * Detect changed/removed docs by checksum diff against the previous index.
258
- *
259
- * Wraps compileCollections (which writes index + all docs to store), then
260
- * reports which ids changed (checksum differs from prev) or were removed
261
- * (present in prev, absent in next). Used by Dev flows (memory store)
262
- * to drive incremental re-render signals. Writing all docs is acceptable
263
- * for ephemeral in-memory Dev stores. Watcher / preview server
264
- * integrations are out of scope.
265
- */
266
- export async function recompileChanged (
267
- source: SourceAdapter,
268
- collections: ResolvedCollection[],
269
- registry: DomainParserRegistry,
270
- store: ArtifactStore
271
- ): Promise<{ changed: string[]; removed: string[] }> {
272
- const prev = await store.readIndex()
273
- const prevMap = new Map((prev?.entries ?? []).map(e => [e.id, e]))
274
-
275
- const { index } = await compileCollections(source, collections, registry, store)
276
-
277
- const nextIds = new Set(index.entries.map(e => e.id))
278
- const changed: string[] = []
279
- const removed: string[] = []
280
-
281
- for (const e of index.entries) {
282
- if (prevMap.get(e.id)?.checksum !== e.checksum) changed.push(e.id)
283
- }
284
- for (const id of prevMap.keys()) {
285
- if (!nextIds.has(id)) removed.push(id)
286
- }
287
-
288
- return { changed, removed }
289
- }
1
+ // ─────────────────────────────────────────────────────────────
2
+ // dexin-content/collection.ts
3
+ //
4
+ // Collection declaration + source adaptation + discovery +
5
+ // batch compile orchestration (API-FREEZE §3.1–§3.4).
6
+ //
7
+ // The batch compileCollections() wraps the formal per-file
8
+ // core/compiler.ts compile() — it discovers files, builds
9
+ // CompileInput per file, calls compile, and writes PositiveArtifact
10
+ // docs + ContentIndex to the store.
11
+ //
12
+ // grep-zero: no business vocabulary appears anywhere in this
13
+ // module — including error messages.
14
+ // ─────────────────────────────────────────────────────────────
15
+
16
+ import { readdir, readFile } from 'node:fs/promises'
17
+ import { join } from 'node:path'
18
+ import { createHash } from 'node:crypto'
19
+ import type {
20
+ PositiveArtifact,
21
+ Identity,
22
+ DomainName,
23
+ Meta,
24
+ ParseError
25
+ } from '../types'
26
+ import type { DomainParserRegistry, CompileInput } from '../compiler/compiler'
27
+ import { compile } from '../compiler/compiler'
28
+ import { buildDocumentIdentity } from '../compiler/discovery'
29
+ import { splitFrontmatter, parseFrontmatter, validateSchema } from '../parser/frontmatter'
30
+ import type { ArtifactStore, ContentIndex, IndexEntry } from './store'
31
+
32
+ // ── Collection types (API-FREEZE §3.1) ──
33
+
34
+ export interface CollectionDefinition {
35
+ /** Collection name (appears in identity.collection + IndexEntry.collection). */
36
+ name: string
37
+ /** Directory under content root; .md files collected recursively. */
38
+ source: string
39
+ /** Domain tag; defaults to 'document' when omitted. */
40
+ domain?: DomainName | string
41
+ /** Frontmatter schema (reuses core Schema; validator-only, no meta projection). */
42
+ schema?: { required?: string[]; types?: Record<string, 'string' | 'number' | 'boolean'> }
43
+ }
44
+
45
+ export interface ResolvedCollection extends CollectionDefinition {
46
+ domain: string
47
+ sourceRoot: string
48
+ }
49
+
50
+ /** Declaration passthrough (declarative API semantic); resolution in resolveCollections. */
51
+ export function defineCollection (def: CollectionDefinition): CollectionDefinition {
52
+ return def
53
+ }
54
+
55
+ /** Resolve definitions: fill domain default + bind absolute sourceRoot. */
56
+ export function resolveCollections (
57
+ defs: CollectionDefinition[],
58
+ contentRoot: string
59
+ ): ResolvedCollection[] {
60
+ return defs.map(d => ({
61
+ ...d,
62
+ domain: d.domain ?? 'document',
63
+ sourceRoot: join(contentRoot, d.source)
64
+ }))
65
+ }
66
+
67
+ // ── SourceAdapter (API-FREEZE §3.2) ──
68
+
69
+ export interface SourceAdapter {
70
+ list(): Promise<string[]>
71
+ read(relPath: string): Promise<string>
72
+ }
73
+
74
+ /**
75
+ * Local filesystem source. Reads .md files recursively from root.
76
+ * CRLF protection (R-d): read() throws LINE_ENDING_CONTAMINATION
77
+ * on CRLF so the batch path enjoys the same guard as run-p0.
78
+ */
79
+ export function createLocalSource (root: string): SourceAdapter {
80
+ async function walk (dir: string, base: string): Promise<string[]> {
81
+ const entries = await readdir(dir, { withFileTypes: true })
82
+ const out: string[] = []
83
+ for (const e of entries) {
84
+ const rel = base ? `${base}/${e.name}` : e.name
85
+ if (e.isDirectory()) out.push(...await walk(join(dir, e.name), rel))
86
+ else if (e.isFile() && e.name.endsWith('.md')) out.push(rel)
87
+ }
88
+ return out.sort()
89
+ }
90
+ return {
91
+ list: () => walk(root, ''),
92
+ read: async (rel) => {
93
+ const raw = await readFile(join(root, rel), 'utf8')
94
+ const src = raw.charCodeAt(0) === 0xfeff ? raw.slice(1) : raw
95
+ if (src.includes('\r')) {
96
+ const err = new Error(
97
+ `[LINE_ENDING_CONTAMINATION] Source file '${rel}' contains CRLF line endings. ` +
98
+ `Normalize to LF before processing.`
99
+ ) as ParseError
100
+ err.code = 'LINE_ENDING_CONTAMINATION'
101
+ err.file = rel
102
+ throw err
103
+ }
104
+ return src
105
+ }
106
+ }
107
+ }
108
+
109
+ /** Virtual in-memory source (test / fixture use). */
110
+ export function createMemorySource (files: Record<string, string>): SourceAdapter {
111
+ return {
112
+ list: async () => Object.keys(files).sort(),
113
+ read: async (rel) => {
114
+ if (!(rel in files)) throw new Error(`[MemorySource] Not found: ${rel}`)
115
+ return files[rel]
116
+ }
117
+ }
118
+ }
119
+
120
+ // ── Discovery (API-FREEZE §3.3) ──
121
+
122
+ export interface ContentFile {
123
+ id: string
124
+ path: string
125
+ collection: string
126
+ file: string
127
+ frontmatter: string
128
+ body: string
129
+ }
130
+
131
+ /**
132
+ * Scan all collections via source. Files not belonging to any
133
+ * collection → throw (fail-fast, no silent drop).
134
+ *
135
+ * Identity derivation: document-domain path convention via
136
+ * buildDocumentIdentity (id = rel minus .md minus trailing /index,
137
+ * path = '/' + id, file = rel, collection = name). Non-document
138
+ * domains are not yet supported in the batch path (T5 lands the
139
+ * structured-ext-A domain discovery rules); a thrown error makes
140
+ * the gap explicit.
141
+ */
142
+ export async function discover (
143
+ source: SourceAdapter,
144
+ collections: ResolvedCollection[]
145
+ ): Promise<ContentFile[]> {
146
+ const byDir = new Map(collections.map(c => [c.source, c]))
147
+ const all = await source.list()
148
+ const files: ContentFile[] = []
149
+ for (const rel of all) {
150
+ const top = rel.split('/')[0]
151
+ const col = byDir.get(top)
152
+ if (!col) {
153
+ throw new Error(
154
+ `[discover] File '${rel}' belongs to no collection (top dir '${top}' undefined).`
155
+ )
156
+ }
157
+ if (col.domain !== 'document') {
158
+ throw new Error(
159
+ `[discover] Domain '${col.domain}' discovery not yet implemented in batch path; ` +
160
+ `only 'document' domain is supported here. Non-document domains require ` +
161
+ `explicit identity construction (see run-p0 FixtureSpec pattern).`
162
+ )
163
+ }
164
+ const identity = buildDocumentIdentity(rel, col.name)
165
+ const raw = await source.read(rel)
166
+ const { frontmatter, body } = splitFrontmatter(raw)
167
+ files.push({
168
+ id: identity.id,
169
+ path: identity.path,
170
+ collection: col.name,
171
+ file: identity.file,
172
+ frontmatter,
173
+ body
174
+ })
175
+ }
176
+ return files
177
+ }
178
+
179
+ // ── Batch compile orchestration (API-FREEZE §3.3) ──
180
+
181
+ function checksum (s: string): string {
182
+ return createHash('md5').update(s, 'utf-8').digest('hex').slice(0, 12)
183
+ }
184
+
185
+ /**
186
+ * Batch compile: discover → per-file compile → write docs + index to store.
187
+ *
188
+ * For each ContentFile:
189
+ * 1. Build CompileInput (fixture = file.id, identity = buildDocumentIdentity,
190
+ * source = raw, schema = col.schema)
191
+ * 2. Call formal compile(input, registry)
192
+ * 3. On positive → collect PositiveArtifact + IndexEntry
193
+ * On error → collect error message (fail-fast: throw after all files processed)
194
+ *
195
+ * Store receives only positive artifacts. compile failures are aggregated
196
+ * and thrown as a single Error (prototype-aligned behaviour).
197
+ */
198
+ export async function compileCollections (
199
+ source: SourceAdapter,
200
+ collections: ResolvedCollection[],
201
+ registry: DomainParserRegistry,
202
+ store: ArtifactStore
203
+ ): Promise<{ docs: PositiveArtifact[]; index: ContentIndex }> {
204
+ const files = await discover(source, collections)
205
+ const docs: PositiveArtifact[] = []
206
+ const entries: IndexEntry[] = []
207
+ const errors: string[] = []
208
+
209
+ for (const f of files) {
210
+ const col = collections.find(c => c.name === f.collection)!
211
+ const raw = await source.read(f.file)
212
+ const input: CompileInput = {
213
+ fixture: f.id,
214
+ domain: col.domain,
215
+ identity: buildDocumentIdentity(f.file, col.name),
216
+ source: raw,
217
+ file: f.file,
218
+ schema: col.schema
219
+ }
220
+ const result = compile(input, registry)
221
+ if (result.kind !== 'positive') {
222
+ errors.push(` ✗ ${f.file}: ${result.error?.message ?? 'unknown'}`)
223
+ continue
224
+ }
225
+ const artifact = result.artifact as PositiveArtifact
226
+ docs.push(artifact)
227
+ entries.push({
228
+ id: f.id,
229
+ path: f.path,
230
+ collection: f.collection,
231
+ domain: col.domain,
232
+ file: f.file,
233
+ meta: result.meta,
234
+ checksum: checksum(f.body)
235
+ })
236
+ }
237
+
238
+ if (errors.length > 0) {
239
+ throw new Error(`Compile failed (${errors.length} error(s)):\n${errors.join('\n')}`)
240
+ }
241
+
242
+ const index: ContentIndex = {
243
+ generator: 'dexin-content',
244
+ builtAt: new Date().toISOString(),
245
+ entries
246
+ }
247
+
248
+ await store.writeIndex(index)
249
+ for (const doc of docs) {
250
+ await store.writeDoc(doc)
251
+ }
252
+
253
+ return { docs, index }
254
+ }
255
+
256
+ // ── Incremental recompile (API-FREEZE §3.3, prototype L663 移植) ──
257
+
258
+ /**
259
+ * Detect changed/removed docs by checksum diff against the previous index.
260
+ *
261
+ * Wraps compileCollections (which writes index + all docs to store), then
262
+ * reports which ids changed (checksum differs from prev) or were removed
263
+ * (present in prev, absent in next). Used by the Dev flow (memory store)
264
+ * to drive incremental re-render signals.
265
+ *
266
+ * Adapts prototype's recompileChanged to the formal layer: prototype's
267
+ * `compile(source, collections, domains)` returned { docs, index } without
268
+ * writing to store; our compileCollections writes index + all docs to
269
+ * store AND returns { docs, index }. The changed/removed detection is
270
+ * preserved; for in-memory Dev stores writing all docs is acceptable
271
+ * (ephemeral). Per B1-SPEC T5.4 "最小形态", watcher + preview server
272
+ * are explicitly out of scope.
273
+ */
274
+ export async function recompileChanged (
275
+ source: SourceAdapter,
276
+ collections: ResolvedCollection[],
277
+ registry: DomainParserRegistry,
278
+ store: ArtifactStore
279
+ ): Promise<{ changed: string[]; removed: string[] }> {
280
+ const prev = await store.readIndex()
281
+ const prevMap = new Map((prev?.entries ?? []).map(e => [e.id, e]))
282
+
283
+ const { index } = await compileCollections(source, collections, registry, store)
284
+
285
+ const nextIds = new Set(index.entries.map(e => e.id))
286
+ const changed: string[] = []
287
+ const removed: string[] = []
288
+
289
+ for (const e of index.entries) {
290
+ if (prevMap.get(e.id)?.checksum !== e.checksum) changed.push(e.id)
291
+ }
292
+ for (const id of prevMap.keys()) {
293
+ if (!nextIds.has(id)) removed.push(id)
294
+ }
295
+
296
+ return { changed, removed }
297
+ }