@lokascript/framework 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/dist/aot/index.d.ts +1 -0
  2. package/dist/aot/index.d.ts.map +1 -1
  3. package/dist/aot/registry-bridge.d.ts +27 -0
  4. package/dist/aot/registry-bridge.d.ts.map +1 -0
  5. package/dist/api/create-dsl.d.ts.map +1 -1
  6. package/dist/api/dispatcher.d.ts +22 -0
  7. package/dist/api/dispatcher.d.ts.map +1 -1
  8. package/dist/api/domain-registry.d.ts +41 -0
  9. package/dist/api/domain-registry.d.ts.map +1 -1
  10. package/dist/api/index.js +1351 -21
  11. package/dist/api/index.js.map +1 -1
  12. package/dist/core/index.js +114 -10
  13. package/dist/core/index.js.map +1 -1
  14. package/dist/core/pattern-matching/index.js +74 -10
  15. package/dist/core/pattern-matching/index.js.map +1 -1
  16. package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -1
  17. package/dist/core/types.d.ts +80 -2
  18. package/dist/core/types.d.ts.map +1 -1
  19. package/dist/core/types.js +40 -0
  20. package/dist/core/types.js.map +1 -1
  21. package/dist/feedback/confidence-gate.d.ts +28 -0
  22. package/dist/feedback/confidence-gate.d.ts.map +1 -0
  23. package/dist/feedback/feedback-formatter.d.ts +21 -0
  24. package/dist/feedback/feedback-formatter.d.ts.map +1 -0
  25. package/dist/feedback/index.d.ts +11 -0
  26. package/dist/feedback/index.d.ts.map +1 -0
  27. package/dist/feedback/pattern-tracker.d.ts +47 -0
  28. package/dist/feedback/pattern-tracker.d.ts.map +1 -0
  29. package/dist/feedback/types.d.ts +113 -0
  30. package/dist/feedback/types.d.ts.map +1 -0
  31. package/dist/generation/index.js +2 -1
  32. package/dist/generation/index.js.map +1 -1
  33. package/dist/index.cjs +2685 -71
  34. package/dist/index.cjs.map +1 -1
  35. package/dist/index.d.ts +6 -2
  36. package/dist/index.d.ts.map +1 -1
  37. package/dist/index.js +2652 -71
  38. package/dist/index.js.map +1 -1
  39. package/dist/ir/explicit-parser.d.ts +31 -0
  40. package/dist/ir/explicit-parser.d.ts.map +1 -0
  41. package/dist/ir/explicit-renderer.d.ts +17 -0
  42. package/dist/ir/explicit-renderer.d.ts.map +1 -0
  43. package/dist/ir/from-interchange.d.ts +51 -0
  44. package/dist/ir/from-interchange.d.ts.map +1 -0
  45. package/dist/ir/index.d.ts +29 -0
  46. package/dist/ir/index.d.ts.map +1 -0
  47. package/dist/ir/index.js +1557 -0
  48. package/dist/ir/index.js.map +1 -0
  49. package/dist/ir/json-schema.d.ts +40 -0
  50. package/dist/ir/json-schema.d.ts.map +1 -0
  51. package/dist/ir/protocol-json.d.ts +73 -0
  52. package/dist/ir/protocol-json.d.ts.map +1 -0
  53. package/dist/ir/references.d.ts +19 -0
  54. package/dist/ir/references.d.ts.map +1 -0
  55. package/dist/ir/types.d.ts +141 -0
  56. package/dist/ir/types.d.ts.map +1 -0
  57. package/dist/parsing/index.js +74 -10
  58. package/dist/parsing/index.js.map +1 -1
  59. package/dist/prompts/index.d.ts +9 -0
  60. package/dist/prompts/index.d.ts.map +1 -0
  61. package/dist/prompts/prompt-generator.d.ts +49 -0
  62. package/dist/prompts/prompt-generator.d.ts.map +1 -0
  63. package/dist/prompts/types.d.ts +64 -0
  64. package/dist/prompts/types.d.ts.map +1 -0
  65. package/dist/schema/command-schema.d.ts +17 -0
  66. package/dist/schema/command-schema.d.ts.map +1 -1
  67. package/dist/schema/index.js.map +1 -1
  68. package/dist/testing/index.js +3 -0
  69. package/dist/testing/index.js.map +1 -1
  70. package/dist/training/index.d.ts +11 -0
  71. package/dist/training/index.d.ts.map +1 -0
  72. package/dist/training/jsonl-writer.d.ts +42 -0
  73. package/dist/training/jsonl-writer.d.ts.map +1 -0
  74. package/dist/training/schema-synthesizer.d.ts +20 -0
  75. package/dist/training/schema-synthesizer.d.ts.map +1 -0
  76. package/dist/training/types.d.ts +73 -0
  77. package/dist/training/types.d.ts.map +1 -0
  78. package/package.json +5 -1
  79. package/src/aot/index.ts +1 -0
  80. package/src/aot/registry-bridge.test.ts +104 -0
  81. package/src/aot/registry-bridge.ts +66 -0
  82. package/src/api/create-dsl.test.ts +306 -0
  83. package/src/api/create-dsl.ts +21 -0
  84. package/src/api/dispatcher.test.ts +66 -0
  85. package/src/api/dispatcher.ts +77 -9
  86. package/src/api/domain-registry.ts +120 -4
  87. package/src/core/pattern-matching/pattern-matcher.test.ts +131 -0
  88. package/src/core/pattern-matching/pattern-matcher.ts +102 -19
  89. package/src/core/types.ts +145 -2
  90. package/src/feedback/confidence-gate.test.ts +111 -0
  91. package/src/feedback/confidence-gate.ts +70 -0
  92. package/src/feedback/feedback-formatter.test.ts +198 -0
  93. package/src/feedback/feedback-formatter.ts +298 -0
  94. package/src/feedback/index.ts +24 -0
  95. package/src/feedback/pattern-tracker.test.ts +198 -0
  96. package/src/feedback/pattern-tracker.ts +140 -0
  97. package/src/feedback/types.ts +149 -0
  98. package/src/generation/pattern-generator.ts +2 -2
  99. package/src/index.ts +22 -0
  100. package/src/integration/llm-lse-integration.test.ts +457 -0
  101. package/src/ir/explicit-parser.test.ts +306 -0
  102. package/src/ir/explicit-parser.ts +419 -0
  103. package/src/ir/explicit-renderer.test.ts +144 -0
  104. package/src/ir/explicit-renderer.ts +146 -0
  105. package/src/ir/from-interchange.test.ts +853 -0
  106. package/src/ir/from-interchange.ts +594 -0
  107. package/src/ir/index.ts +66 -0
  108. package/src/ir/json-schema.test.ts +281 -0
  109. package/src/ir/json-schema.ts +299 -0
  110. package/src/ir/protocol-json.test.ts +1185 -0
  111. package/src/ir/protocol-json.ts +745 -0
  112. package/src/ir/references.test.ts +38 -0
  113. package/src/ir/references.ts +33 -0
  114. package/src/ir/round-trip.test.ts +557 -0
  115. package/src/ir/types.ts +181 -0
  116. package/src/prompts/index.ts +15 -0
  117. package/src/prompts/prompt-generator.test.ts +386 -0
  118. package/src/prompts/prompt-generator.ts +531 -0
  119. package/src/prompts/types.ts +89 -0
  120. package/src/schema/command-schema.ts +19 -0
  121. package/src/training/index.ts +13 -0
  122. package/src/training/jsonl-writer.test.ts +138 -0
  123. package/src/training/jsonl-writer.ts +67 -0
  124. package/src/training/schema-synthesizer.test.ts +250 -0
  125. package/src/training/schema-synthesizer.ts +306 -0
  126. package/src/training/types.ts +103 -0
@@ -0,0 +1,531 @@
1
+ /**
2
+ * LLM Prompt Generator
3
+ *
4
+ * Auto-generates LLM system prompts from CommandSchema[]. Given a set of
5
+ * domain schemas, produces a structured markdown prompt that teaches LLMs
6
+ * to output valid LSE bracket syntax and/or LLM-simplified JSON.
7
+ *
8
+ * The generated prompt includes:
9
+ * - Protocol overview (LSE syntax summary)
10
+ * - Value type reference
11
+ * - Per-command documentation with role descriptions
12
+ * - Synthetic examples in both formats
13
+ * - Output format specification
14
+ * - Error recovery instructions
15
+ */
16
+
17
+ import type { CommandSchema, RoleSpec } from '../schema/command-schema';
18
+ import type { SemanticJSON, SemanticJSONValue } from '../ir/types';
19
+ import { renderExplicit } from '../ir/explicit-renderer';
20
+ import { createCommandNode } from '../core/types';
21
+ import type {
22
+ PromptGeneratorConfig,
23
+ GeneratedPrompt,
24
+ PromptSection,
25
+ PromptMetadata,
26
+ } from './types';
27
+
28
+ // =============================================================================
29
+ // Public API
30
+ // =============================================================================
31
+
32
+ /**
33
+ * Generate an LLM system prompt from domain command schemas.
34
+ *
35
+ * @example
36
+ * ```typescript
37
+ * import { generatePrompt } from '@lokascript/framework';
38
+ * import { allSchemas } from '@lokascript/domain-flow';
39
+ *
40
+ * const prompt = generatePrompt({
41
+ * domain: 'flow',
42
+ * description: 'Reactive data flow pipelines',
43
+ * schemas: allSchemas,
44
+ * });
45
+ *
46
+ * console.log(prompt.text); // Full markdown system prompt
47
+ * ```
48
+ */
49
+ export function generatePrompt(config: PromptGeneratorConfig): GeneratedPrompt {
50
+ const outputFormat = config.outputFormat ?? 'both';
51
+ const examplesPerCommand = config.examplesPerCommand ?? 2;
52
+
53
+ const sections: PromptSection[] = [
54
+ buildProtocolSection(),
55
+ buildValueTypeSection(),
56
+ buildCommandsSection(config.schemas, outputFormat, examplesPerCommand),
57
+ buildOutputFormatSection(outputFormat),
58
+ buildErrorRecoverySection(),
59
+ ];
60
+
61
+ // Apply token budget if specified
62
+ let finalSections = sections;
63
+ if (config.maxTokens && config.maxTokens > 0) {
64
+ finalSections = truncateToTokenBudget(sections, config.maxTokens);
65
+ }
66
+
67
+ const text = finalSections.map(s => `## ${s.title}\n\n${s.content}`).join('\n\n---\n\n');
68
+
69
+ const totalRoles = config.schemas.reduce((sum, s) => sum + s.roles.length, 0);
70
+ const approximateTokens = estimateTokens(text);
71
+
72
+ const metadata: PromptMetadata = {
73
+ domain: config.domain,
74
+ commandCount: config.schemas.length,
75
+ roleCount: totalRoles,
76
+ approximateTokens,
77
+ };
78
+
79
+ return { text, sections: finalSections, metadata };
80
+ }
81
+
82
+ /**
83
+ * Generate synthetic LSE examples from a single command schema.
84
+ * Used by both the prompt generator and training data synthesizer.
85
+ */
86
+ export function generateExamples(
87
+ schema: CommandSchema,
88
+ count: number = 2
89
+ ): Array<{ explicit: string; json: SemanticJSON }> {
90
+ const combos = generateRoleCombinations(schema);
91
+ const examples: Array<{ explicit: string; json: SemanticJSON }> = [];
92
+
93
+ for (let i = 0; i < Math.min(count, combos.length); i++) {
94
+ const combo = combos[i];
95
+ const roles = new Map<string, { type: string; value: string | number | boolean }>();
96
+
97
+ for (const role of combo) {
98
+ const sample = sampleValue(role);
99
+ roles.set(role.role, sample);
100
+ }
101
+
102
+ // Build SemanticNode for rendering to bracket syntax
103
+ const nodeRoles = new Map<
104
+ string,
105
+ {
106
+ type: string;
107
+ value: string | number | boolean;
108
+ raw?: string;
109
+ dataType?: string;
110
+ selectorKind?: string;
111
+ }
112
+ >();
113
+ for (const [roleName, sample] of roles) {
114
+ nodeRoles.set(roleName, toSemanticValue(sample));
115
+ }
116
+
117
+ const node = createCommandNode(schema.action, nodeRoles as never);
118
+ const explicit = renderExplicit(node);
119
+
120
+ // Build JSON format
121
+ const jsonRoles: Record<string, SemanticJSONValue> = {};
122
+ for (const [roleName, sample] of roles) {
123
+ jsonRoles[roleName] = sample as SemanticJSONValue;
124
+ }
125
+ const json: SemanticJSON = { action: schema.action, roles: jsonRoles };
126
+
127
+ examples.push({ explicit, json });
128
+ }
129
+
130
+ return examples;
131
+ }
132
+
133
+ /**
134
+ * Generate a condensed LSE protocol reference (for MCP resources).
135
+ */
136
+ export function generateProtocolReference(): string {
137
+ const protocol = buildProtocolSection();
138
+ const valueTypes = buildValueTypeSection();
139
+ return `# LokaScript Explicit Syntax (LSE) Quick Reference\n\n${protocol.content}\n\n${valueTypes.content}`;
140
+ }
141
+
142
+ // =============================================================================
143
+ // Section Builders
144
+ // =============================================================================
145
+
146
+ function buildProtocolSection(): PromptSection {
147
+ const content = `LokaScript Explicit Syntax (LSE) is a bracket-based format for imperative commands.
148
+
149
+ ### Syntax
150
+
151
+ \`\`\`
152
+ [command role1:value1 role2:value2 +flag1 ~flag2]
153
+ \`\`\`
154
+
155
+ - **Command**: The first token inside brackets (lowercased)
156
+ - **Role pair**: \`name:value\` — a named semantic role with a typed value (no space around colon)
157
+ - **Enabled flag**: \`+name\` — boolean attribute present
158
+ - **Disabled flag**: \`~name\` — boolean attribute negated
159
+ - **Nested body**: \`body:[command ...]\` — a bracket command inside a role value
160
+
161
+ ### Rules
162
+
163
+ 1. Commands are always lowercased
164
+ 2. Role names preserve their original case
165
+ 3. No spaces around the colon in role:value pairs
166
+ 4. Strings with spaces must be quoted: \`patient:"hello world"\`
167
+ 5. Selectors start with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`
168
+ 6. Output must be valid bracket syntax: \`[action role:value ...]\``;
169
+
170
+ return {
171
+ id: 'protocol',
172
+ title: 'LSE Protocol',
173
+ content,
174
+ approximateTokens: estimateTokens(content),
175
+ };
176
+ }
177
+
178
+ function buildValueTypeSection(): PromptSection {
179
+ const content = `Values in LSE are classified by their syntactic form (first match wins):
180
+
181
+ | Type | Syntax | Example |
182
+ |------|--------|---------|
183
+ | Selector | Starts with \`#\` \`.\` \`[\` \`@\` \`*\` | \`#button\`, \`.active\`, \`[data-id]\` |
184
+ | String | Quoted with \`"\` or \`'\` | \`"hello world"\`, \`'json'\` |
185
+ | Boolean | Exact: \`true\` / \`false\` | \`visible:true\` |
186
+ | Reference | Built-in name | \`me\`, \`you\`, \`it\`, \`result\`, \`event\`, \`target\`, \`body\` |
187
+ | Duration | Number + suffix | \`500ms\`, \`2s\`, \`1m\`, \`1h\` |
188
+ | Number | Digits with optional decimal | \`5\`, \`3.14\`, \`-1\` |
189
+ | Plain | Fallback (any non-whitespace) | \`/api/users\`, \`json\`, \`production\` |
190
+
191
+ **Important**: Classification is by prefix, not by intent. \`#true\` is a selector (not boolean). \`event\` as a value is a reference (not plain text).`;
192
+
193
+ return {
194
+ id: 'value-types',
195
+ title: 'Value Types',
196
+ content,
197
+ approximateTokens: estimateTokens(content),
198
+ };
199
+ }
200
+
201
+ function buildCommandsSection(
202
+ schemas: readonly CommandSchema[],
203
+ outputFormat: 'explicit' | 'json' | 'both',
204
+ examplesPerCommand: number
205
+ ): PromptSection {
206
+ const parts: string[] = [];
207
+
208
+ for (const schema of schemas) {
209
+ parts.push(formatCommandDoc(schema, outputFormat, examplesPerCommand));
210
+ }
211
+
212
+ const content = parts.join('\n\n');
213
+
214
+ return {
215
+ id: 'commands',
216
+ title: 'Available Commands',
217
+ content,
218
+ approximateTokens: estimateTokens(content),
219
+ };
220
+ }
221
+
222
+ function buildOutputFormatSection(outputFormat: 'explicit' | 'json' | 'both'): PromptSection {
223
+ let content: string;
224
+
225
+ if (outputFormat === 'explicit') {
226
+ content = `Output valid LSE bracket syntax. Each command must be wrapped in brackets:
227
+
228
+ \`\`\`
229
+ [command role:value ...]
230
+ \`\`\`
231
+
232
+ Do NOT output JSON. Only use bracket syntax.`;
233
+ } else if (outputFormat === 'json') {
234
+ content = `Output valid JSON in the LLM-simplified format:
235
+
236
+ \`\`\`json
237
+ {
238
+ "action": "command-name",
239
+ "roles": {
240
+ "roleName": { "type": "valueType", "value": "theValue" }
241
+ }
242
+ }
243
+ \`\`\`
244
+
245
+ Valid value types: \`selector\`, \`literal\`, \`reference\`, \`expression\`.
246
+ Do NOT output bracket syntax. Only use JSON.`;
247
+ } else {
248
+ content = `You may output EITHER format:
249
+
250
+ **Bracket syntax** (preferred for single commands):
251
+ \`\`\`
252
+ [command role:value ...]
253
+ \`\`\`
254
+
255
+ **JSON format** (preferred when structured data is needed):
256
+ \`\`\`json
257
+ {
258
+ "action": "command-name",
259
+ "roles": {
260
+ "roleName": { "type": "valueType", "value": "theValue" }
261
+ }
262
+ }
263
+ \`\`\`
264
+
265
+ Both formats are equally valid. Use whichever is more natural for the context.`;
266
+ }
267
+
268
+ return {
269
+ id: 'output-format',
270
+ title: 'Output Format',
271
+ content,
272
+ approximateTokens: estimateTokens(content),
273
+ };
274
+ }
275
+
276
+ function buildErrorRecoverySection(): PromptSection {
277
+ const content = `If you're unsure about a command or role:
278
+
279
+ 1. **Unknown command**: Use the closest matching command from the Available Commands section. Do not invent commands.
280
+ 2. **Unknown role**: Use only the roles listed for each command. Do not add roles not in the schema.
281
+ 3. **Ambiguous value type**: When in doubt, use \`expression\` type (it's the most permissive).
282
+ 4. **Missing required role**: Always include all required roles. Check the Required/Optional labels.
283
+ 5. **Selector vs plain value**: If a value starts with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`, it's a selector. URLs like \`/api/data\` are plain values.
284
+
285
+ When generating LSE, prefer explicit role labeling over positional guessing. Every value should have a role name.`;
286
+
287
+ return {
288
+ id: 'error-recovery',
289
+ title: 'Error Recovery',
290
+ content,
291
+ approximateTokens: estimateTokens(content),
292
+ };
293
+ }
294
+
295
+ // =============================================================================
296
+ // Command Documentation
297
+ // =============================================================================
298
+
299
+ function formatCommandDoc(
300
+ schema: CommandSchema,
301
+ outputFormat: 'explicit' | 'json' | 'both',
302
+ examplesPerCommand: number
303
+ ): string {
304
+ const lines: string[] = [];
305
+
306
+ // Header
307
+ lines.push(`### \`${schema.action}\` — ${schema.description}`);
308
+ lines.push('');
309
+
310
+ // Roles
311
+ const required = schema.roles.filter(r => r.required);
312
+ const optional = schema.roles.filter(r => !r.required);
313
+
314
+ if (required.length > 0) {
315
+ lines.push('**Required roles:**');
316
+ for (const role of required) {
317
+ lines.push(`- \`${role.role}\`: ${role.description} (type: ${formatExpectedTypes(role)})`);
318
+ }
319
+ }
320
+
321
+ if (optional.length > 0) {
322
+ lines.push('**Optional roles:**');
323
+ for (const role of optional) {
324
+ lines.push(`- \`${role.role}\`: ${role.description} (type: ${formatExpectedTypes(role)})`);
325
+ }
326
+ }
327
+
328
+ // Examples
329
+ const examples = generateExamples(schema, examplesPerCommand);
330
+ if (examples.length > 0) {
331
+ lines.push('');
332
+ if (outputFormat === 'explicit' || outputFormat === 'both') {
333
+ lines.push('**Bracket syntax:**');
334
+ for (const ex of examples) {
335
+ lines.push(`\`\`\`\n${ex.explicit}\n\`\`\``);
336
+ }
337
+ }
338
+ if (outputFormat === 'json' || outputFormat === 'both') {
339
+ lines.push('**JSON format:**');
340
+ for (const ex of examples) {
341
+ lines.push(`\`\`\`json\n${JSON.stringify(ex.json, null, 2)}\n\`\`\``);
342
+ }
343
+ }
344
+ }
345
+
346
+ return lines.join('\n');
347
+ }
348
+
349
+ function formatExpectedTypes(role: RoleSpec): string {
350
+ return role.expectedTypes.join(' | ');
351
+ }
352
+
353
+ // =============================================================================
354
+ // Example Generation Helpers
355
+ // =============================================================================
356
+
357
+ /**
358
+ * Generate role combinations for example synthesis.
359
+ * Returns arrays of RoleSpec[] representing different valid input combinations:
360
+ * 1. Required roles only
361
+ * 2. Required + each optional role individually
362
+ * 3. Required + all optional roles (if different from #1 and #2)
363
+ */
364
+ function generateRoleCombinations(schema: CommandSchema): RoleSpec[][] {
365
+ const required = schema.roles.filter(r => r.required);
366
+ const optional = schema.roles.filter(r => !r.required);
367
+
368
+ const combos: RoleSpec[][] = [];
369
+
370
+ // 1. Required only
371
+ if (required.length > 0) {
372
+ combos.push(required);
373
+ }
374
+
375
+ // 2. Required + each optional individually
376
+ for (const opt of optional) {
377
+ combos.push([...required, opt]);
378
+ }
379
+
380
+ // 3. Required + all optional (if there are 2+ optional roles)
381
+ if (optional.length >= 2) {
382
+ combos.push([...required, ...optional]);
383
+ }
384
+
385
+ // Fallback: if schema has no required roles, use all roles
386
+ if (combos.length === 0) {
387
+ combos.push(schema.roles.slice());
388
+ }
389
+
390
+ return combos;
391
+ }
392
+
393
+ /**
394
+ * Sample a value for a role based on its expected types and name.
395
+ */
396
+ function sampleValue(role: RoleSpec): { type: string; value: string | number | boolean } {
397
+ const primaryType = role.expectedTypes[0] || 'expression';
398
+
399
+ // Use role name as a heuristic for realistic values
400
+ switch (primaryType) {
401
+ case 'selector':
402
+ return sampleSelector(role.role);
403
+ case 'literal':
404
+ return sampleLiteral(role.role);
405
+ case 'reference':
406
+ return { type: 'reference', value: 'me' };
407
+ case 'expression':
408
+ return sampleExpression(role.role);
409
+ case 'flag':
410
+ return { type: 'flag', value: true };
411
+ default:
412
+ return { type: 'literal', value: 'example' };
413
+ }
414
+ }
415
+
416
+ function sampleSelector(roleName: string): { type: string; value: string } {
417
+ const selectors: Record<string, string> = {
418
+ patient: '.active',
419
+ destination: '#output',
420
+ source: '#input',
421
+ target: '#target',
422
+ };
423
+ return { type: 'selector', value: selectors[roleName] || `#${roleName}` };
424
+ }
425
+
426
+ function sampleLiteral(roleName: string): { type: string; value: string | number } {
427
+ const literals: Record<string, string | number> = {
428
+ duration: '30s',
429
+ interval: '5s',
430
+ delay: '500ms',
431
+ timeout: '10s',
432
+ quantity: 5,
433
+ limit: 10,
434
+ style: 'json',
435
+ manner: 'rolling',
436
+ format: 'html',
437
+ };
438
+ return { type: 'literal', value: literals[roleName] || roleName };
439
+ }
440
+
441
+ function sampleExpression(roleName: string): { type: string; value: string } {
442
+ const expressions: Record<string, string> = {
443
+ source: '/api/data',
444
+ destination: '#output',
445
+ patient: '.active',
446
+ instrument: 'toUpperCase',
447
+ condition: 'age > 18',
448
+ style: 'json',
449
+ duration: '5s',
450
+ url: '/api/users',
451
+ };
452
+ return { type: 'expression', value: expressions[roleName] || `/api/${roleName}` };
453
+ }
454
+
455
+ /**
456
+ * Convert a sampled value into a SemanticValue for renderExplicit().
457
+ */
458
+ function toSemanticValue(sample: { type: string; value: string | number | boolean }): {
459
+ type: string;
460
+ value: string | number | boolean;
461
+ raw?: string;
462
+ dataType?: string;
463
+ selectorKind?: string;
464
+ } {
465
+ switch (sample.type) {
466
+ case 'selector':
467
+ return {
468
+ type: 'selector',
469
+ value: String(sample.value),
470
+ selectorKind: detectSelectorKind(String(sample.value)),
471
+ };
472
+ case 'literal':
473
+ return {
474
+ type: 'literal',
475
+ value: sample.value,
476
+ dataType:
477
+ typeof sample.value === 'number'
478
+ ? 'number'
479
+ : typeof sample.value === 'boolean'
480
+ ? 'boolean'
481
+ : 'string',
482
+ };
483
+ case 'reference':
484
+ return { type: 'reference', value: String(sample.value) };
485
+ case 'expression':
486
+ return { type: 'expression', value: String(sample.value), raw: String(sample.value) };
487
+ case 'flag':
488
+ return { type: 'flag', value: sample.value };
489
+ default:
490
+ return { type: 'literal', value: String(sample.value), dataType: 'string' };
491
+ }
492
+ }
493
+
494
+ function detectSelectorKind(selector: string): string {
495
+ if (selector.startsWith('#')) return 'id';
496
+ if (selector.startsWith('.')) return 'class';
497
+ if (selector.startsWith('[')) return 'attribute';
498
+ return 'complex';
499
+ }
500
+
501
+ // =============================================================================
502
+ // Token Budget
503
+ // =============================================================================
504
+
505
+ function estimateTokens(text: string): number {
506
+ return Math.ceil(text.length / 4);
507
+ }
508
+
509
+ function truncateToTokenBudget(sections: PromptSection[], maxTokens: number): PromptSection[] {
510
+ const result: PromptSection[] = [];
511
+ let budget = maxTokens;
512
+
513
+ for (const section of sections) {
514
+ if (section.approximateTokens <= budget) {
515
+ result.push(section);
516
+ budget -= section.approximateTokens;
517
+ } else {
518
+ // Truncate the section to fit remaining budget (always include at least something)
519
+ const charBudget = Math.max(budget * 4, 40);
520
+ const truncatedContent = section.content.slice(0, charBudget) + '\n\n*(truncated)*';
521
+ result.push({
522
+ ...section,
523
+ content: truncatedContent,
524
+ approximateTokens: estimateTokens(truncatedContent),
525
+ });
526
+ break;
527
+ }
528
+ }
529
+
530
+ return result;
531
+ }
@@ -0,0 +1,89 @@
1
+ /**
2
+ * LLM Prompt Generation Types
3
+ *
4
+ * Types for auto-generating LLM system prompts from CommandSchema[].
5
+ * The prompt generator reads domain schemas and produces markdown
6
+ * instructions that teach LLMs to output valid LSE bracket syntax
7
+ * and/or LLM-simplified JSON.
8
+ */
9
+
10
+ import type { CommandSchema } from '../schema/command-schema';
11
+
12
+ // =============================================================================
13
+ // Configuration
14
+ // =============================================================================
15
+
16
+ /**
17
+ * Configuration for generating an LLM system prompt from domain schemas.
18
+ */
19
+ export interface PromptGeneratorConfig {
20
+ /** Domain identifier (e.g., 'flow', 'sql', 'bdd') */
21
+ readonly domain: string;
22
+
23
+ /** Human-readable domain description */
24
+ readonly description: string;
25
+
26
+ /** Command schemas to document */
27
+ readonly schemas: readonly CommandSchema[];
28
+
29
+ /** What output format to teach the LLM. Default: 'both' */
30
+ readonly outputFormat?: 'explicit' | 'json' | 'both';
31
+
32
+ /** Number of examples to generate per command. Default: 2 */
33
+ readonly examplesPerCommand?: number;
34
+
35
+ /** Approximate token budget for truncation. Default: no limit */
36
+ readonly maxTokens?: number;
37
+ }
38
+
39
+ // =============================================================================
40
+ // Output
41
+ // =============================================================================
42
+
43
+ /**
44
+ * Section of a generated prompt, for partial inclusion.
45
+ */
46
+ export interface PromptSection {
47
+ /** Section identifier */
48
+ readonly id: string;
49
+
50
+ /** Human-readable title */
51
+ readonly title: string;
52
+
53
+ /** Markdown content */
54
+ readonly content: string;
55
+
56
+ /** Approximate token count (chars / 4 heuristic) */
57
+ readonly approximateTokens: number;
58
+ }
59
+
60
+ /**
61
+ * A complete generated system prompt with metadata.
62
+ */
63
+ export interface GeneratedPrompt {
64
+ /** Full system prompt as markdown */
65
+ readonly text: string;
66
+
67
+ /** Individual sections for partial inclusion */
68
+ readonly sections: PromptSection[];
69
+
70
+ /** Metadata about the generated prompt */
71
+ readonly metadata: PromptMetadata;
72
+ }
73
+
74
+ /**
75
+ * Metadata about a generated prompt.
76
+ */
77
+ export interface PromptMetadata {
78
+ /** Domain the prompt was generated for */
79
+ readonly domain: string;
80
+
81
+ /** Number of commands documented */
82
+ readonly commandCount: number;
83
+
84
+ /** Total number of roles across all commands */
85
+ readonly roleCount: number;
86
+
87
+ /** Approximate token count (chars / 4 heuristic) */
88
+ readonly approximateTokens: number;
89
+ }
@@ -73,6 +73,15 @@ export interface RoleSpec {
73
73
  */
74
74
  readonly renderOverride?: Record<string, string>;
75
75
 
76
+ /**
77
+ * Override the marker position for this role in the generated pattern.
78
+ * 'before' = marker precedes the role value (preposition: "set X").
79
+ * 'after' = marker follows the role value (postposition: "X から").
80
+ * If not set, falls back to the language profile's roleMarkers position,
81
+ * then to the word-order default (SOV='after', else='before').
82
+ */
83
+ readonly markerPosition?: 'before' | 'after';
84
+
76
85
  /**
77
86
  * When true, this role captures all remaining tokens until the next
78
87
  * recognized marker keyword or end of input, joining their values
@@ -82,6 +91,16 @@ export interface RoleSpec {
82
91
  * by a marked role. Default: false.
83
92
  */
84
93
  readonly greedy?: boolean;
94
+
95
+ /**
96
+ * Restricts which selector subtypes are valid for this role (v1.2).
97
+ * Only meaningful when expectedTypes includes 'selector'.
98
+ * If omitted, all selector kinds are accepted.
99
+ *
100
+ * Example: `selectorKinds: ['class', 'attribute']` means only `.class`
101
+ * and `[attr]` selectors are valid, not `#id` or `*wildcard`.
102
+ */
103
+ readonly selectorKinds?: ReadonlyArray<'id' | 'class' | 'attribute' | 'element' | 'complex'>;
85
104
  }
86
105
 
87
106
  /**
@@ -0,0 +1,13 @@
1
+ /**
2
+ * Training Data Generation
3
+ *
4
+ * Generates (natural_language, LSE) pairs from domain schemas for
5
+ * fine-tuning or few-shot prompting LLMs. Exports as JSONL.
6
+ */
7
+
8
+ export type { TrainingPair, SynthesisConfig, SynthesisResult, SynthesisMetadata } from './types';
9
+
10
+ export { synthesizeFromSchemas } from './schema-synthesizer';
11
+
12
+ export { toJSONL, toJSONLRow, parseJSONL } from './jsonl-writer';
13
+ export type { JSONLRow } from './jsonl-writer';