@lokascript/framework 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/dist/api/index.js +111 -24
  2. package/dist/api/index.js.map +1 -1
  3. package/dist/core/index.js +133 -0
  4. package/dist/core/index.js.map +1 -1
  5. package/dist/core/pattern-matching/index.js.map +1 -1
  6. package/dist/core/tokenization/index.js +133 -0
  7. package/dist/core/tokenization/index.js.map +1 -1
  8. package/dist/core/tokenization/morphology/base-normalizer.d.ts +94 -0
  9. package/dist/core/tokenization/morphology/base-normalizer.d.ts.map +1 -0
  10. package/dist/core/tokenization/morphology/index.d.ts +3 -1
  11. package/dist/core/tokenization/morphology/index.d.ts.map +1 -1
  12. package/dist/core/types.d.ts +3 -2
  13. package/dist/core/types.d.ts.map +1 -1
  14. package/dist/core/types.js.map +1 -1
  15. package/dist/generation/index.js.map +1 -1
  16. package/dist/index.cjs +564 -46
  17. package/dist/index.cjs.map +1 -1
  18. package/dist/index.d.ts +2 -0
  19. package/dist/index.d.ts.map +1 -1
  20. package/dist/index.js +556 -46
  21. package/dist/index.js.map +1 -1
  22. package/dist/ir/explicit-parser.d.ts +62 -7
  23. package/dist/ir/explicit-parser.d.ts.map +1 -1
  24. package/dist/ir/explicit-renderer.d.ts +18 -1
  25. package/dist/ir/explicit-renderer.d.ts.map +1 -1
  26. package/dist/ir/index.d.ts +4 -2
  27. package/dist/ir/index.d.ts.map +1 -1
  28. package/dist/ir/index.js +423 -46
  29. package/dist/ir/index.js.map +1 -1
  30. package/dist/ir/protocol-json.d.ts.map +1 -1
  31. package/dist/ir/to-runtime-ast.d.ts +53 -0
  32. package/dist/ir/to-runtime-ast.d.ts.map +1 -0
  33. package/dist/ir/types.d.ts +14 -4
  34. package/dist/ir/types.d.ts.map +1 -1
  35. package/dist/parsing/index.js.map +1 -1
  36. package/dist/testing/index.js +8671 -8261
  37. package/dist/testing/index.js.map +1 -1
  38. package/package.json +1 -1
  39. package/src/core/tokenization/morphology/base-normalizer.ts +249 -0
  40. package/src/core/tokenization/morphology/index.ts +3 -1
  41. package/src/core/types.ts +4 -2
  42. package/src/index.ts +7 -0
  43. package/src/ir/conformance.test.ts +120 -0
  44. package/src/ir/explicit-parser.test.ts +307 -1
  45. package/src/ir/explicit-parser.ts +403 -29
  46. package/src/ir/explicit-renderer.test.ts +87 -2
  47. package/src/ir/explicit-renderer.ts +57 -5
  48. package/src/ir/index.ts +13 -2
  49. package/src/ir/protocol-json.test.ts +4 -2
  50. package/src/ir/protocol-json.ts +30 -24
  51. package/src/ir/to-runtime-ast.ts +203 -0
  52. package/src/ir/types.ts +14 -4
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/framework",
3
- "version": "2.1.0",
3
+ "version": "2.2.0",
4
4
  "description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
5
5
  "type": "module",
6
6
  "main": "dist/index.cjs",
@@ -0,0 +1,249 @@
1
+ /**
2
+ * BaseMorphologicalNormalizer — Shared base class for language normalizers.
3
+ *
4
+ * Provides the common suffix/prefix stripping loop, reflexive verb handling,
5
+ * and normalize() pipeline. Language-specific normalizers extend this class
6
+ * and provide their conjugation rules.
7
+ *
8
+ * Phase 3.1 of parser-ecosystem-plan-v3.
9
+ */
10
+
11
+ import type {
12
+ MorphologicalNormalizer,
13
+ NormalizationResult,
14
+ ConjugationType,
15
+ SuffixRule,
16
+ PrefixRule,
17
+ } from './types';
18
+ import { noChange, normalized } from './types';
19
+
20
+ /**
21
+ * Conjugation ending rule for verb classes (Romance languages etc.)
22
+ * Broader than SuffixRule — includes the replacement stem (e.g., strip -ando, add -ar).
23
+ */
24
+ export interface ConjugationEnding {
25
+ readonly ending: string;
26
+ readonly stem: string;
27
+ readonly confidence: number;
28
+ readonly type: ConjugationType;
29
+ }
30
+
31
+ /**
32
+ * Configuration for BaseMorphologicalNormalizer.
33
+ * Subclasses provide this in their constructor.
34
+ */
35
+ export interface NormalizerConfig {
36
+ /** Language code */
37
+ readonly language: string;
38
+
39
+ /** Minimum word length to attempt normalization */
40
+ readonly minWordLength?: number;
41
+
42
+ /** Minimum stem length after stripping (default: 2) */
43
+ readonly minStemLength?: number;
44
+
45
+ /** Conjugation endings sorted longest-first */
46
+ readonly endings?: readonly ConjugationEnding[];
47
+
48
+ /** Suffix rules (for SuffixRule-style normalizers) */
49
+ readonly suffixRules?: readonly SuffixRule[];
50
+
51
+ /** Prefix rules */
52
+ readonly prefixRules?: readonly PrefixRule[];
53
+
54
+ /** Reflexive suffixes (for Romance languages) */
55
+ readonly reflexiveSuffixes?: readonly string[];
56
+
57
+ /** Infinitive endings (for checking if already normalized) */
58
+ readonly infinitiveEndings?: readonly string[];
59
+ }
60
+
61
+ /**
62
+ * Abstract base class for morphological normalizers.
63
+ *
64
+ * Subclasses must implement `isNormalizable()` and can override any
65
+ * normalization step. The default `normalize()` pipeline is:
66
+ *
67
+ * 1. Check if already in dictionary form → noChange
68
+ * 2. Try reflexive normalization (if reflexiveSuffixes configured)
69
+ * 3. Try conjugation endings (if endings configured)
70
+ * 4. Try suffix rules (if suffixRules configured)
71
+ * 5. Try prefix rules (if prefixRules configured)
72
+ * 6. Return noChange
73
+ */
74
+ export abstract class BaseMorphologicalNormalizer implements MorphologicalNormalizer {
75
+ readonly language: string;
76
+ protected readonly config: NormalizerConfig;
77
+
78
+ constructor(config: NormalizerConfig) {
79
+ this.language = config.language;
80
+ this.config = {
81
+ minWordLength: 3,
82
+ minStemLength: 2,
83
+ ...config,
84
+ };
85
+ }
86
+
87
+ /**
88
+ * Check if a word can be normalized. Subclasses must implement this
89
+ * with language-specific script/character detection.
90
+ */
91
+ abstract isNormalizable(word: string): boolean;
92
+
93
+ /**
94
+ * Standard normalization pipeline. Override for custom behavior.
95
+ */
96
+ normalize(word: string): NormalizationResult {
97
+ const lower = word.toLowerCase();
98
+
99
+ // Check if already in dictionary form
100
+ if (this.isAlreadyNormalized(lower)) {
101
+ return noChange(word);
102
+ }
103
+
104
+ // Try reflexive normalization (Romance languages)
105
+ if (this.config.reflexiveSuffixes) {
106
+ const reflexive = this.tryReflexiveNormalization(lower);
107
+ if (reflexive) return reflexive;
108
+ }
109
+
110
+ // Try conjugation endings
111
+ if (this.config.endings) {
112
+ const conjugation = this.tryConjugationEndings(lower);
113
+ if (conjugation) return conjugation;
114
+ }
115
+
116
+ // Try suffix rules
117
+ if (this.config.suffixRules) {
118
+ const suffix = this.trySuffixRules(lower);
119
+ if (suffix) return suffix;
120
+ }
121
+
122
+ // Try prefix rules
123
+ if (this.config.prefixRules) {
124
+ const prefix = this.tryPrefixRules(lower);
125
+ if (prefix) return prefix;
126
+ }
127
+
128
+ return noChange(word);
129
+ }
130
+
131
+ /**
132
+ * Check if word is already in dictionary form (e.g., ends in -ar/-er/-ir).
133
+ * Override for language-specific checks.
134
+ */
135
+ protected isAlreadyNormalized(word: string): boolean {
136
+ if (this.config.infinitiveEndings) {
137
+ return this.config.infinitiveEndings.some(e => word.endsWith(e));
138
+ }
139
+ return false;
140
+ }
141
+
142
+ /**
143
+ * Try to strip reflexive suffixes and normalize the remainder.
144
+ * Common in Romance languages (Spanish, Portuguese, French).
145
+ */
146
+ protected tryReflexiveNormalization(word: string): NormalizationResult | null {
147
+ const suffixes = this.config.reflexiveSuffixes;
148
+ if (!suffixes) return null;
149
+
150
+ for (const suffix of suffixes) {
151
+ if (!word.endsWith(suffix)) continue;
152
+ const remainder = word.slice(0, -suffix.length);
153
+
154
+ // Check if remainder is already an infinitive
155
+ if (this.isAlreadyNormalized(remainder)) {
156
+ return normalized(remainder, 0.88, {
157
+ removedSuffixes: [suffix],
158
+ conjugationType: 'reflexive',
159
+ });
160
+ }
161
+
162
+ // Try to normalize the remainder
163
+ const inner = this.tryConjugationEndings(remainder) || this.trySuffixRules(remainder);
164
+ if (inner && inner.stem !== remainder) {
165
+ return normalized(inner.stem, inner.confidence * 0.95, {
166
+ removedSuffixes: [suffix, ...(inner.metadata?.removedSuffixes || [])],
167
+ conjugationType: 'reflexive',
168
+ });
169
+ }
170
+ }
171
+
172
+ return null;
173
+ }
174
+
175
+ /**
176
+ * Try conjugation endings (verb class endings like -ar/-er/-ir patterns).
177
+ * Endings must be pre-sorted longest-first.
178
+ */
179
+ protected tryConjugationEndings(word: string): NormalizationResult | null {
180
+ const endings = this.config.endings;
181
+ if (!endings) return null;
182
+
183
+ const minStem = this.config.minStemLength ?? 2;
184
+
185
+ for (const rule of endings) {
186
+ if (!word.endsWith(rule.ending)) continue;
187
+
188
+ const stemBase = word.slice(0, -rule.ending.length);
189
+ if (stemBase.length < minStem) continue;
190
+
191
+ const infinitive = stemBase + rule.stem;
192
+ return normalized(infinitive, rule.confidence, {
193
+ removedSuffixes: [rule.ending],
194
+ conjugationType: rule.type,
195
+ });
196
+ }
197
+
198
+ return null;
199
+ }
200
+
201
+ /**
202
+ * Try SuffixRule-style normalization.
203
+ * Rules must be pre-sorted longest-first.
204
+ */
205
+ protected trySuffixRules(word: string): NormalizationResult | null {
206
+ const rules = this.config.suffixRules;
207
+ if (!rules) return null;
208
+
209
+ const defaultMinStem = this.config.minStemLength ?? 2;
210
+
211
+ for (const rule of rules) {
212
+ if (!word.endsWith(rule.pattern)) continue;
213
+
214
+ const stem = word.slice(0, -rule.pattern.length);
215
+ const minStem = rule.minStemLength ?? defaultMinStem;
216
+ if (stem.length < minStem) continue;
217
+
218
+ const result = stem + (rule.replacement || '');
219
+ return normalized(result, rule.confidence, {
220
+ removedSuffixes: [rule.pattern],
221
+ ...(rule.conjugationType && { conjugationType: rule.conjugationType }),
222
+ });
223
+ }
224
+
225
+ return null;
226
+ }
227
+
228
+ /**
229
+ * Try PrefixRule-style normalization.
230
+ */
231
+ protected tryPrefixRules(word: string): NormalizationResult | null {
232
+ const rules = this.config.prefixRules;
233
+ if (!rules) return null;
234
+
235
+ for (const rule of rules) {
236
+ if (!word.startsWith(rule.pattern)) continue;
237
+
238
+ const remainder = word.slice(rule.pattern.length);
239
+ const minRemaining = rule.minRemaining ?? this.config.minStemLength ?? 2;
240
+ if (remainder.length < minRemaining) continue;
241
+
242
+ return normalized(remainder, 1.0 - rule.confidencePenalty, {
243
+ removedPrefixes: [rule.pattern],
244
+ });
245
+ }
246
+
247
+ return null;
248
+ }
249
+ }
@@ -1,5 +1,7 @@
1
1
  /**
2
- * Morphological normalization types and interfaces
2
+ * Morphological normalization types, interfaces, and base class
3
3
  */
4
4
 
5
5
  export * from './types';
6
+ export { BaseMorphologicalNormalizer } from './base-normalizer';
7
+ export type { ConjugationEnding, NormalizerConfig } from './base-normalizer';
package/src/core/types.ts CHANGED
@@ -8,6 +8,8 @@
8
8
  * These types are domain-agnostic - they work for any DSL (SQL, animations, etc.)
9
9
  */
10
10
 
11
+ import type { Diagnostic } from '../generation/diagnostics';
12
+
11
13
  // =============================================================================
12
14
  // Action Types (Generic)
13
15
  // =============================================================================
@@ -152,8 +154,8 @@ export interface SemanticNode {
152
154
  readonly metadata?: SemanticMetadata;
153
155
  /** Metadata annotations (v1.2). */
154
156
  readonly annotations?: readonly Annotation[];
155
- /** Type constraint diagnostics (v1.2). */
156
- readonly diagnostics?: readonly ProtocolDiagnostic[];
157
+ /** Diagnostics from parsing, validation, and schema checks (v1.2, extended v1.2.1). */
158
+ readonly diagnostics?: readonly Diagnostic[];
157
159
  }
158
160
 
159
161
  /**
package/src/index.ts CHANGED
@@ -151,6 +151,13 @@ export { defineCommand, defineRole } from './schema';
151
151
  export { createSimpleTokenizer } from './core/tokenization/base-tokenizer';
152
152
  export type { SimpleTokenizerConfig } from './core/tokenization/base-tokenizer';
153
153
 
154
+ // Morphological normalization base class and types
155
+ export { BaseMorphologicalNormalizer } from './core/tokenization/morphology/base-normalizer';
156
+ export type {
157
+ ConjugationEnding,
158
+ NormalizerConfig,
159
+ } from './core/tokenization/morphology/base-normalizer';
160
+
154
161
  // IR (Intermediate Representation) — explicit syntax, JSON conversion, reference validation
155
162
  export * from './ir';
156
163
 
@@ -0,0 +1,120 @@
1
+ /**
2
+ * Conformance Fixture Integration Tests
3
+ *
4
+ * Drives the protocol test fixtures through the bracket-syntax
5
+ * parseCompound/renderExplicit round-trip pipeline.
6
+ */
7
+
8
+ import { describe, it, expect } from 'vitest';
9
+ import { parseCompound, parseDocument } from './explicit-parser';
10
+ import { renderExplicit, renderDocument } from './explicit-renderer';
11
+ import { fromProtocolJSON, toProtocolJSON } from './protocol-json';
12
+ import type { CompoundSemanticNode } from '../core/types';
13
+
14
+ import compoundFixtures from '../../../../protocol/test-fixtures/compound.json';
15
+ import annotationFixtures from '../../../../protocol/test-fixtures/annotations.json';
16
+ import versionFixtures from '../../../../protocol/test-fixtures/version-envelope.json';
17
+
18
+ // =============================================================================
19
+ // Compound Fixtures
20
+ // =============================================================================
21
+
22
+ describe('conformance: compound.json', () => {
23
+ for (const fixture of compoundFixtures) {
24
+ it(`${fixture.id}: ${fixture.description}`, () => {
25
+ // Verify rendering matches expected output
26
+ const node = fromProtocolJSON(fixture.node as any);
27
+ const rendered = renderExplicit(node);
28
+ expect(rendered).toBe(fixture.rendered);
29
+
30
+ // Now verify we can PARSE it back (previously renderOnly)
31
+ const reparsed = parseCompound(fixture.rendered);
32
+ expect(reparsed.kind).toBe('compound');
33
+ const compound = reparsed as CompoundSemanticNode;
34
+ expect(compound.chainType).toBe(fixture.node.chainType);
35
+ expect(compound.statements).toHaveLength(fixture.node.statements!.length);
36
+
37
+ for (let i = 0; i < compound.statements.length; i++) {
38
+ expect(compound.statements[i].action).toBe(fixture.node.statements![i].action);
39
+ }
40
+ });
41
+ }
42
+ });
43
+
44
+ // =============================================================================
45
+ // Annotation Fixtures
46
+ // =============================================================================
47
+
48
+ describe('conformance: annotations.json', () => {
49
+ for (const fixture of annotationFixtures) {
50
+ it(`${fixture.id}: ${fixture.description}`, () => {
51
+ // Convert JSON → SemanticNode → render → parse → compare
52
+ const node = fromProtocolJSON(fixture.jsonInput as any);
53
+ const rendered = renderExplicit(node);
54
+ const reparsed = parseCompound(rendered);
55
+
56
+ // Verify action preserved
57
+ expect(reparsed.action).toBe(fixture.jsonInput.action);
58
+
59
+ // Verify annotations round-trip
60
+ if (fixture.jsonInput.annotations && fixture.jsonInput.annotations.length > 0) {
61
+ expect(reparsed.annotations).toHaveLength(fixture.jsonInput.annotations.length);
62
+ for (let i = 0; i < fixture.jsonInput.annotations.length; i++) {
63
+ expect(reparsed.annotations![i].name).toBe(fixture.jsonInput.annotations[i].name);
64
+ if (fixture.jsonInput.annotations[i].value !== undefined) {
65
+ expect(reparsed.annotations![i].value).toBe(fixture.jsonInput.annotations[i].value);
66
+ }
67
+ }
68
+ } else {
69
+ // ann-006: no annotations → should be undefined after parse
70
+ expect(reparsed.annotations).toBeUndefined();
71
+ }
72
+
73
+ // Verify re-render produces same output
74
+ const rerendered = renderExplicit(reparsed);
75
+ expect(rerendered).toBe(rendered);
76
+ });
77
+ }
78
+ });
79
+
80
+ // =============================================================================
81
+ // Version Envelope Fixtures (streaming format)
82
+ // =============================================================================
83
+
84
+ describe('conformance: version-envelope.json (streaming)', () => {
85
+ for (const fixture of versionFixtures) {
86
+ const f = fixture as any;
87
+ if (!f.streamingInput) continue;
88
+
89
+ // ver-007 describes v1.0/v1.1 fallback behavior — our parser is v1.2,
90
+ // so we parse the version header correctly instead of treating it as a comment.
91
+ if (f.v1Fallback) {
92
+ it(`${f.id}: v1.2 parser parses version header — ${f.description}`, () => {
93
+ const envelope = parseDocument(f.streamingInput);
94
+ // v1.2 parser extracts the version (v1.0 parsers would ignore it)
95
+ expect(envelope.lseVersion).toBe('1.2');
96
+ expect(envelope.nodes).toHaveLength(f.v1Fallback.expectedNodeCount);
97
+ });
98
+ continue;
99
+ }
100
+
101
+ it(`${f.id}: ${f.description}`, () => {
102
+ const envelope = parseDocument(f.streamingInput);
103
+ expect(envelope.lseVersion).toBe(f.expectedVersion);
104
+ expect(envelope.nodes).toHaveLength(f.expectedNodeCount);
105
+ });
106
+ }
107
+
108
+ // Also test JSON-based envelope round-trips
109
+ for (const fixture of versionFixtures) {
110
+ const f = fixture as any;
111
+ if (f.streamingInput) continue;
112
+ if (!f.expectedRoundTrip) continue;
113
+
114
+ it(`${f.id}: JSON envelope round-trip — ${f.description}`, () => {
115
+ // For JSON envelopes: verify node count and version
116
+ expect(f.jsonInput.lseVersion).toBe(f.expectedVersion);
117
+ expect(f.jsonInput.nodes).toHaveLength(f.expectedNodeCount);
118
+ });
119
+ }
120
+ });