@lokascript/framework 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/index.js +111 -24
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +133 -0
- package/dist/core/index.js.map +1 -1
- package/dist/core/pattern-matching/index.js.map +1 -1
- package/dist/core/tokenization/index.js +133 -0
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/morphology/base-normalizer.d.ts +94 -0
- package/dist/core/tokenization/morphology/base-normalizer.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +3 -1
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -1
- package/dist/core/types.d.ts +3 -2
- package/dist/core/types.d.ts.map +1 -1
- package/dist/core/types.js.map +1 -1
- package/dist/generation/index.js.map +1 -1
- package/dist/index.cjs +564 -46
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +556 -46
- package/dist/index.js.map +1 -1
- package/dist/ir/explicit-parser.d.ts +62 -7
- package/dist/ir/explicit-parser.d.ts.map +1 -1
- package/dist/ir/explicit-renderer.d.ts +18 -1
- package/dist/ir/explicit-renderer.d.ts.map +1 -1
- package/dist/ir/index.d.ts +4 -2
- package/dist/ir/index.d.ts.map +1 -1
- package/dist/ir/index.js +423 -46
- package/dist/ir/index.js.map +1 -1
- package/dist/ir/protocol-json.d.ts.map +1 -1
- package/dist/ir/to-runtime-ast.d.ts +53 -0
- package/dist/ir/to-runtime-ast.d.ts.map +1 -0
- package/dist/ir/types.d.ts +14 -4
- package/dist/ir/types.d.ts.map +1 -1
- package/dist/parsing/index.js.map +1 -1
- package/dist/testing/index.js +8671 -8261
- package/dist/testing/index.js.map +1 -1
- package/package.json +1 -1
- package/src/core/tokenization/morphology/base-normalizer.ts +249 -0
- package/src/core/tokenization/morphology/index.ts +3 -1
- package/src/core/types.ts +4 -2
- package/src/index.ts +7 -0
- package/src/ir/conformance.test.ts +120 -0
- package/src/ir/explicit-parser.test.ts +307 -1
- package/src/ir/explicit-parser.ts +403 -29
- package/src/ir/explicit-renderer.test.ts +87 -2
- package/src/ir/explicit-renderer.ts +57 -5
- package/src/ir/index.ts +13 -2
- package/src/ir/protocol-json.test.ts +4 -2
- package/src/ir/protocol-json.ts +30 -24
- package/src/ir/to-runtime-ast.ts +203 -0
- package/src/ir/types.ts +14 -4
|
@@ -1312,7 +1312,140 @@ function normalized(stem, confidence, metadata) {
|
|
|
1312
1312
|
}
|
|
1313
1313
|
return { stem, confidence };
|
|
1314
1314
|
}
|
|
1315
|
+
|
|
1316
|
+
// src/core/tokenization/morphology/base-normalizer.ts
|
|
1317
|
+
var BaseMorphologicalNormalizer = class {
|
|
1318
|
+
constructor(config) {
|
|
1319
|
+
this.language = config.language;
|
|
1320
|
+
this.config = {
|
|
1321
|
+
minWordLength: 3,
|
|
1322
|
+
minStemLength: 2,
|
|
1323
|
+
...config
|
|
1324
|
+
};
|
|
1325
|
+
}
|
|
1326
|
+
/**
|
|
1327
|
+
* Standard normalization pipeline. Override for custom behavior.
|
|
1328
|
+
*/
|
|
1329
|
+
normalize(word) {
|
|
1330
|
+
const lower = word.toLowerCase();
|
|
1331
|
+
if (this.isAlreadyNormalized(lower)) {
|
|
1332
|
+
return noChange(word);
|
|
1333
|
+
}
|
|
1334
|
+
if (this.config.reflexiveSuffixes) {
|
|
1335
|
+
const reflexive = this.tryReflexiveNormalization(lower);
|
|
1336
|
+
if (reflexive) return reflexive;
|
|
1337
|
+
}
|
|
1338
|
+
if (this.config.endings) {
|
|
1339
|
+
const conjugation = this.tryConjugationEndings(lower);
|
|
1340
|
+
if (conjugation) return conjugation;
|
|
1341
|
+
}
|
|
1342
|
+
if (this.config.suffixRules) {
|
|
1343
|
+
const suffix = this.trySuffixRules(lower);
|
|
1344
|
+
if (suffix) return suffix;
|
|
1345
|
+
}
|
|
1346
|
+
if (this.config.prefixRules) {
|
|
1347
|
+
const prefix = this.tryPrefixRules(lower);
|
|
1348
|
+
if (prefix) return prefix;
|
|
1349
|
+
}
|
|
1350
|
+
return noChange(word);
|
|
1351
|
+
}
|
|
1352
|
+
/**
|
|
1353
|
+
* Check if word is already in dictionary form (e.g., ends in -ar/-er/-ir).
|
|
1354
|
+
* Override for language-specific checks.
|
|
1355
|
+
*/
|
|
1356
|
+
isAlreadyNormalized(word) {
|
|
1357
|
+
if (this.config.infinitiveEndings) {
|
|
1358
|
+
return this.config.infinitiveEndings.some((e) => word.endsWith(e));
|
|
1359
|
+
}
|
|
1360
|
+
return false;
|
|
1361
|
+
}
|
|
1362
|
+
/**
|
|
1363
|
+
* Try to strip reflexive suffixes and normalize the remainder.
|
|
1364
|
+
* Common in Romance languages (Spanish, Portuguese, French).
|
|
1365
|
+
*/
|
|
1366
|
+
tryReflexiveNormalization(word) {
|
|
1367
|
+
const suffixes = this.config.reflexiveSuffixes;
|
|
1368
|
+
if (!suffixes) return null;
|
|
1369
|
+
for (const suffix of suffixes) {
|
|
1370
|
+
if (!word.endsWith(suffix)) continue;
|
|
1371
|
+
const remainder = word.slice(0, -suffix.length);
|
|
1372
|
+
if (this.isAlreadyNormalized(remainder)) {
|
|
1373
|
+
return normalized(remainder, 0.88, {
|
|
1374
|
+
removedSuffixes: [suffix],
|
|
1375
|
+
conjugationType: "reflexive"
|
|
1376
|
+
});
|
|
1377
|
+
}
|
|
1378
|
+
const inner = this.tryConjugationEndings(remainder) || this.trySuffixRules(remainder);
|
|
1379
|
+
if (inner && inner.stem !== remainder) {
|
|
1380
|
+
return normalized(inner.stem, inner.confidence * 0.95, {
|
|
1381
|
+
removedSuffixes: [suffix, ...inner.metadata?.removedSuffixes || []],
|
|
1382
|
+
conjugationType: "reflexive"
|
|
1383
|
+
});
|
|
1384
|
+
}
|
|
1385
|
+
}
|
|
1386
|
+
return null;
|
|
1387
|
+
}
|
|
1388
|
+
/**
|
|
1389
|
+
* Try conjugation endings (verb class endings like -ar/-er/-ir patterns).
|
|
1390
|
+
* Endings must be pre-sorted longest-first.
|
|
1391
|
+
*/
|
|
1392
|
+
tryConjugationEndings(word) {
|
|
1393
|
+
const endings = this.config.endings;
|
|
1394
|
+
if (!endings) return null;
|
|
1395
|
+
const minStem = this.config.minStemLength ?? 2;
|
|
1396
|
+
for (const rule of endings) {
|
|
1397
|
+
if (!word.endsWith(rule.ending)) continue;
|
|
1398
|
+
const stemBase = word.slice(0, -rule.ending.length);
|
|
1399
|
+
if (stemBase.length < minStem) continue;
|
|
1400
|
+
const infinitive = stemBase + rule.stem;
|
|
1401
|
+
return normalized(infinitive, rule.confidence, {
|
|
1402
|
+
removedSuffixes: [rule.ending],
|
|
1403
|
+
conjugationType: rule.type
|
|
1404
|
+
});
|
|
1405
|
+
}
|
|
1406
|
+
return null;
|
|
1407
|
+
}
|
|
1408
|
+
/**
|
|
1409
|
+
* Try SuffixRule-style normalization.
|
|
1410
|
+
* Rules must be pre-sorted longest-first.
|
|
1411
|
+
*/
|
|
1412
|
+
trySuffixRules(word) {
|
|
1413
|
+
const rules = this.config.suffixRules;
|
|
1414
|
+
if (!rules) return null;
|
|
1415
|
+
const defaultMinStem = this.config.minStemLength ?? 2;
|
|
1416
|
+
for (const rule of rules) {
|
|
1417
|
+
if (!word.endsWith(rule.pattern)) continue;
|
|
1418
|
+
const stem = word.slice(0, -rule.pattern.length);
|
|
1419
|
+
const minStem = rule.minStemLength ?? defaultMinStem;
|
|
1420
|
+
if (stem.length < minStem) continue;
|
|
1421
|
+
const result = stem + (rule.replacement || "");
|
|
1422
|
+
return normalized(result, rule.confidence, {
|
|
1423
|
+
removedSuffixes: [rule.pattern],
|
|
1424
|
+
...rule.conjugationType && { conjugationType: rule.conjugationType }
|
|
1425
|
+
});
|
|
1426
|
+
}
|
|
1427
|
+
return null;
|
|
1428
|
+
}
|
|
1429
|
+
/**
|
|
1430
|
+
* Try PrefixRule-style normalization.
|
|
1431
|
+
*/
|
|
1432
|
+
tryPrefixRules(word) {
|
|
1433
|
+
const rules = this.config.prefixRules;
|
|
1434
|
+
if (!rules) return null;
|
|
1435
|
+
for (const rule of rules) {
|
|
1436
|
+
if (!word.startsWith(rule.pattern)) continue;
|
|
1437
|
+
const remainder = word.slice(rule.pattern.length);
|
|
1438
|
+
const minRemaining = rule.minRemaining ?? this.config.minStemLength ?? 2;
|
|
1439
|
+
if (remainder.length < minRemaining) continue;
|
|
1440
|
+
return normalized(remainder, 1 - rule.confidencePenalty, {
|
|
1441
|
+
removedPrefixes: [rule.pattern]
|
|
1442
|
+
});
|
|
1443
|
+
}
|
|
1444
|
+
return null;
|
|
1445
|
+
}
|
|
1446
|
+
};
|
|
1315
1447
|
export {
|
|
1448
|
+
BaseMorphologicalNormalizer,
|
|
1316
1449
|
BaseTokenizer,
|
|
1317
1450
|
DEFAULT_OPERATORS,
|
|
1318
1451
|
DEFAULT_PUNCTUATION,
|