@lokascript/framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +238 -20
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +67 -7
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +67 -7
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +75 -48
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +349 -118
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +345 -118
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +66 -7
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +78 -9
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +32 -20
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
package/dist/index.cjs
CHANGED
|
@@ -24,6 +24,7 @@ __export(index_exports, {
|
|
|
24
24
|
BaseMorphologicalNormalizer: () => BaseMorphologicalNormalizer,
|
|
25
25
|
BaseTokenizer: () => BaseTokenizer,
|
|
26
26
|
CrossDomainDispatcher: () => CrossDomainDispatcher,
|
|
27
|
+
CssSelectorExtractor: () => CssSelectorExtractor,
|
|
27
28
|
DEFAULT_OPERATORS: () => DEFAULT_OPERATORS,
|
|
28
29
|
DEFAULT_PUNCTUATION: () => DEFAULT_PUNCTUATION,
|
|
29
30
|
DEFAULT_REFERENCES: () => import_intent2.DEFAULT_REFERENCES,
|
|
@@ -60,6 +61,7 @@ __export(index_exports, {
|
|
|
60
61
|
createCompoundNode: () => import_intent.createCompoundNode,
|
|
61
62
|
createConditionalNode: () => import_intent.createConditionalNode,
|
|
62
63
|
createDiagnosticCollector: () => import_intent7.createDiagnosticCollector,
|
|
64
|
+
createDomainRenderer: () => createDomainRenderer,
|
|
63
65
|
createEventHandlerNode: () => import_intent.createEventHandlerNode,
|
|
64
66
|
createExpression: () => import_intent.createExpression,
|
|
65
67
|
createFlag: () => import_intent.createFlag,
|
|
@@ -151,6 +153,8 @@ __export(index_exports, {
|
|
|
151
153
|
semanticNodeToJSON: () => semanticNodeToJSON,
|
|
152
154
|
semanticNodeToRuntimeAST: () => semanticNodeToRuntimeAST,
|
|
153
155
|
semanticValueToAST: () => semanticValueToAST,
|
|
156
|
+
sortRolesByWordOrder: () => sortRolesByWordOrder,
|
|
157
|
+
stripOptionalDiacritics: () => stripOptionalDiacritics,
|
|
154
158
|
synthesizeFromSchemas: () => synthesizeFromSchemas,
|
|
155
159
|
toEnvelopeJSON: () => import_intent5.toEnvelopeJSON,
|
|
156
160
|
toJSONL: () => toJSONL,
|
|
@@ -1477,10 +1481,12 @@ function buildTokens(schema, profile, keyword, alternatives) {
|
|
|
1477
1481
|
addRoleWithMarker(tokens, role, profile);
|
|
1478
1482
|
}
|
|
1479
1483
|
} else if (profile.wordOrder === "SOV") {
|
|
1484
|
+
const preVerb = [];
|
|
1485
|
+
const postVerb = [];
|
|
1480
1486
|
for (const role of sortedRoles) {
|
|
1481
|
-
addRoleWithMarker(
|
|
1487
|
+
addRoleWithMarker(role.sovSlot === "postVerb" ? postVerb : preVerb, role, profile);
|
|
1482
1488
|
}
|
|
1483
|
-
tokens.push(keywordToken);
|
|
1489
|
+
tokens.push(...preVerb, keywordToken, ...postVerb);
|
|
1484
1490
|
} else if (profile.wordOrder === "VSO") {
|
|
1485
1491
|
tokens.push(keywordToken);
|
|
1486
1492
|
for (const role of sortedRoles) {
|
|
@@ -1563,28 +1569,28 @@ function buildExtractionRules(schema, profile) {
|
|
|
1563
1569
|
return rules;
|
|
1564
1570
|
}
|
|
1565
1571
|
function buildFormatString(schema, profile, keyword) {
|
|
1566
|
-
const
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
for (const role of
|
|
1572
|
+
const isSOV = profile.wordOrder === "SOV";
|
|
1573
|
+
const preVerb = [];
|
|
1574
|
+
const postVerb = [];
|
|
1575
|
+
const sortedRoles = sortRolesByWordOrder(schema.roles, profile.wordOrder);
|
|
1576
|
+
for (const role of sortedRoles) {
|
|
1571
1577
|
const marker = getMarkerForRole(role, profile);
|
|
1572
1578
|
const roleName = `{${role.role}}`;
|
|
1579
|
+
const bucket = isSOV && role.sovSlot === "postVerb" ? postVerb : preVerb;
|
|
1573
1580
|
if (marker) {
|
|
1574
1581
|
const markerInfo = profile.roleMarkers?.[role.role];
|
|
1575
1582
|
if (markerInfo?.position === "after") {
|
|
1576
|
-
|
|
1583
|
+
bucket.push(`${roleName} ${marker}`);
|
|
1577
1584
|
} else {
|
|
1578
|
-
|
|
1585
|
+
bucket.push(`${marker} ${roleName}`);
|
|
1579
1586
|
}
|
|
1580
1587
|
} else {
|
|
1581
|
-
|
|
1588
|
+
bucket.push(roleName);
|
|
1582
1589
|
}
|
|
1583
1590
|
}
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
return parts.join(" ");
|
|
1591
|
+
return (isSOV ? [...preVerb, keyword, ...postVerb] : [keyword, ...preVerb, ...postVerb]).join(
|
|
1592
|
+
" "
|
|
1593
|
+
);
|
|
1588
1594
|
}
|
|
1589
1595
|
function generatePatternVariants(schema, profile, config = defaultConfig) {
|
|
1590
1596
|
const patterns = [];
|
|
@@ -1592,6 +1598,119 @@ function generatePatternVariants(schema, profile, config = defaultConfig) {
|
|
|
1592
1598
|
return patterns;
|
|
1593
1599
|
}
|
|
1594
1600
|
|
|
1601
|
+
// src/generation/renderer.ts
|
|
1602
|
+
function lookupKeyword(keywords, action, language) {
|
|
1603
|
+
return keywords[action]?.[language] ?? action;
|
|
1604
|
+
}
|
|
1605
|
+
function lookupMarker(markers, marker, language) {
|
|
1606
|
+
return markers[marker]?.[language] ?? marker;
|
|
1607
|
+
}
|
|
1608
|
+
function buildPhrase(...parts) {
|
|
1609
|
+
return parts.filter(Boolean).join(" ");
|
|
1610
|
+
}
|
|
1611
|
+
function buildTablesFromProfiles(schemas, profiles) {
|
|
1612
|
+
const keywords = {};
|
|
1613
|
+
const markers = {};
|
|
1614
|
+
for (const profile of profiles) {
|
|
1615
|
+
for (const [action, kw] of Object.entries(profile.keywords)) {
|
|
1616
|
+
if (!keywords[action]) keywords[action] = {};
|
|
1617
|
+
keywords[action][profile.code] = kw.primary;
|
|
1618
|
+
}
|
|
1619
|
+
}
|
|
1620
|
+
for (const schema of schemas) {
|
|
1621
|
+
for (const role of schema.roles) {
|
|
1622
|
+
if (role.markerOverride) {
|
|
1623
|
+
const markerKey = role.role;
|
|
1624
|
+
if (!markers[markerKey]) markers[markerKey] = {};
|
|
1625
|
+
for (const [lang, markerText] of Object.entries(role.markerOverride)) {
|
|
1626
|
+
markers[markerKey][lang] = markerText;
|
|
1627
|
+
}
|
|
1628
|
+
}
|
|
1629
|
+
}
|
|
1630
|
+
}
|
|
1631
|
+
for (const profile of profiles) {
|
|
1632
|
+
if (profile.roleMarkers) {
|
|
1633
|
+
for (const [markerKey, markerDef] of Object.entries(profile.roleMarkers)) {
|
|
1634
|
+
if (!markers[markerKey]) markers[markerKey] = {};
|
|
1635
|
+
markers[markerKey][profile.code] = markerDef.primary;
|
|
1636
|
+
}
|
|
1637
|
+
}
|
|
1638
|
+
}
|
|
1639
|
+
return { keywords, markers };
|
|
1640
|
+
}
|
|
1641
|
+
function detectWordOrders(profiles) {
|
|
1642
|
+
const sovLanguages = /* @__PURE__ */ new Set();
|
|
1643
|
+
const vsoLanguages = /* @__PURE__ */ new Set();
|
|
1644
|
+
for (const profile of profiles) {
|
|
1645
|
+
if (profile.wordOrder === "SOV") sovLanguages.add(profile.code);
|
|
1646
|
+
if (profile.wordOrder === "VSO") vsoLanguages.add(profile.code);
|
|
1647
|
+
}
|
|
1648
|
+
return { sovLanguages, vsoLanguages };
|
|
1649
|
+
}
|
|
1650
|
+
function createSchemaRenderer(schemas, profiles) {
|
|
1651
|
+
const tables = buildRenderTables(schemas, profiles);
|
|
1652
|
+
return {
|
|
1653
|
+
render(node, language) {
|
|
1654
|
+
return renderFromSchema(node, language, tables) ?? node.action;
|
|
1655
|
+
}
|
|
1656
|
+
};
|
|
1657
|
+
}
|
|
1658
|
+
function createDomainRenderer(config) {
|
|
1659
|
+
const { schemas, profiles, overrides } = config;
|
|
1660
|
+
let tables;
|
|
1661
|
+
return (node, language) => {
|
|
1662
|
+
const override = overrides?.[node.action];
|
|
1663
|
+
if (override) return override(node, language);
|
|
1664
|
+
tables ?? (tables = buildRenderTables(schemas, profiles));
|
|
1665
|
+
return renderFromSchema(node, language, tables);
|
|
1666
|
+
};
|
|
1667
|
+
}
|
|
1668
|
+
function buildRenderTables(schemas, profiles) {
|
|
1669
|
+
const { keywords, markers } = buildTablesFromProfiles(schemas, profiles);
|
|
1670
|
+
const { sovLanguages } = detectWordOrders(profiles);
|
|
1671
|
+
const markerPositions = {};
|
|
1672
|
+
for (const profile of profiles) {
|
|
1673
|
+
for (const [role, markerDef] of Object.entries(profile.roleMarkers ?? {})) {
|
|
1674
|
+
if (!markerDef.position) continue;
|
|
1675
|
+
markerPositions[role] ?? (markerPositions[role] = {});
|
|
1676
|
+
markerPositions[role][profile.code] = markerDef.position;
|
|
1677
|
+
}
|
|
1678
|
+
}
|
|
1679
|
+
const schemaMap = /* @__PURE__ */ new Map();
|
|
1680
|
+
for (const s of schemas) schemaMap.set(s.action, s);
|
|
1681
|
+
return { keywords, markers, markerPositions, sovLanguages, schemaMap };
|
|
1682
|
+
}
|
|
1683
|
+
function renderFromSchema(node, language, tables) {
|
|
1684
|
+
const schema = tables.schemaMap.get(node.action);
|
|
1685
|
+
if (!schema) return null;
|
|
1686
|
+
const keyword = lookupKeyword(tables.keywords, node.action, language);
|
|
1687
|
+
const isSOV = tables.sovLanguages.has(language);
|
|
1688
|
+
const ordered = sortRolesByWordOrder([...schema.roles], isSOV ? "SOV" : "SVO");
|
|
1689
|
+
const preVerb = [];
|
|
1690
|
+
const postVerb = [];
|
|
1691
|
+
for (const role of ordered) {
|
|
1692
|
+
let value = (0, import_intent.extractRoleValue)(node, role.role);
|
|
1693
|
+
if (!value && role.default !== void 0) {
|
|
1694
|
+
value = String((0, import_intent.extractValue)(role.default));
|
|
1695
|
+
}
|
|
1696
|
+
if (!value) continue;
|
|
1697
|
+
if (role.quoteMultiword && /\s/.test(value) && !/^(["']).*\1$/.test(value)) {
|
|
1698
|
+
value = `"${value}"`;
|
|
1699
|
+
}
|
|
1700
|
+
const markerText = role.renderOverride?.[language] ?? role.markerOverride?.[language] ?? role.renderOverride?.["*"] ?? tables.markers[role.role]?.[language];
|
|
1701
|
+
const markerPosition = role.markerPositionOverride?.[language] ?? role.markerPosition ?? tables.markerPositions[role.role]?.[language] ?? (isSOV ? "after" : "before");
|
|
1702
|
+
const bucket = isSOV && role.sovSlot === "postVerb" ? postVerb : preVerb;
|
|
1703
|
+
if (markerText && markerPosition === "before") {
|
|
1704
|
+
bucket.push(markerText, value);
|
|
1705
|
+
} else if (markerText) {
|
|
1706
|
+
bucket.push(value, markerText);
|
|
1707
|
+
} else {
|
|
1708
|
+
bucket.push(value);
|
|
1709
|
+
}
|
|
1710
|
+
}
|
|
1711
|
+
return isSOV ? buildPhrase(...preVerb, keyword, ...postVerb) : buildPhrase(keyword, ...preVerb, ...postVerb);
|
|
1712
|
+
}
|
|
1713
|
+
|
|
1595
1714
|
// src/grammar/types.ts
|
|
1596
1715
|
function reorderRoles(roles, targetOrder) {
|
|
1597
1716
|
const result = [];
|
|
@@ -2028,6 +2147,31 @@ var LatinExtendedIdentifierExtractor = class {
|
|
|
2028
2147
|
return { value: input.slice(position, end), length: end - position };
|
|
2029
2148
|
}
|
|
2030
2149
|
};
|
|
2150
|
+
var SELECTOR_BODY_CHAR = /[\p{L}\p{N}_-]/u;
|
|
2151
|
+
var PARTICLE_SCRIPT_CHAR = /[\p{sc=Han}\p{sc=Hiragana}\p{sc=Katakana}\p{sc=Hangul}]/u;
|
|
2152
|
+
function isSelectorBodyChar(char) {
|
|
2153
|
+
return SELECTOR_BODY_CHAR.test(char) && !PARTICLE_SCRIPT_CHAR.test(char);
|
|
2154
|
+
}
|
|
2155
|
+
var CssSelectorExtractor = class {
|
|
2156
|
+
constructor() {
|
|
2157
|
+
this.name = "css-selector";
|
|
2158
|
+
}
|
|
2159
|
+
canExtract(input, position) {
|
|
2160
|
+
const char = input[position];
|
|
2161
|
+
if (char !== "#" && char !== ".") return false;
|
|
2162
|
+
const next = input[position + 1];
|
|
2163
|
+
if (next === void 0) return false;
|
|
2164
|
+
return next === "_" || next === "-" || /\p{L}/u.test(next) && !PARTICLE_SCRIPT_CHAR.test(next);
|
|
2165
|
+
}
|
|
2166
|
+
extract(input, position) {
|
|
2167
|
+
let end = position + 1;
|
|
2168
|
+
while (end < input.length && isSelectorBodyChar(input[end])) {
|
|
2169
|
+
end++;
|
|
2170
|
+
}
|
|
2171
|
+
if (end === position + 1) return null;
|
|
2172
|
+
return { value: input.slice(position, end), length: end - position };
|
|
2173
|
+
}
|
|
2174
|
+
};
|
|
2031
2175
|
var WhitespaceExtractor = class {
|
|
2032
2176
|
constructor() {
|
|
2033
2177
|
this.name = "whitespace";
|
|
@@ -2767,10 +2911,17 @@ function semanticValueToAST(value) {
|
|
|
2767
2911
|
|
|
2768
2912
|
// src/api/create-dsl.ts
|
|
2769
2913
|
var DSLRegistry = class {
|
|
2770
|
-
|
|
2914
|
+
/**
|
|
2915
|
+
* @param partiallyTranslatedActions Actions permitted to lack a keyword in
|
|
2916
|
+
* some configured languages — extension commands, which may be added for a
|
|
2917
|
+
* subset of the DSL's languages. A built-in command missing a keyword stays
|
|
2918
|
+
* an error, since that is a domain-authoring mistake.
|
|
2919
|
+
*/
|
|
2920
|
+
constructor(config, partiallyTranslatedActions = /* @__PURE__ */ new Set()) {
|
|
2771
2921
|
this.patterns = /* @__PURE__ */ new Map();
|
|
2772
2922
|
this.tokenizers = /* @__PURE__ */ new Map();
|
|
2773
2923
|
this.schemas = config.schemas;
|
|
2924
|
+
this.partiallyTranslatedActions = partiallyTranslatedActions;
|
|
2774
2925
|
for (const lang of config.languages) {
|
|
2775
2926
|
this.registerLanguage(lang);
|
|
2776
2927
|
}
|
|
@@ -2779,6 +2930,9 @@ var DSLRegistry = class {
|
|
|
2779
2930
|
this.tokenizers.set(lang.code, lang.tokenizer);
|
|
2780
2931
|
const patterns = [];
|
|
2781
2932
|
for (const schema of this.schemas) {
|
|
2933
|
+
if (this.partiallyTranslatedActions.has(schema.action) && !lang.patternProfile.keywords[schema.action]) {
|
|
2934
|
+
continue;
|
|
2935
|
+
}
|
|
2782
2936
|
const pattern = generatePattern(schema, lang.patternProfile);
|
|
2783
2937
|
patterns.push(pattern);
|
|
2784
2938
|
}
|
|
@@ -2795,13 +2949,25 @@ var DSLRegistry = class {
|
|
|
2795
2949
|
}
|
|
2796
2950
|
};
|
|
2797
2951
|
var MultilingualDSLImpl = class {
|
|
2798
|
-
constructor(config, registry, transformer) {
|
|
2952
|
+
constructor(config, registry, transformer, profileProvider) {
|
|
2799
2953
|
this.registry = registry;
|
|
2800
2954
|
this.matcher = new PatternMatcher();
|
|
2801
2955
|
this.transformer = transformer;
|
|
2956
|
+
this.profileProvider = profileProvider;
|
|
2802
2957
|
if (config.codeGenerator) {
|
|
2803
2958
|
this.codeGenerator = config.codeGenerator;
|
|
2804
2959
|
}
|
|
2960
|
+
if (config.renderer) {
|
|
2961
|
+
this.domainRenderer = config.renderer;
|
|
2962
|
+
}
|
|
2963
|
+
this.extensionRenderers = /* @__PURE__ */ new Map();
|
|
2964
|
+
for (const extension of config.extensions ?? []) {
|
|
2965
|
+
if (extension.render) this.extensionRenderers.set(extension.schema.action, extension.render);
|
|
2966
|
+
}
|
|
2967
|
+
this.schemaFallbackRenderer = createDomainRenderer({
|
|
2968
|
+
schemas: config.schemas,
|
|
2969
|
+
profiles: config.languages.map((l) => l.patternProfile)
|
|
2970
|
+
});
|
|
2805
2971
|
const schemaMap = new Map(config.schemas.map((s) => [s.action, s]));
|
|
2806
2972
|
this.schemaLookup = {
|
|
2807
2973
|
getSchema: (action) => schemaMap.get(action)
|
|
@@ -2863,6 +3029,13 @@ var MultilingualDSLImpl = class {
|
|
|
2863
3029
|
}
|
|
2864
3030
|
translate(input, fromLanguage, toLanguage) {
|
|
2865
3031
|
if ((0, import_intent3.isExplicitSyntax)(input)) return input;
|
|
3032
|
+
for (const language of [fromLanguage, toLanguage]) {
|
|
3033
|
+
if (!this.profileProvider.getProfile(language)) {
|
|
3034
|
+
throw new Error(
|
|
3035
|
+
`translate() requires a grammar profile for language "${language}", but none is configured. Set 'grammarProfile' on the LanguageConfig for "${language}" in createMultilingualDSL() (parse/validate/compile do not need it), or inject a custom 'profileProvider'.`
|
|
3036
|
+
);
|
|
3037
|
+
}
|
|
3038
|
+
}
|
|
2866
3039
|
return this.transformer.transform(input, fromLanguage, toLanguage);
|
|
2867
3040
|
}
|
|
2868
3041
|
compile(input, language) {
|
|
@@ -2891,6 +3064,18 @@ var MultilingualDSLImpl = class {
|
|
|
2891
3064
|
};
|
|
2892
3065
|
}
|
|
2893
3066
|
}
|
|
3067
|
+
render(node, language) {
|
|
3068
|
+
const extensionRenderer = this.extensionRenderers.get(node.action);
|
|
3069
|
+
if (extensionRenderer) {
|
|
3070
|
+
const rendered = extensionRenderer(node, language);
|
|
3071
|
+
if (rendered != null) return rendered;
|
|
3072
|
+
}
|
|
3073
|
+
if (this.domainRenderer) {
|
|
3074
|
+
const rendered = this.domainRenderer(node, language);
|
|
3075
|
+
if (rendered != null) return rendered;
|
|
3076
|
+
}
|
|
3077
|
+
return this.schemaFallbackRenderer(node, language);
|
|
3078
|
+
}
|
|
2894
3079
|
getSupportedLanguages() {
|
|
2895
3080
|
return this.registry.getSupportedLanguages();
|
|
2896
3081
|
}
|
|
@@ -2915,15 +3100,80 @@ function createDefaultProfileProvider(config) {
|
|
|
2915
3100
|
}
|
|
2916
3101
|
return new InMemoryProfileProvider(profiles);
|
|
2917
3102
|
}
|
|
3103
|
+
function applyExtensions(config) {
|
|
3104
|
+
const extensions = config.extensions;
|
|
3105
|
+
if (!extensions || extensions.length === 0) return config;
|
|
3106
|
+
const configuredLanguages = new Set(config.languages.map((l) => l.code));
|
|
3107
|
+
const actions = new Set(config.schemas.map((s) => s.action));
|
|
3108
|
+
for (const extension of extensions) {
|
|
3109
|
+
const { action } = extension.schema;
|
|
3110
|
+
if (actions.has(action)) {
|
|
3111
|
+
throw new Error(
|
|
3112
|
+
`Extension command "${action}" collides with a command this DSL already defines. Pick a different action name.`
|
|
3113
|
+
);
|
|
3114
|
+
}
|
|
3115
|
+
actions.add(action);
|
|
3116
|
+
for (const code of Object.keys(extension.vocabulary)) {
|
|
3117
|
+
if (!configuredLanguages.has(code)) {
|
|
3118
|
+
throw new Error(
|
|
3119
|
+
`Extension command "${action}" supplies vocabulary for language "${code}", which is not configured on this DSL. Configured languages: ${[...configuredLanguages].join(", ")}.`
|
|
3120
|
+
);
|
|
3121
|
+
}
|
|
3122
|
+
}
|
|
3123
|
+
}
|
|
3124
|
+
const schemas = [...config.schemas, ...extensions.map((e) => e.schema)];
|
|
3125
|
+
const languages = config.languages.map((lang) => {
|
|
3126
|
+
const keywords = { ...lang.patternProfile.keywords };
|
|
3127
|
+
let roleMarkers = lang.patternProfile.roleMarkers;
|
|
3128
|
+
for (const extension of extensions) {
|
|
3129
|
+
const vocabulary = extension.vocabulary[lang.code];
|
|
3130
|
+
if (!vocabulary) continue;
|
|
3131
|
+
keywords[extension.schema.action] = { ...vocabulary.keyword };
|
|
3132
|
+
if (vocabulary.roleMarkers) {
|
|
3133
|
+
roleMarkers = { ...roleMarkers, ...vocabulary.roleMarkers };
|
|
3134
|
+
}
|
|
3135
|
+
}
|
|
3136
|
+
return {
|
|
3137
|
+
...lang,
|
|
3138
|
+
patternProfile: {
|
|
3139
|
+
...lang.patternProfile,
|
|
3140
|
+
keywords,
|
|
3141
|
+
...roleMarkers !== void 0 && { roleMarkers }
|
|
3142
|
+
}
|
|
3143
|
+
};
|
|
3144
|
+
});
|
|
3145
|
+
const extensionGenerators = new Map(
|
|
3146
|
+
extensions.filter((e) => e.generate).map((e) => [e.schema.action, e.generate])
|
|
3147
|
+
);
|
|
3148
|
+
const baseGenerator = config.codeGenerator;
|
|
3149
|
+
const codeGenerator = extensionGenerators.size > 0 ? {
|
|
3150
|
+
generate(node) {
|
|
3151
|
+
const generate = extensionGenerators.get(node.action);
|
|
3152
|
+
if (generate) return generate(node);
|
|
3153
|
+
if (!baseGenerator) {
|
|
3154
|
+
throw new Error(`No code generator for action "${node.action}"`);
|
|
3155
|
+
}
|
|
3156
|
+
return baseGenerator.generate(node);
|
|
3157
|
+
}
|
|
3158
|
+
} : baseGenerator;
|
|
3159
|
+
return {
|
|
3160
|
+
...config,
|
|
3161
|
+
schemas,
|
|
3162
|
+
languages,
|
|
3163
|
+
...codeGenerator !== void 0 && { codeGenerator }
|
|
3164
|
+
};
|
|
3165
|
+
}
|
|
2918
3166
|
function createMultilingualDSL(config) {
|
|
2919
|
-
const
|
|
2920
|
-
const
|
|
3167
|
+
const effectiveConfig = applyExtensions(config);
|
|
3168
|
+
const extensionActions = new Set((config.extensions ?? []).map((e) => e.schema.action));
|
|
3169
|
+
const dictionary = effectiveConfig.dictionary ?? createDefaultDictionary(effectiveConfig);
|
|
3170
|
+
const profileProvider = effectiveConfig.profileProvider ?? createDefaultProfileProvider(effectiveConfig);
|
|
2921
3171
|
const transformer = new GrammarTransformer({
|
|
2922
3172
|
dictionary,
|
|
2923
3173
|
profileProvider
|
|
2924
3174
|
});
|
|
2925
|
-
const registry = new DSLRegistry(
|
|
2926
|
-
return new MultilingualDSLImpl(
|
|
3175
|
+
const registry = new DSLRegistry(effectiveConfig, extensionActions);
|
|
3176
|
+
return new MultilingualDSLImpl(effectiveConfig, registry, transformer, profileProvider);
|
|
2927
3177
|
}
|
|
2928
3178
|
|
|
2929
3179
|
// src/prompts/prompt-generator.ts
|
|
@@ -3010,7 +3260,9 @@ function buildProtocolSection() {
|
|
|
3010
3260
|
3. No spaces around the colon in role:value pairs
|
|
3011
3261
|
4. Strings with spaces must be quoted: \`patient:"hello world"\`
|
|
3012
3262
|
5. Selectors start with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`
|
|
3013
|
-
6.
|
|
3263
|
+
6. A selector containing a space, a combinator (\`>\` \`+\` \`~\`), or a comma must use a selector literal: \`patient:<ul > li/>\`, \`patient:<.a, .b/>\`
|
|
3264
|
+
7. Inside a structural role (\`body\`, \`then\`, \`else\`, \`condition\`, \`loop-body\`, \`variable\`, \`catch\`, \`finally\`) a \`[...]\` value is always a nested command. Write an attribute selector there as \`condition:<[data-active]/>\`
|
|
3265
|
+
8. Output must be valid bracket syntax: \`[action role:value ...]\``;
|
|
3014
3266
|
return {
|
|
3015
3267
|
id: "protocol",
|
|
3016
3268
|
title: "LSE Protocol",
|
|
@@ -3023,6 +3275,7 @@ function buildValueTypeSection() {
|
|
|
3023
3275
|
|
|
3024
3276
|
| Type | Syntax | Example |
|
|
3025
3277
|
|------|--------|---------|
|
|
3278
|
+
| Selector literal | Delimited by \`<\` and \`/>\` | \`<ul > li/>\`, \`<.a, .b/>\`, \`<[data-id]/>\` |
|
|
3026
3279
|
| Selector | Starts with \`#\` \`.\` \`[\` \`@\` \`*\` | \`#button\`, \`.active\`, \`[data-id]\` |
|
|
3027
3280
|
| String | Quoted with \`"\` or \`'\` | \`"hello world"\`, \`'json'\` |
|
|
3028
3281
|
| Boolean | Exact: \`true\` / \`false\` | \`visible:true\` |
|
|
@@ -3891,6 +4144,11 @@ var DomainRegistry = class {
|
|
|
3891
4144
|
rendered = typeof renderer === "function" ? renderer(node, to) : renderer.render(node, to);
|
|
3892
4145
|
} catch {
|
|
3893
4146
|
}
|
|
4147
|
+
} else if (dsl.render) {
|
|
4148
|
+
try {
|
|
4149
|
+
rendered = dsl.render(node, to);
|
|
4150
|
+
} catch {
|
|
4151
|
+
}
|
|
3894
4152
|
}
|
|
3895
4153
|
const roles = {};
|
|
3896
4154
|
for (const [key, value] of node.roles) {
|
|
@@ -4448,96 +4706,6 @@ async function registryToAOTBackends(registry) {
|
|
|
4448
4706
|
// src/schema/command-schema.ts
|
|
4449
4707
|
var import_intent6 = require("@lokascript/intent");
|
|
4450
4708
|
|
|
4451
|
-
// src/generation/renderer.ts
|
|
4452
|
-
function lookupKeyword(keywords, action, language) {
|
|
4453
|
-
return keywords[action]?.[language] ?? action;
|
|
4454
|
-
}
|
|
4455
|
-
function lookupMarker(markers, marker, language) {
|
|
4456
|
-
return markers[marker]?.[language] ?? marker;
|
|
4457
|
-
}
|
|
4458
|
-
function buildPhrase(...parts) {
|
|
4459
|
-
return parts.filter(Boolean).join(" ");
|
|
4460
|
-
}
|
|
4461
|
-
function buildTablesFromProfiles(schemas, profiles) {
|
|
4462
|
-
const keywords = {};
|
|
4463
|
-
const markers = {};
|
|
4464
|
-
for (const profile of profiles) {
|
|
4465
|
-
for (const [action, kw] of Object.entries(profile.keywords)) {
|
|
4466
|
-
if (!keywords[action]) keywords[action] = {};
|
|
4467
|
-
keywords[action][profile.code] = kw.primary;
|
|
4468
|
-
}
|
|
4469
|
-
}
|
|
4470
|
-
for (const schema of schemas) {
|
|
4471
|
-
for (const role of schema.roles) {
|
|
4472
|
-
if (role.markerOverride) {
|
|
4473
|
-
const markerKey = role.role;
|
|
4474
|
-
if (!markers[markerKey]) markers[markerKey] = {};
|
|
4475
|
-
for (const [lang, markerText] of Object.entries(role.markerOverride)) {
|
|
4476
|
-
markers[markerKey][lang] = markerText;
|
|
4477
|
-
}
|
|
4478
|
-
}
|
|
4479
|
-
}
|
|
4480
|
-
}
|
|
4481
|
-
for (const profile of profiles) {
|
|
4482
|
-
if (profile.roleMarkers) {
|
|
4483
|
-
for (const [markerKey, markerDef] of Object.entries(profile.roleMarkers)) {
|
|
4484
|
-
if (!markers[markerKey]) markers[markerKey] = {};
|
|
4485
|
-
markers[markerKey][profile.code] = markerDef.primary;
|
|
4486
|
-
}
|
|
4487
|
-
}
|
|
4488
|
-
}
|
|
4489
|
-
return { keywords, markers };
|
|
4490
|
-
}
|
|
4491
|
-
function detectWordOrders(profiles) {
|
|
4492
|
-
const sovLanguages = /* @__PURE__ */ new Set();
|
|
4493
|
-
const vsoLanguages = /* @__PURE__ */ new Set();
|
|
4494
|
-
for (const profile of profiles) {
|
|
4495
|
-
if (profile.wordOrder === "SOV") sovLanguages.add(profile.code);
|
|
4496
|
-
if (profile.wordOrder === "VSO") vsoLanguages.add(profile.code);
|
|
4497
|
-
}
|
|
4498
|
-
return { sovLanguages, vsoLanguages };
|
|
4499
|
-
}
|
|
4500
|
-
function createSchemaRenderer(schemas, profiles) {
|
|
4501
|
-
const { keywords, markers } = buildTablesFromProfiles(schemas, profiles);
|
|
4502
|
-
const { sovLanguages } = detectWordOrders(profiles);
|
|
4503
|
-
const schemaMap = /* @__PURE__ */ new Map();
|
|
4504
|
-
for (const s of schemas) schemaMap.set(s.action, s);
|
|
4505
|
-
return {
|
|
4506
|
-
render(node, language) {
|
|
4507
|
-
const schema = schemaMap.get(node.action);
|
|
4508
|
-
if (!schema) return node.action;
|
|
4509
|
-
const keyword = lookupKeyword(keywords, node.action, language);
|
|
4510
|
-
const isSOV = sovLanguages.has(language);
|
|
4511
|
-
const roleParts = [];
|
|
4512
|
-
for (const role of schema.roles) {
|
|
4513
|
-
const value = (0, import_intent.extractRoleValue)(node, role.role);
|
|
4514
|
-
if (!value && !role.required) continue;
|
|
4515
|
-
const markerText = role.markerOverride?.[language] ?? markers[role.role]?.[language] ?? void 0;
|
|
4516
|
-
roleParts.push({
|
|
4517
|
-
...markerText != null && { marker: markerText },
|
|
4518
|
-
value: value || "",
|
|
4519
|
-
role
|
|
4520
|
-
});
|
|
4521
|
-
}
|
|
4522
|
-
const parts = [];
|
|
4523
|
-
if (isSOV) {
|
|
4524
|
-
for (const rp of roleParts) {
|
|
4525
|
-
if (rp.value) parts.push(rp.value);
|
|
4526
|
-
if (rp.marker) parts.push(rp.marker);
|
|
4527
|
-
}
|
|
4528
|
-
parts.push(keyword);
|
|
4529
|
-
} else {
|
|
4530
|
-
parts.push(keyword);
|
|
4531
|
-
for (const rp of roleParts) {
|
|
4532
|
-
if (rp.marker) parts.push(rp.marker);
|
|
4533
|
-
if (rp.value) parts.push(rp.value);
|
|
4534
|
-
}
|
|
4535
|
-
}
|
|
4536
|
-
return buildPhrase(...parts);
|
|
4537
|
-
}
|
|
4538
|
-
};
|
|
4539
|
-
}
|
|
4540
|
-
|
|
4541
4709
|
// src/generation/diagnostics.ts
|
|
4542
4710
|
var import_intent7 = require("@lokascript/intent");
|
|
4543
4711
|
|
|
@@ -4647,6 +4815,9 @@ function isQuote(char) {
|
|
|
4647
4815
|
function isDigit(char) {
|
|
4648
4816
|
return /\d/.test(char);
|
|
4649
4817
|
}
|
|
4818
|
+
function stripOptionalDiacritics(word) {
|
|
4819
|
+
return word.replace(/[ً-ْٰ]/g, "");
|
|
4820
|
+
}
|
|
4650
4821
|
function isAsciiLetter(char) {
|
|
4651
4822
|
return /[a-zA-Z]/.test(char);
|
|
4652
4823
|
}
|
|
@@ -5218,7 +5389,39 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5218
5389
|
pos++;
|
|
5219
5390
|
}
|
|
5220
5391
|
}
|
|
5221
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
5392
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
5393
|
+
}
|
|
5394
|
+
/**
|
|
5395
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
5396
|
+
*
|
|
5397
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
5398
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
5399
|
+
* sigil. The English tokenizer already merges these inside
|
|
5400
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
5401
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
5402
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
5403
|
+
* local-variable reference survives untouched.
|
|
5404
|
+
*
|
|
5405
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
5406
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
5407
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
5408
|
+
*/
|
|
5409
|
+
mergeColonQualifiedNames(tokens) {
|
|
5410
|
+
const out = [];
|
|
5411
|
+
for (const tok of tokens) {
|
|
5412
|
+
const prev = out[out.length - 1];
|
|
5413
|
+
if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
|
|
5414
|
+
const merged = prev.value + tok.value;
|
|
5415
|
+
out[out.length - 1] = createToken(
|
|
5416
|
+
merged,
|
|
5417
|
+
this.classifyToken(merged),
|
|
5418
|
+
createPosition(prev.position.start, tok.position.end)
|
|
5419
|
+
);
|
|
5420
|
+
continue;
|
|
5421
|
+
}
|
|
5422
|
+
out.push(tok);
|
|
5423
|
+
}
|
|
5424
|
+
return out;
|
|
5222
5425
|
}
|
|
5223
5426
|
/**
|
|
5224
5427
|
* Classify an unknown character when no extractor matches.
|
|
@@ -5351,7 +5554,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5351
5554
|
* @returns Word without diacritics
|
|
5352
5555
|
*/
|
|
5353
5556
|
removeDiacritics(word) {
|
|
5354
|
-
return word
|
|
5557
|
+
return stripOptionalDiacritics(word);
|
|
5355
5558
|
}
|
|
5356
5559
|
/**
|
|
5357
5560
|
* Try to match a keyword from profile at the current position.
|
|
@@ -5442,24 +5645,40 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5442
5645
|
});
|
|
5443
5646
|
}
|
|
5444
5647
|
/**
|
|
5445
|
-
* Look up a keyword by native word (case-insensitive).
|
|
5648
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
5446
5649
|
* O(1) lookup using the keyword map.
|
|
5447
5650
|
*
|
|
5651
|
+
* The map is INDEXED both with and without diacritics (see
|
|
5652
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
5653
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
5654
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
5655
|
+
* every previously-matching word resolves byte-identically.
|
|
5656
|
+
*
|
|
5657
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
5658
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
5659
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
5660
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
5661
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
5662
|
+
*
|
|
5448
5663
|
* @param native - Native word to look up
|
|
5449
5664
|
* @returns KeywordEntry if found, undefined otherwise
|
|
5450
5665
|
*/
|
|
5451
5666
|
lookupKeyword(native) {
|
|
5452
|
-
|
|
5667
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
5668
|
+
if (exact) return exact;
|
|
5669
|
+
const stripped = this.removeDiacritics(native);
|
|
5670
|
+
if (stripped === native) return void 0;
|
|
5671
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
5453
5672
|
}
|
|
5454
5673
|
/**
|
|
5455
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
5456
|
-
* O(1) lookup using the keyword map.
|
|
5674
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
5675
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
5457
5676
|
*
|
|
5458
5677
|
* @param native - Native word to check
|
|
5459
5678
|
* @returns true if the word is a keyword
|
|
5460
5679
|
*/
|
|
5461
5680
|
isKeyword(native) {
|
|
5462
|
-
return this.
|
|
5681
|
+
return this.lookupKeyword(native) !== void 0;
|
|
5463
5682
|
}
|
|
5464
5683
|
/**
|
|
5465
5684
|
* Set the morphological normalizer for this tokenizer.
|
|
@@ -5724,6 +5943,14 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5724
5943
|
return null;
|
|
5725
5944
|
}
|
|
5726
5945
|
};
|
|
5946
|
+
/**
|
|
5947
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
5948
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
5949
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
5950
|
+
*/
|
|
5951
|
+
_BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
5952
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
5953
|
+
_BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
5727
5954
|
/**
|
|
5728
5955
|
* Configuration for native language time units.
|
|
5729
5956
|
* Maps patterns to their standard suffix (ms, s, m, h).
|
|
@@ -6429,6 +6656,7 @@ function accumulateIndented(statements, config) {
|
|
|
6429
6656
|
BaseMorphologicalNormalizer,
|
|
6430
6657
|
BaseTokenizer,
|
|
6431
6658
|
CrossDomainDispatcher,
|
|
6659
|
+
CssSelectorExtractor,
|
|
6432
6660
|
DEFAULT_OPERATORS,
|
|
6433
6661
|
DEFAULT_PUNCTUATION,
|
|
6434
6662
|
DEFAULT_REFERENCES,
|
|
@@ -6465,6 +6693,7 @@ function accumulateIndented(statements, config) {
|
|
|
6465
6693
|
createCompoundNode,
|
|
6466
6694
|
createConditionalNode,
|
|
6467
6695
|
createDiagnosticCollector,
|
|
6696
|
+
createDomainRenderer,
|
|
6468
6697
|
createEventHandlerNode,
|
|
6469
6698
|
createExpression,
|
|
6470
6699
|
createFlag,
|
|
@@ -6556,6 +6785,8 @@ function accumulateIndented(statements, config) {
|
|
|
6556
6785
|
semanticNodeToJSON,
|
|
6557
6786
|
semanticNodeToRuntimeAST,
|
|
6558
6787
|
semanticValueToAST,
|
|
6788
|
+
sortRolesByWordOrder,
|
|
6789
|
+
stripOptionalDiacritics,
|
|
6559
6790
|
synthesizeFromSchemas,
|
|
6560
6791
|
toEnvelopeJSON,
|
|
6561
6792
|
toJSONL,
|