@lokascript/framework 2.7.2 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/api/create-dsl.d.ts +93 -1
  3. package/dist/api/create-dsl.d.ts.map +1 -1
  4. package/dist/api/domain-registry.d.ts +5 -3
  5. package/dist/api/domain-registry.d.ts.map +1 -1
  6. package/dist/api/index.js +238 -20
  7. package/dist/api/index.js.map +1 -1
  8. package/dist/core/index.js +67 -7
  9. package/dist/core/index.js.map +1 -1
  10. package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
  11. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  12. package/dist/core/tokenization/extractors.d.ts +6 -0
  13. package/dist/core/tokenization/extractors.d.ts.map +1 -1
  14. package/dist/core/tokenization/index.js +67 -7
  15. package/dist/core/tokenization/index.js.map +1 -1
  16. package/dist/core/tokenization/token-utils.d.ts +15 -0
  17. package/dist/core/tokenization/token-utils.d.ts.map +1 -1
  18. package/dist/generation/index.js +75 -48
  19. package/dist/generation/index.js.map +1 -1
  20. package/dist/generation/pattern-generator.d.ts +8 -1
  21. package/dist/generation/pattern-generator.d.ts.map +1 -1
  22. package/dist/generation/renderer.d.ts +53 -1
  23. package/dist/generation/renderer.d.ts.map +1 -1
  24. package/dist/index.cjs +349 -118
  25. package/dist/index.cjs.map +1 -1
  26. package/dist/index.d.ts +4 -4
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +345 -118
  29. package/dist/index.js.map +1 -1
  30. package/dist/interfaces/value-extractor.d.ts +5 -0
  31. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  32. package/dist/multilingual/index.js +66 -7
  33. package/dist/multilingual/index.js.map +1 -1
  34. package/dist/testing/index.js +4 -12
  35. package/dist/testing/index.js.map +1 -1
  36. package/package.json +4 -3
  37. package/src/api/create-dsl.test.ts +11 -0
  38. package/src/api/create-dsl.ts +278 -9
  39. package/src/api/domain-registry.ts +15 -10
  40. package/src/api/extensions.test.ts +322 -0
  41. package/src/core/tokenization/base-tokenizer.ts +78 -9
  42. package/src/core/tokenization/colon-qualifier.test.ts +129 -0
  43. package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
  44. package/src/core/tokenization/extractors.ts +6 -0
  45. package/src/core/tokenization/token-utils.ts +18 -0
  46. package/src/generation/domain-renderer.test.ts +172 -0
  47. package/src/generation/pattern-generator.test.ts +102 -0
  48. package/src/generation/pattern-generator.ts +32 -20
  49. package/src/generation/renderer.test.ts +243 -4
  50. package/src/generation/renderer.ts +188 -45
  51. package/src/index.ts +9 -1
  52. package/src/interfaces/value-extractor.ts +50 -0
  53. package/src/ir/protocol-json.test.ts +21 -0
  54. package/src/ir/references.test.ts +5 -2
  55. package/src/prompts/prompt-generator.ts +4 -1
package/dist/index.cjs CHANGED
@@ -24,6 +24,7 @@ __export(index_exports, {
24
24
  BaseMorphologicalNormalizer: () => BaseMorphologicalNormalizer,
25
25
  BaseTokenizer: () => BaseTokenizer,
26
26
  CrossDomainDispatcher: () => CrossDomainDispatcher,
27
+ CssSelectorExtractor: () => CssSelectorExtractor,
27
28
  DEFAULT_OPERATORS: () => DEFAULT_OPERATORS,
28
29
  DEFAULT_PUNCTUATION: () => DEFAULT_PUNCTUATION,
29
30
  DEFAULT_REFERENCES: () => import_intent2.DEFAULT_REFERENCES,
@@ -60,6 +61,7 @@ __export(index_exports, {
60
61
  createCompoundNode: () => import_intent.createCompoundNode,
61
62
  createConditionalNode: () => import_intent.createConditionalNode,
62
63
  createDiagnosticCollector: () => import_intent7.createDiagnosticCollector,
64
+ createDomainRenderer: () => createDomainRenderer,
63
65
  createEventHandlerNode: () => import_intent.createEventHandlerNode,
64
66
  createExpression: () => import_intent.createExpression,
65
67
  createFlag: () => import_intent.createFlag,
@@ -151,6 +153,8 @@ __export(index_exports, {
151
153
  semanticNodeToJSON: () => semanticNodeToJSON,
152
154
  semanticNodeToRuntimeAST: () => semanticNodeToRuntimeAST,
153
155
  semanticValueToAST: () => semanticValueToAST,
156
+ sortRolesByWordOrder: () => sortRolesByWordOrder,
157
+ stripOptionalDiacritics: () => stripOptionalDiacritics,
154
158
  synthesizeFromSchemas: () => synthesizeFromSchemas,
155
159
  toEnvelopeJSON: () => import_intent5.toEnvelopeJSON,
156
160
  toJSONL: () => toJSONL,
@@ -1477,10 +1481,12 @@ function buildTokens(schema, profile, keyword, alternatives) {
1477
1481
  addRoleWithMarker(tokens, role, profile);
1478
1482
  }
1479
1483
  } else if (profile.wordOrder === "SOV") {
1484
+ const preVerb = [];
1485
+ const postVerb = [];
1480
1486
  for (const role of sortedRoles) {
1481
- addRoleWithMarker(tokens, role, profile);
1487
+ addRoleWithMarker(role.sovSlot === "postVerb" ? postVerb : preVerb, role, profile);
1482
1488
  }
1483
- tokens.push(keywordToken);
1489
+ tokens.push(...preVerb, keywordToken, ...postVerb);
1484
1490
  } else if (profile.wordOrder === "VSO") {
1485
1491
  tokens.push(keywordToken);
1486
1492
  for (const role of sortedRoles) {
@@ -1563,28 +1569,28 @@ function buildExtractionRules(schema, profile) {
1563
1569
  return rules;
1564
1570
  }
1565
1571
  function buildFormatString(schema, profile, keyword) {
1566
- const parts = [];
1567
- if (profile.wordOrder === "SVO" || profile.wordOrder === "VSO") {
1568
- parts.push(keyword);
1569
- }
1570
- for (const role of schema.roles) {
1572
+ const isSOV = profile.wordOrder === "SOV";
1573
+ const preVerb = [];
1574
+ const postVerb = [];
1575
+ const sortedRoles = sortRolesByWordOrder(schema.roles, profile.wordOrder);
1576
+ for (const role of sortedRoles) {
1571
1577
  const marker = getMarkerForRole(role, profile);
1572
1578
  const roleName = `{${role.role}}`;
1579
+ const bucket = isSOV && role.sovSlot === "postVerb" ? postVerb : preVerb;
1573
1580
  if (marker) {
1574
1581
  const markerInfo = profile.roleMarkers?.[role.role];
1575
1582
  if (markerInfo?.position === "after") {
1576
- parts.push(`${roleName} ${marker}`);
1583
+ bucket.push(`${roleName} ${marker}`);
1577
1584
  } else {
1578
- parts.push(`${marker} ${roleName}`);
1585
+ bucket.push(`${marker} ${roleName}`);
1579
1586
  }
1580
1587
  } else {
1581
- parts.push(roleName);
1588
+ bucket.push(roleName);
1582
1589
  }
1583
1590
  }
1584
- if (profile.wordOrder === "SOV") {
1585
- parts.push(keyword);
1586
- }
1587
- return parts.join(" ");
1591
+ return (isSOV ? [...preVerb, keyword, ...postVerb] : [keyword, ...preVerb, ...postVerb]).join(
1592
+ " "
1593
+ );
1588
1594
  }
1589
1595
  function generatePatternVariants(schema, profile, config = defaultConfig) {
1590
1596
  const patterns = [];
@@ -1592,6 +1598,119 @@ function generatePatternVariants(schema, profile, config = defaultConfig) {
1592
1598
  return patterns;
1593
1599
  }
1594
1600
 
1601
+ // src/generation/renderer.ts
1602
+ function lookupKeyword(keywords, action, language) {
1603
+ return keywords[action]?.[language] ?? action;
1604
+ }
1605
+ function lookupMarker(markers, marker, language) {
1606
+ return markers[marker]?.[language] ?? marker;
1607
+ }
1608
+ function buildPhrase(...parts) {
1609
+ return parts.filter(Boolean).join(" ");
1610
+ }
1611
+ function buildTablesFromProfiles(schemas, profiles) {
1612
+ const keywords = {};
1613
+ const markers = {};
1614
+ for (const profile of profiles) {
1615
+ for (const [action, kw] of Object.entries(profile.keywords)) {
1616
+ if (!keywords[action]) keywords[action] = {};
1617
+ keywords[action][profile.code] = kw.primary;
1618
+ }
1619
+ }
1620
+ for (const schema of schemas) {
1621
+ for (const role of schema.roles) {
1622
+ if (role.markerOverride) {
1623
+ const markerKey = role.role;
1624
+ if (!markers[markerKey]) markers[markerKey] = {};
1625
+ for (const [lang, markerText] of Object.entries(role.markerOverride)) {
1626
+ markers[markerKey][lang] = markerText;
1627
+ }
1628
+ }
1629
+ }
1630
+ }
1631
+ for (const profile of profiles) {
1632
+ if (profile.roleMarkers) {
1633
+ for (const [markerKey, markerDef] of Object.entries(profile.roleMarkers)) {
1634
+ if (!markers[markerKey]) markers[markerKey] = {};
1635
+ markers[markerKey][profile.code] = markerDef.primary;
1636
+ }
1637
+ }
1638
+ }
1639
+ return { keywords, markers };
1640
+ }
1641
+ function detectWordOrders(profiles) {
1642
+ const sovLanguages = /* @__PURE__ */ new Set();
1643
+ const vsoLanguages = /* @__PURE__ */ new Set();
1644
+ for (const profile of profiles) {
1645
+ if (profile.wordOrder === "SOV") sovLanguages.add(profile.code);
1646
+ if (profile.wordOrder === "VSO") vsoLanguages.add(profile.code);
1647
+ }
1648
+ return { sovLanguages, vsoLanguages };
1649
+ }
1650
+ function createSchemaRenderer(schemas, profiles) {
1651
+ const tables = buildRenderTables(schemas, profiles);
1652
+ return {
1653
+ render(node, language) {
1654
+ return renderFromSchema(node, language, tables) ?? node.action;
1655
+ }
1656
+ };
1657
+ }
1658
+ function createDomainRenderer(config) {
1659
+ const { schemas, profiles, overrides } = config;
1660
+ let tables;
1661
+ return (node, language) => {
1662
+ const override = overrides?.[node.action];
1663
+ if (override) return override(node, language);
1664
+ tables ?? (tables = buildRenderTables(schemas, profiles));
1665
+ return renderFromSchema(node, language, tables);
1666
+ };
1667
+ }
1668
+ function buildRenderTables(schemas, profiles) {
1669
+ const { keywords, markers } = buildTablesFromProfiles(schemas, profiles);
1670
+ const { sovLanguages } = detectWordOrders(profiles);
1671
+ const markerPositions = {};
1672
+ for (const profile of profiles) {
1673
+ for (const [role, markerDef] of Object.entries(profile.roleMarkers ?? {})) {
1674
+ if (!markerDef.position) continue;
1675
+ markerPositions[role] ?? (markerPositions[role] = {});
1676
+ markerPositions[role][profile.code] = markerDef.position;
1677
+ }
1678
+ }
1679
+ const schemaMap = /* @__PURE__ */ new Map();
1680
+ for (const s of schemas) schemaMap.set(s.action, s);
1681
+ return { keywords, markers, markerPositions, sovLanguages, schemaMap };
1682
+ }
1683
+ function renderFromSchema(node, language, tables) {
1684
+ const schema = tables.schemaMap.get(node.action);
1685
+ if (!schema) return null;
1686
+ const keyword = lookupKeyword(tables.keywords, node.action, language);
1687
+ const isSOV = tables.sovLanguages.has(language);
1688
+ const ordered = sortRolesByWordOrder([...schema.roles], isSOV ? "SOV" : "SVO");
1689
+ const preVerb = [];
1690
+ const postVerb = [];
1691
+ for (const role of ordered) {
1692
+ let value = (0, import_intent.extractRoleValue)(node, role.role);
1693
+ if (!value && role.default !== void 0) {
1694
+ value = String((0, import_intent.extractValue)(role.default));
1695
+ }
1696
+ if (!value) continue;
1697
+ if (role.quoteMultiword && /\s/.test(value) && !/^(["']).*\1$/.test(value)) {
1698
+ value = `"${value}"`;
1699
+ }
1700
+ const markerText = role.renderOverride?.[language] ?? role.markerOverride?.[language] ?? role.renderOverride?.["*"] ?? tables.markers[role.role]?.[language];
1701
+ const markerPosition = role.markerPositionOverride?.[language] ?? role.markerPosition ?? tables.markerPositions[role.role]?.[language] ?? (isSOV ? "after" : "before");
1702
+ const bucket = isSOV && role.sovSlot === "postVerb" ? postVerb : preVerb;
1703
+ if (markerText && markerPosition === "before") {
1704
+ bucket.push(markerText, value);
1705
+ } else if (markerText) {
1706
+ bucket.push(value, markerText);
1707
+ } else {
1708
+ bucket.push(value);
1709
+ }
1710
+ }
1711
+ return isSOV ? buildPhrase(...preVerb, keyword, ...postVerb) : buildPhrase(keyword, ...preVerb, ...postVerb);
1712
+ }
1713
+
1595
1714
  // src/grammar/types.ts
1596
1715
  function reorderRoles(roles, targetOrder) {
1597
1716
  const result = [];
@@ -2028,6 +2147,31 @@ var LatinExtendedIdentifierExtractor = class {
2028
2147
  return { value: input.slice(position, end), length: end - position };
2029
2148
  }
2030
2149
  };
2150
+ var SELECTOR_BODY_CHAR = /[\p{L}\p{N}_-]/u;
2151
+ var PARTICLE_SCRIPT_CHAR = /[\p{sc=Han}\p{sc=Hiragana}\p{sc=Katakana}\p{sc=Hangul}]/u;
2152
+ function isSelectorBodyChar(char) {
2153
+ return SELECTOR_BODY_CHAR.test(char) && !PARTICLE_SCRIPT_CHAR.test(char);
2154
+ }
2155
+ var CssSelectorExtractor = class {
2156
+ constructor() {
2157
+ this.name = "css-selector";
2158
+ }
2159
+ canExtract(input, position) {
2160
+ const char = input[position];
2161
+ if (char !== "#" && char !== ".") return false;
2162
+ const next = input[position + 1];
2163
+ if (next === void 0) return false;
2164
+ return next === "_" || next === "-" || /\p{L}/u.test(next) && !PARTICLE_SCRIPT_CHAR.test(next);
2165
+ }
2166
+ extract(input, position) {
2167
+ let end = position + 1;
2168
+ while (end < input.length && isSelectorBodyChar(input[end])) {
2169
+ end++;
2170
+ }
2171
+ if (end === position + 1) return null;
2172
+ return { value: input.slice(position, end), length: end - position };
2173
+ }
2174
+ };
2031
2175
  var WhitespaceExtractor = class {
2032
2176
  constructor() {
2033
2177
  this.name = "whitespace";
@@ -2767,10 +2911,17 @@ function semanticValueToAST(value) {
2767
2911
 
2768
2912
  // src/api/create-dsl.ts
2769
2913
  var DSLRegistry = class {
2770
- constructor(config) {
2914
+ /**
2915
+ * @param partiallyTranslatedActions Actions permitted to lack a keyword in
2916
+ * some configured languages — extension commands, which may be added for a
2917
+ * subset of the DSL's languages. A built-in command missing a keyword stays
2918
+ * an error, since that is a domain-authoring mistake.
2919
+ */
2920
+ constructor(config, partiallyTranslatedActions = /* @__PURE__ */ new Set()) {
2771
2921
  this.patterns = /* @__PURE__ */ new Map();
2772
2922
  this.tokenizers = /* @__PURE__ */ new Map();
2773
2923
  this.schemas = config.schemas;
2924
+ this.partiallyTranslatedActions = partiallyTranslatedActions;
2774
2925
  for (const lang of config.languages) {
2775
2926
  this.registerLanguage(lang);
2776
2927
  }
@@ -2779,6 +2930,9 @@ var DSLRegistry = class {
2779
2930
  this.tokenizers.set(lang.code, lang.tokenizer);
2780
2931
  const patterns = [];
2781
2932
  for (const schema of this.schemas) {
2933
+ if (this.partiallyTranslatedActions.has(schema.action) && !lang.patternProfile.keywords[schema.action]) {
2934
+ continue;
2935
+ }
2782
2936
  const pattern = generatePattern(schema, lang.patternProfile);
2783
2937
  patterns.push(pattern);
2784
2938
  }
@@ -2795,13 +2949,25 @@ var DSLRegistry = class {
2795
2949
  }
2796
2950
  };
2797
2951
  var MultilingualDSLImpl = class {
2798
- constructor(config, registry, transformer) {
2952
+ constructor(config, registry, transformer, profileProvider) {
2799
2953
  this.registry = registry;
2800
2954
  this.matcher = new PatternMatcher();
2801
2955
  this.transformer = transformer;
2956
+ this.profileProvider = profileProvider;
2802
2957
  if (config.codeGenerator) {
2803
2958
  this.codeGenerator = config.codeGenerator;
2804
2959
  }
2960
+ if (config.renderer) {
2961
+ this.domainRenderer = config.renderer;
2962
+ }
2963
+ this.extensionRenderers = /* @__PURE__ */ new Map();
2964
+ for (const extension of config.extensions ?? []) {
2965
+ if (extension.render) this.extensionRenderers.set(extension.schema.action, extension.render);
2966
+ }
2967
+ this.schemaFallbackRenderer = createDomainRenderer({
2968
+ schemas: config.schemas,
2969
+ profiles: config.languages.map((l) => l.patternProfile)
2970
+ });
2805
2971
  const schemaMap = new Map(config.schemas.map((s) => [s.action, s]));
2806
2972
  this.schemaLookup = {
2807
2973
  getSchema: (action) => schemaMap.get(action)
@@ -2863,6 +3029,13 @@ var MultilingualDSLImpl = class {
2863
3029
  }
2864
3030
  translate(input, fromLanguage, toLanguage) {
2865
3031
  if ((0, import_intent3.isExplicitSyntax)(input)) return input;
3032
+ for (const language of [fromLanguage, toLanguage]) {
3033
+ if (!this.profileProvider.getProfile(language)) {
3034
+ throw new Error(
3035
+ `translate() requires a grammar profile for language "${language}", but none is configured. Set 'grammarProfile' on the LanguageConfig for "${language}" in createMultilingualDSL() (parse/validate/compile do not need it), or inject a custom 'profileProvider'.`
3036
+ );
3037
+ }
3038
+ }
2866
3039
  return this.transformer.transform(input, fromLanguage, toLanguage);
2867
3040
  }
2868
3041
  compile(input, language) {
@@ -2891,6 +3064,18 @@ var MultilingualDSLImpl = class {
2891
3064
  };
2892
3065
  }
2893
3066
  }
3067
+ render(node, language) {
3068
+ const extensionRenderer = this.extensionRenderers.get(node.action);
3069
+ if (extensionRenderer) {
3070
+ const rendered = extensionRenderer(node, language);
3071
+ if (rendered != null) return rendered;
3072
+ }
3073
+ if (this.domainRenderer) {
3074
+ const rendered = this.domainRenderer(node, language);
3075
+ if (rendered != null) return rendered;
3076
+ }
3077
+ return this.schemaFallbackRenderer(node, language);
3078
+ }
2894
3079
  getSupportedLanguages() {
2895
3080
  return this.registry.getSupportedLanguages();
2896
3081
  }
@@ -2915,15 +3100,80 @@ function createDefaultProfileProvider(config) {
2915
3100
  }
2916
3101
  return new InMemoryProfileProvider(profiles);
2917
3102
  }
3103
+ function applyExtensions(config) {
3104
+ const extensions = config.extensions;
3105
+ if (!extensions || extensions.length === 0) return config;
3106
+ const configuredLanguages = new Set(config.languages.map((l) => l.code));
3107
+ const actions = new Set(config.schemas.map((s) => s.action));
3108
+ for (const extension of extensions) {
3109
+ const { action } = extension.schema;
3110
+ if (actions.has(action)) {
3111
+ throw new Error(
3112
+ `Extension command "${action}" collides with a command this DSL already defines. Pick a different action name.`
3113
+ );
3114
+ }
3115
+ actions.add(action);
3116
+ for (const code of Object.keys(extension.vocabulary)) {
3117
+ if (!configuredLanguages.has(code)) {
3118
+ throw new Error(
3119
+ `Extension command "${action}" supplies vocabulary for language "${code}", which is not configured on this DSL. Configured languages: ${[...configuredLanguages].join(", ")}.`
3120
+ );
3121
+ }
3122
+ }
3123
+ }
3124
+ const schemas = [...config.schemas, ...extensions.map((e) => e.schema)];
3125
+ const languages = config.languages.map((lang) => {
3126
+ const keywords = { ...lang.patternProfile.keywords };
3127
+ let roleMarkers = lang.patternProfile.roleMarkers;
3128
+ for (const extension of extensions) {
3129
+ const vocabulary = extension.vocabulary[lang.code];
3130
+ if (!vocabulary) continue;
3131
+ keywords[extension.schema.action] = { ...vocabulary.keyword };
3132
+ if (vocabulary.roleMarkers) {
3133
+ roleMarkers = { ...roleMarkers, ...vocabulary.roleMarkers };
3134
+ }
3135
+ }
3136
+ return {
3137
+ ...lang,
3138
+ patternProfile: {
3139
+ ...lang.patternProfile,
3140
+ keywords,
3141
+ ...roleMarkers !== void 0 && { roleMarkers }
3142
+ }
3143
+ };
3144
+ });
3145
+ const extensionGenerators = new Map(
3146
+ extensions.filter((e) => e.generate).map((e) => [e.schema.action, e.generate])
3147
+ );
3148
+ const baseGenerator = config.codeGenerator;
3149
+ const codeGenerator = extensionGenerators.size > 0 ? {
3150
+ generate(node) {
3151
+ const generate = extensionGenerators.get(node.action);
3152
+ if (generate) return generate(node);
3153
+ if (!baseGenerator) {
3154
+ throw new Error(`No code generator for action "${node.action}"`);
3155
+ }
3156
+ return baseGenerator.generate(node);
3157
+ }
3158
+ } : baseGenerator;
3159
+ return {
3160
+ ...config,
3161
+ schemas,
3162
+ languages,
3163
+ ...codeGenerator !== void 0 && { codeGenerator }
3164
+ };
3165
+ }
2918
3166
  function createMultilingualDSL(config) {
2919
- const dictionary = config.dictionary ?? createDefaultDictionary(config);
2920
- const profileProvider = config.profileProvider ?? createDefaultProfileProvider(config);
3167
+ const effectiveConfig = applyExtensions(config);
3168
+ const extensionActions = new Set((config.extensions ?? []).map((e) => e.schema.action));
3169
+ const dictionary = effectiveConfig.dictionary ?? createDefaultDictionary(effectiveConfig);
3170
+ const profileProvider = effectiveConfig.profileProvider ?? createDefaultProfileProvider(effectiveConfig);
2921
3171
  const transformer = new GrammarTransformer({
2922
3172
  dictionary,
2923
3173
  profileProvider
2924
3174
  });
2925
- const registry = new DSLRegistry(config);
2926
- return new MultilingualDSLImpl(config, registry, transformer);
3175
+ const registry = new DSLRegistry(effectiveConfig, extensionActions);
3176
+ return new MultilingualDSLImpl(effectiveConfig, registry, transformer, profileProvider);
2927
3177
  }
2928
3178
 
2929
3179
  // src/prompts/prompt-generator.ts
@@ -3010,7 +3260,9 @@ function buildProtocolSection() {
3010
3260
  3. No spaces around the colon in role:value pairs
3011
3261
  4. Strings with spaces must be quoted: \`patient:"hello world"\`
3012
3262
  5. Selectors start with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`
3013
- 6. Output must be valid bracket syntax: \`[action role:value ...]\``;
3263
+ 6. A selector containing a space, a combinator (\`>\` \`+\` \`~\`), or a comma must use a selector literal: \`patient:<ul > li/>\`, \`patient:<.a, .b/>\`
3264
+ 7. Inside a structural role (\`body\`, \`then\`, \`else\`, \`condition\`, \`loop-body\`, \`variable\`, \`catch\`, \`finally\`) a \`[...]\` value is always a nested command. Write an attribute selector there as \`condition:<[data-active]/>\`
3265
+ 8. Output must be valid bracket syntax: \`[action role:value ...]\``;
3014
3266
  return {
3015
3267
  id: "protocol",
3016
3268
  title: "LSE Protocol",
@@ -3023,6 +3275,7 @@ function buildValueTypeSection() {
3023
3275
 
3024
3276
  | Type | Syntax | Example |
3025
3277
  |------|--------|---------|
3278
+ | Selector literal | Delimited by \`<\` and \`/>\` | \`<ul > li/>\`, \`<.a, .b/>\`, \`<[data-id]/>\` |
3026
3279
  | Selector | Starts with \`#\` \`.\` \`[\` \`@\` \`*\` | \`#button\`, \`.active\`, \`[data-id]\` |
3027
3280
  | String | Quoted with \`"\` or \`'\` | \`"hello world"\`, \`'json'\` |
3028
3281
  | Boolean | Exact: \`true\` / \`false\` | \`visible:true\` |
@@ -3891,6 +4144,11 @@ var DomainRegistry = class {
3891
4144
  rendered = typeof renderer === "function" ? renderer(node, to) : renderer.render(node, to);
3892
4145
  } catch {
3893
4146
  }
4147
+ } else if (dsl.render) {
4148
+ try {
4149
+ rendered = dsl.render(node, to);
4150
+ } catch {
4151
+ }
3894
4152
  }
3895
4153
  const roles = {};
3896
4154
  for (const [key, value] of node.roles) {
@@ -4448,96 +4706,6 @@ async function registryToAOTBackends(registry) {
4448
4706
  // src/schema/command-schema.ts
4449
4707
  var import_intent6 = require("@lokascript/intent");
4450
4708
 
4451
- // src/generation/renderer.ts
4452
- function lookupKeyword(keywords, action, language) {
4453
- return keywords[action]?.[language] ?? action;
4454
- }
4455
- function lookupMarker(markers, marker, language) {
4456
- return markers[marker]?.[language] ?? marker;
4457
- }
4458
- function buildPhrase(...parts) {
4459
- return parts.filter(Boolean).join(" ");
4460
- }
4461
- function buildTablesFromProfiles(schemas, profiles) {
4462
- const keywords = {};
4463
- const markers = {};
4464
- for (const profile of profiles) {
4465
- for (const [action, kw] of Object.entries(profile.keywords)) {
4466
- if (!keywords[action]) keywords[action] = {};
4467
- keywords[action][profile.code] = kw.primary;
4468
- }
4469
- }
4470
- for (const schema of schemas) {
4471
- for (const role of schema.roles) {
4472
- if (role.markerOverride) {
4473
- const markerKey = role.role;
4474
- if (!markers[markerKey]) markers[markerKey] = {};
4475
- for (const [lang, markerText] of Object.entries(role.markerOverride)) {
4476
- markers[markerKey][lang] = markerText;
4477
- }
4478
- }
4479
- }
4480
- }
4481
- for (const profile of profiles) {
4482
- if (profile.roleMarkers) {
4483
- for (const [markerKey, markerDef] of Object.entries(profile.roleMarkers)) {
4484
- if (!markers[markerKey]) markers[markerKey] = {};
4485
- markers[markerKey][profile.code] = markerDef.primary;
4486
- }
4487
- }
4488
- }
4489
- return { keywords, markers };
4490
- }
4491
- function detectWordOrders(profiles) {
4492
- const sovLanguages = /* @__PURE__ */ new Set();
4493
- const vsoLanguages = /* @__PURE__ */ new Set();
4494
- for (const profile of profiles) {
4495
- if (profile.wordOrder === "SOV") sovLanguages.add(profile.code);
4496
- if (profile.wordOrder === "VSO") vsoLanguages.add(profile.code);
4497
- }
4498
- return { sovLanguages, vsoLanguages };
4499
- }
4500
- function createSchemaRenderer(schemas, profiles) {
4501
- const { keywords, markers } = buildTablesFromProfiles(schemas, profiles);
4502
- const { sovLanguages } = detectWordOrders(profiles);
4503
- const schemaMap = /* @__PURE__ */ new Map();
4504
- for (const s of schemas) schemaMap.set(s.action, s);
4505
- return {
4506
- render(node, language) {
4507
- const schema = schemaMap.get(node.action);
4508
- if (!schema) return node.action;
4509
- const keyword = lookupKeyword(keywords, node.action, language);
4510
- const isSOV = sovLanguages.has(language);
4511
- const roleParts = [];
4512
- for (const role of schema.roles) {
4513
- const value = (0, import_intent.extractRoleValue)(node, role.role);
4514
- if (!value && !role.required) continue;
4515
- const markerText = role.markerOverride?.[language] ?? markers[role.role]?.[language] ?? void 0;
4516
- roleParts.push({
4517
- ...markerText != null && { marker: markerText },
4518
- value: value || "",
4519
- role
4520
- });
4521
- }
4522
- const parts = [];
4523
- if (isSOV) {
4524
- for (const rp of roleParts) {
4525
- if (rp.value) parts.push(rp.value);
4526
- if (rp.marker) parts.push(rp.marker);
4527
- }
4528
- parts.push(keyword);
4529
- } else {
4530
- parts.push(keyword);
4531
- for (const rp of roleParts) {
4532
- if (rp.marker) parts.push(rp.marker);
4533
- if (rp.value) parts.push(rp.value);
4534
- }
4535
- }
4536
- return buildPhrase(...parts);
4537
- }
4538
- };
4539
- }
4540
-
4541
4709
  // src/generation/diagnostics.ts
4542
4710
  var import_intent7 = require("@lokascript/intent");
4543
4711
 
@@ -4647,6 +4815,9 @@ function isQuote(char) {
4647
4815
  function isDigit(char) {
4648
4816
  return /\d/.test(char);
4649
4817
  }
4818
+ function stripOptionalDiacritics(word) {
4819
+ return word.replace(/[ً-ْٰ]/g, "");
4820
+ }
4650
4821
  function isAsciiLetter(char) {
4651
4822
  return /[a-zA-Z]/.test(char);
4652
4823
  }
@@ -5218,7 +5389,39 @@ var _BaseTokenizer = class _BaseTokenizer {
5218
5389
  pos++;
5219
5390
  }
5220
5391
  }
5221
- return new TokenStreamImpl(tokens, this.language);
5392
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
5393
+ }
5394
+ /**
5395
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
5396
+ *
5397
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
5398
+ * preceded by an identifier is a qualifier (custom event namespace), not a
5399
+ * sigil. The English tokenizer already merges these inside
5400
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
5401
+ * same stream. Strict position adjacency is the discriminator: whitespace
5402
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
5403
+ * local-variable reference survives untouched.
5404
+ *
5405
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
5406
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
5407
+ * COLON_QUALIFIER, so this pass is a no-op for them.
5408
+ */
5409
+ mergeColonQualifiedNames(tokens) {
5410
+ const out = [];
5411
+ for (const tok of tokens) {
5412
+ const prev = out[out.length - 1];
5413
+ if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
5414
+ const merged = prev.value + tok.value;
5415
+ out[out.length - 1] = createToken(
5416
+ merged,
5417
+ this.classifyToken(merged),
5418
+ createPosition(prev.position.start, tok.position.end)
5419
+ );
5420
+ continue;
5421
+ }
5422
+ out.push(tok);
5423
+ }
5424
+ return out;
5222
5425
  }
5223
5426
  /**
5224
5427
  * Classify an unknown character when no extractor matches.
@@ -5351,7 +5554,7 @@ var _BaseTokenizer = class _BaseTokenizer {
5351
5554
  * @returns Word without diacritics
5352
5555
  */
5353
5556
  removeDiacritics(word) {
5354
- return word.replace(/[\u064B-\u0652\u0670]/g, "");
5557
+ return stripOptionalDiacritics(word);
5355
5558
  }
5356
5559
  /**
5357
5560
  * Try to match a keyword from profile at the current position.
@@ -5442,24 +5645,40 @@ var _BaseTokenizer = class _BaseTokenizer {
5442
5645
  });
5443
5646
  }
5444
5647
  /**
5445
- * Look up a keyword by native word (case-insensitive).
5648
+ * Look up a keyword by native word (case-insensitive, diacritic-insensitive).
5446
5649
  * O(1) lookup using the keyword map.
5447
5650
  *
5651
+ * The map is INDEXED both with and without diacritics (see
5652
+ * `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
5653
+ * that: it lets a surface form carrying harakat the profile does not happen to
5654
+ * spell still find its entry. Only consulted after the exact lookup misses, so
5655
+ * every previously-matching word resolves byte-identically.
5656
+ *
5657
+ * Half-implementing this — indexing stripped but querying exact — is what made
5658
+ * diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
5659
+ * `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
5660
+ * exists to prevent exactly that handed the word on, and the single-char `ب`
5661
+ * bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
5662
+ *
5448
5663
  * @param native - Native word to look up
5449
5664
  * @returns KeywordEntry if found, undefined otherwise
5450
5665
  */
5451
5666
  lookupKeyword(native) {
5452
- return this.profileKeywordMap.get(native.toLowerCase());
5667
+ const exact = this.profileKeywordMap.get(native.toLowerCase());
5668
+ if (exact) return exact;
5669
+ const stripped = this.removeDiacritics(native);
5670
+ if (stripped === native) return void 0;
5671
+ return this.profileKeywordMap.get(stripped.toLowerCase());
5453
5672
  }
5454
5673
  /**
5455
- * Check if a word is a known keyword (case-insensitive).
5456
- * O(1) lookup using the keyword map.
5674
+ * Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
5675
+ * O(1) lookup using the keyword map. See {@link lookupKeyword}.
5457
5676
  *
5458
5677
  * @param native - Native word to check
5459
5678
  * @returns true if the word is a keyword
5460
5679
  */
5461
5680
  isKeyword(native) {
5462
- return this.profileKeywordMap.has(native.toLowerCase());
5681
+ return this.lookupKeyword(native) !== void 0;
5463
5682
  }
5464
5683
  /**
5465
5684
  * Set the morphological normalizer for this tokenizer.
@@ -5724,6 +5943,14 @@ var _BaseTokenizer = class _BaseTokenizer {
5724
5943
  return null;
5725
5944
  }
5726
5945
  };
5946
+ /**
5947
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
5948
+ * token that already carries a qualifier never merges again — `a:b:c` yields
5949
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
5950
+ */
5951
+ _BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
5952
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
5953
+ _BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
5727
5954
  /**
5728
5955
  * Configuration for native language time units.
5729
5956
  * Maps patterns to their standard suffix (ms, s, m, h).
@@ -6429,6 +6656,7 @@ function accumulateIndented(statements, config) {
6429
6656
  BaseMorphologicalNormalizer,
6430
6657
  BaseTokenizer,
6431
6658
  CrossDomainDispatcher,
6659
+ CssSelectorExtractor,
6432
6660
  DEFAULT_OPERATORS,
6433
6661
  DEFAULT_PUNCTUATION,
6434
6662
  DEFAULT_REFERENCES,
@@ -6465,6 +6693,7 @@ function accumulateIndented(statements, config) {
6465
6693
  createCompoundNode,
6466
6694
  createConditionalNode,
6467
6695
  createDiagnosticCollector,
6696
+ createDomainRenderer,
6468
6697
  createEventHandlerNode,
6469
6698
  createExpression,
6470
6699
  createFlag,
@@ -6556,6 +6785,8 @@ function accumulateIndented(statements, config) {
6556
6785
  semanticNodeToJSON,
6557
6786
  semanticNodeToRuntimeAST,
6558
6787
  semanticValueToAST,
6788
+ sortRolesByWordOrder,
6789
+ stripOptionalDiacritics,
6559
6790
  synthesizeFromSchemas,
6560
6791
  toEnvelopeJSON,
6561
6792
  toJSONL,