@lokascript/framework 2.11.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,7 +22,7 @@ import type { PromptGeneratorConfig, GeneratedPrompt } from './types';
22
22
  * @example
23
23
  * ```typescript
24
24
  * import { generatePrompt } from '@lokascript/framework';
25
- * import { allSchemas } from '@lokascript/domain-flow';
25
+ * import { allSchemas } from '@lokascript/domains/flow';
26
26
  *
27
27
  * const prompt = generatePrompt({
28
28
  * domain: 'flow',
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/framework",
3
- "version": "2.11.0",
3
+ "version": "3.0.0",
4
4
  "description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
5
5
  "type": "module",
6
6
  "main": "dist/index.cjs",
@@ -96,7 +96,7 @@
96
96
  "author": "LokaScript Contributors",
97
97
  "license": "MIT",
98
98
  "dependencies": {
99
- "@lokascript/intent": "^2.11.0"
99
+ "@lokascript/intent": "^3.0.0"
100
100
  },
101
101
  "devDependencies": {
102
102
  "@lokascript/domains": "^2.11.1",
@@ -18,8 +18,8 @@
18
18
  * languages: ['en', 'es', 'ja', 'ar'],
19
19
  * inputLabel: 'query',
20
20
  * inputDescription: 'SQL query in natural language',
21
- * getDSL: () => import('@lokascript/domain-sql').then(m => m.createSQLDSL()),
22
- * getRenderer: () => import('@lokascript/domain-sql').then(m => m.renderSQL),
21
+ * getDSL: () => import('@lokascript/domains/sql').then(m => m.createSQLDSL()),
22
+ * getRenderer: () => import('@lokascript/domains/sql').then(m => m.renderSQL),
23
23
  * });
24
24
  *
25
25
  * // Auto-generate MCP tool definitions
@@ -0,0 +1,42 @@
1
+ /**
2
+ * A word-glued apostrophe is the possessive marker, not a string opener.
3
+ *
4
+ * `(#price's value * #qty's value)` used to lex `'s value * #qty'` as ONE
5
+ * string literal — the first apostrophe opened a string that the second closed
6
+ * — which swallowed the operator and hid the property noun from translation in
7
+ * every language that renders the `'s` fallback (hi `#price's मान`, pl
8
+ * `#price's wartość`, …), so their English renders carried a non-ASCII
9
+ * identifier the canonical parser rejects ("Unknown token: ś"). A quote that
10
+ * opens a string always follows whitespace, punctuation, or the start of input.
11
+ */
12
+ import { describe, it, expect } from 'vitest';
13
+ import { StringLiteralExtractor } from '../../interfaces/value-extractor';
14
+
15
+ const extractor = new StringLiteralExtractor();
16
+
17
+ describe('StringLiteralExtractor: apostrophe after a word character', () => {
18
+ it.each([
19
+ ["#price's value", 6],
20
+ ["it's value", 2],
21
+ ["(#qty's value)", 5],
22
+ ["items[0]'s value", 8],
23
+ ])('does not open a string at the possessive in %s', (input, position) => {
24
+ expect(input[position]).toBe("'");
25
+ expect(extractor.canExtract(input, position)).toBe(false);
26
+ });
27
+
28
+ it.each([
29
+ ["put 'hello' into me", 4],
30
+ ["log 'a' + 'b'", 4],
31
+ ["log 'a' + 'b'", 10],
32
+ ["'leading'", 0],
33
+ ["items['a']", 6],
34
+ ["call f('x')", 7],
35
+ ])('still opens a string after whitespace/punctuation/start in %s', (input, position) => {
36
+ expect(extractor.canExtract(input, position)).toBe(true);
37
+ });
38
+
39
+ it('extracts the whole quoted string once opened', () => {
40
+ expect(extractor.extract("'hello' rest", 0)).toEqual({ value: "'hello'", length: 7 });
41
+ });
42
+ });
@@ -88,6 +88,15 @@ const MARKER_CONCEPT_NORMALIZEDS: ReadonlySet<string> = new Set([
88
88
  'without',
89
89
  ]);
90
90
 
91
+ /**
92
+ * A hyphen-JOINED keyword surface (`na-żywo`, `ao-vivo`): letters on both sides
93
+ * of every hyphen. A leading/trailing hyphen (qu's `-kama` / `-manta` suffix
94
+ * alternatives) or a bare `-` is not one — those stay with the word walk.
95
+ */
96
+ function isHyphenatedWord(native: string): boolean {
97
+ return native.includes('-') && /^[\p{L}\p{N}_]+(-[\p{L}\p{N}_]+)+$/u.test(native);
98
+ }
99
+
91
100
  // =============================================================================
92
101
  // Types
93
102
  // =============================================================================
@@ -561,8 +570,15 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
561
570
  (a, b) => b.native.length - a.native.length
562
571
  );
563
572
 
564
- // Multi-word (space-containing) keywords, for longest-phrase matching at a
565
- // token boundary. Already longest-first (profileKeywords is sorted above).
573
+ // Multi-word keywords — space-containing (hi `मेल खाता`, vi `chuyển đổi`)
574
+ // or HYPHENATED (pl `na-żywo`, pt `ao-vivo`, fr `point-arrêt`) — for
575
+ // longest-phrase matching at a token boundary. Already longest-first
576
+ // (profileKeywords is sorted above). Hyphenated ones were left to the word
577
+ // walk, which stops at the `-` and hands the parts to whatever they happen
578
+ // to be: pl `na-żywo` (live) read as `na`(→destination) + `-` + `żywo`, so
579
+ // every pl `live … koniec` render parsed back as a handler with the `live`
580
+ // action gone; es `en-vivo` / pt `ao-vivo` survived only because their
581
+ // second half is a `live` alternative.
566
582
  // Marker/modifier concepts are EXCLUDED: those are matched positionally by
567
583
  // the pattern matcher (role markers), and greedily consuming a multi-word
568
584
  // marker phrase shadows the single-word marker patterns rely on — e.g. id
@@ -572,7 +588,9 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
572
588
  // keydown, bn `তৈরি করুন`=make) are kept — the pattern matcher treats those
573
589
  // as keyword literals, so one-token matching is strictly better.
574
590
  this.multiWordKeywords = this.profileKeywords.filter(
575
- k => k.native.includes(' ') && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
591
+ k =>
592
+ (k.native.includes(' ') || isHyphenatedWord(k.native)) &&
593
+ !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
576
594
  );
577
595
 
578
596
  // Build Map for O(1) lookups (case-insensitive + diacritic-insensitive)
@@ -85,6 +85,16 @@ export class StringLiteralExtractor implements ValueExtractor {
85
85
 
86
86
  canExtract(input: string, position: number): boolean {
87
87
  const char = input[position];
88
+ // An ASCII apostrophe glued to the END of a word is the possessive marker
89
+ // (`#price's value`, `my's` never occurs but `it's`/`#qty's` do), not a
90
+ // string opener. Reading it as a quote paired it with the NEXT possessive:
91
+ // `(#price's value * #qty's value)` lexed `'s value * #qty'` as one string
92
+ // literal, which swallowed the operator and hid the property noun from
93
+ // translation in every language that renders `'s`. A quote that opens a
94
+ // string always follows whitespace, punctuation, or the start of input.
95
+ if (char === "'" && position > 0 && /[\p{L}\p{N}_)\]]/u.test(input[position - 1])) {
96
+ return false;
97
+ }
88
98
  return (
89
99
  char === '"' ||
90
100
  char === "'" ||
@@ -35,7 +35,7 @@ import type {
35
35
  * @example
36
36
  * ```typescript
37
37
  * import { generatePrompt } from '@lokascript/framework';
38
- * import { allSchemas } from '@lokascript/domain-flow';
38
+ * import { allSchemas } from '@lokascript/domains/flow';
39
39
  *
40
40
  * const prompt = generatePrompt({
41
41
  * domain: 'flow',