@lokascript/framework 2.11.1 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +301 -1
- package/dist/api/domain-registry.d.ts +2 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +7 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +7 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +7 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +7 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +7 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/prompts/prompt-generator.d.ts +1 -1
- package/package.json +2 -2
- package/src/api/domain-registry.ts +2 -2
- package/src/core/tokenization/apostrophe-possessive.test.ts +42 -0
- package/src/core/tokenization/base-tokenizer.ts +21 -3
- package/src/interfaces/value-extractor.ts +10 -0
- package/src/prompts/prompt-generator.ts +1 -1
|
@@ -22,7 +22,7 @@ import type { PromptGeneratorConfig, GeneratedPrompt } from './types';
|
|
|
22
22
|
* @example
|
|
23
23
|
* ```typescript
|
|
24
24
|
* import { generatePrompt } from '@lokascript/framework';
|
|
25
|
-
* import { allSchemas } from '@lokascript/
|
|
25
|
+
* import { allSchemas } from '@lokascript/domains/flow';
|
|
26
26
|
*
|
|
27
27
|
* const prompt = generatePrompt({
|
|
28
28
|
* domain: 'flow',
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lokascript/framework",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "3.1.0",
|
|
4
4
|
"description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.cjs",
|
|
@@ -96,7 +96,7 @@
|
|
|
96
96
|
"author": "LokaScript Contributors",
|
|
97
97
|
"license": "MIT",
|
|
98
98
|
"dependencies": {
|
|
99
|
-
"@lokascript/intent": "^
|
|
99
|
+
"@lokascript/intent": "^3.1.0"
|
|
100
100
|
},
|
|
101
101
|
"devDependencies": {
|
|
102
102
|
"@lokascript/domains": "^2.11.1",
|
|
@@ -18,8 +18,8 @@
|
|
|
18
18
|
* languages: ['en', 'es', 'ja', 'ar'],
|
|
19
19
|
* inputLabel: 'query',
|
|
20
20
|
* inputDescription: 'SQL query in natural language',
|
|
21
|
-
* getDSL: () => import('@lokascript/
|
|
22
|
-
* getRenderer: () => import('@lokascript/
|
|
21
|
+
* getDSL: () => import('@lokascript/domains/sql').then(m => m.createSQLDSL()),
|
|
22
|
+
* getRenderer: () => import('@lokascript/domains/sql').then(m => m.renderSQL),
|
|
23
23
|
* });
|
|
24
24
|
*
|
|
25
25
|
* // Auto-generate MCP tool definitions
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A word-glued apostrophe is the possessive marker, not a string opener.
|
|
3
|
+
*
|
|
4
|
+
* `(#price's value * #qty's value)` used to lex `'s value * #qty'` as ONE
|
|
5
|
+
* string literal — the first apostrophe opened a string that the second closed
|
|
6
|
+
* — which swallowed the operator and hid the property noun from translation in
|
|
7
|
+
* every language that renders the `'s` fallback (hi `#price's मान`, pl
|
|
8
|
+
* `#price's wartość`, …), so their English renders carried a non-ASCII
|
|
9
|
+
* identifier the canonical parser rejects ("Unknown token: ś"). A quote that
|
|
10
|
+
* opens a string always follows whitespace, punctuation, or the start of input.
|
|
11
|
+
*/
|
|
12
|
+
import { describe, it, expect } from 'vitest';
|
|
13
|
+
import { StringLiteralExtractor } from '../../interfaces/value-extractor';
|
|
14
|
+
|
|
15
|
+
const extractor = new StringLiteralExtractor();
|
|
16
|
+
|
|
17
|
+
describe('StringLiteralExtractor: apostrophe after a word character', () => {
|
|
18
|
+
it.each([
|
|
19
|
+
["#price's value", 6],
|
|
20
|
+
["it's value", 2],
|
|
21
|
+
["(#qty's value)", 5],
|
|
22
|
+
["items[0]'s value", 8],
|
|
23
|
+
])('does not open a string at the possessive in %s', (input, position) => {
|
|
24
|
+
expect(input[position]).toBe("'");
|
|
25
|
+
expect(extractor.canExtract(input, position)).toBe(false);
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
it.each([
|
|
29
|
+
["put 'hello' into me", 4],
|
|
30
|
+
["log 'a' + 'b'", 4],
|
|
31
|
+
["log 'a' + 'b'", 10],
|
|
32
|
+
["'leading'", 0],
|
|
33
|
+
["items['a']", 6],
|
|
34
|
+
["call f('x')", 7],
|
|
35
|
+
])('still opens a string after whitespace/punctuation/start in %s', (input, position) => {
|
|
36
|
+
expect(extractor.canExtract(input, position)).toBe(true);
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
it('extracts the whole quoted string once opened', () => {
|
|
40
|
+
expect(extractor.extract("'hello' rest", 0)).toEqual({ value: "'hello'", length: 7 });
|
|
41
|
+
});
|
|
42
|
+
});
|
|
@@ -88,6 +88,15 @@ const MARKER_CONCEPT_NORMALIZEDS: ReadonlySet<string> = new Set([
|
|
|
88
88
|
'without',
|
|
89
89
|
]);
|
|
90
90
|
|
|
91
|
+
/**
|
|
92
|
+
* A hyphen-JOINED keyword surface (`na-żywo`, `ao-vivo`): letters on both sides
|
|
93
|
+
* of every hyphen. A leading/trailing hyphen (qu's `-kama` / `-manta` suffix
|
|
94
|
+
* alternatives) or a bare `-` is not one — those stay with the word walk.
|
|
95
|
+
*/
|
|
96
|
+
function isHyphenatedWord(native: string): boolean {
|
|
97
|
+
return native.includes('-') && /^[\p{L}\p{N}_]+(-[\p{L}\p{N}_]+)+$/u.test(native);
|
|
98
|
+
}
|
|
99
|
+
|
|
91
100
|
// =============================================================================
|
|
92
101
|
// Types
|
|
93
102
|
// =============================================================================
|
|
@@ -561,8 +570,15 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
561
570
|
(a, b) => b.native.length - a.native.length
|
|
562
571
|
);
|
|
563
572
|
|
|
564
|
-
// Multi-word
|
|
565
|
-
//
|
|
573
|
+
// Multi-word keywords — space-containing (hi `मेल खाता`, vi `chuyển đổi`)
|
|
574
|
+
// or HYPHENATED (pl `na-żywo`, pt `ao-vivo`, fr `point-arrêt`) — for
|
|
575
|
+
// longest-phrase matching at a token boundary. Already longest-first
|
|
576
|
+
// (profileKeywords is sorted above). Hyphenated ones were left to the word
|
|
577
|
+
// walk, which stops at the `-` and hands the parts to whatever they happen
|
|
578
|
+
// to be: pl `na-żywo` (live) read as `na`(→destination) + `-` + `żywo`, so
|
|
579
|
+
// every pl `live … koniec` render parsed back as a handler with the `live`
|
|
580
|
+
// action gone; es `en-vivo` / pt `ao-vivo` survived only because their
|
|
581
|
+
// second half is a `live` alternative.
|
|
566
582
|
// Marker/modifier concepts are EXCLUDED: those are matched positionally by
|
|
567
583
|
// the pattern matcher (role markers), and greedily consuming a multi-word
|
|
568
584
|
// marker phrase shadows the single-word marker patterns rely on — e.g. id
|
|
@@ -572,7 +588,9 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
572
588
|
// keydown, bn `তৈরি করুন`=make) are kept — the pattern matcher treats those
|
|
573
589
|
// as keyword literals, so one-token matching is strictly better.
|
|
574
590
|
this.multiWordKeywords = this.profileKeywords.filter(
|
|
575
|
-
k =>
|
|
591
|
+
k =>
|
|
592
|
+
(k.native.includes(' ') || isHyphenatedWord(k.native)) &&
|
|
593
|
+
!MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
576
594
|
);
|
|
577
595
|
|
|
578
596
|
// Build Map for O(1) lookups (case-insensitive + diacritic-insensitive)
|
|
@@ -85,6 +85,16 @@ export class StringLiteralExtractor implements ValueExtractor {
|
|
|
85
85
|
|
|
86
86
|
canExtract(input: string, position: number): boolean {
|
|
87
87
|
const char = input[position];
|
|
88
|
+
// An ASCII apostrophe glued to the END of a word is the possessive marker
|
|
89
|
+
// (`#price's value`, `my's` never occurs but `it's`/`#qty's` do), not a
|
|
90
|
+
// string opener. Reading it as a quote paired it with the NEXT possessive:
|
|
91
|
+
// `(#price's value * #qty's value)` lexed `'s value * #qty'` as one string
|
|
92
|
+
// literal, which swallowed the operator and hid the property noun from
|
|
93
|
+
// translation in every language that renders `'s`. A quote that opens a
|
|
94
|
+
// string always follows whitespace, punctuation, or the start of input.
|
|
95
|
+
if (char === "'" && position > 0 && /[\p{L}\p{N}_)\]]/u.test(input[position - 1])) {
|
|
96
|
+
return false;
|
|
97
|
+
}
|
|
88
98
|
return (
|
|
89
99
|
char === '"' ||
|
|
90
100
|
char === "'" ||
|
|
@@ -35,7 +35,7 @@ import type {
|
|
|
35
35
|
* @example
|
|
36
36
|
* ```typescript
|
|
37
37
|
* import { generatePrompt } from '@lokascript/framework';
|
|
38
|
-
* import { allSchemas } from '@lokascript/
|
|
38
|
+
* import { allSchemas } from '@lokascript/domains/flow';
|
|
39
39
|
*
|
|
40
40
|
* const prompt = generatePrompt({
|
|
41
41
|
* domain: 'flow',
|