@lokascript/framework 2.11.1 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +120 -0
- package/dist/api/domain-registry.d.ts +2 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +7 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +7 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +7 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +7 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +7 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/prompts/prompt-generator.d.ts +1 -1
- package/package.json +2 -2
- package/src/api/domain-registry.ts +2 -2
- package/src/core/tokenization/apostrophe-possessive.test.ts +42 -0
- package/src/core/tokenization/base-tokenizer.ts +21 -3
- package/src/interfaces/value-extractor.ts +10 -0
- package/src/prompts/prompt-generator.ts +1 -1
package/dist/core/index.js
CHANGED
|
@@ -478,6 +478,9 @@ var StringLiteralExtractor = class {
|
|
|
478
478
|
}
|
|
479
479
|
canExtract(input, position) {
|
|
480
480
|
const char = input[position];
|
|
481
|
+
if (char === "'" && position > 0 && /[\p{L}\p{N}_)\]]/u.test(input[position - 1])) {
|
|
482
|
+
return false;
|
|
483
|
+
}
|
|
481
484
|
return char === '"' || char === "'" || char === "`" || char === "\u201C" || // Chinese double quote open "
|
|
482
485
|
char === "\u2018";
|
|
483
486
|
}
|
|
@@ -744,6 +747,9 @@ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
|
744
747
|
"through",
|
|
745
748
|
"without"
|
|
746
749
|
]);
|
|
750
|
+
function isHyphenatedWord(native) {
|
|
751
|
+
return native.includes("-") && /^[\p{L}\p{N}_]+(-[\p{L}\p{N}_]+)+$/u.test(native);
|
|
752
|
+
}
|
|
747
753
|
var ENGLISH_DOM_EVENT_NAMES = [
|
|
748
754
|
"click",
|
|
749
755
|
"dblclick",
|
|
@@ -1063,7 +1069,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1063
1069
|
(a, b) => b.native.length - a.native.length
|
|
1064
1070
|
);
|
|
1065
1071
|
this.multiWordKeywords = this.profileKeywords.filter(
|
|
1066
|
-
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
1072
|
+
(k) => (k.native.includes(" ") || isHyphenatedWord(k.native)) && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
1067
1073
|
);
|
|
1068
1074
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
1069
1075
|
for (const keyword of this.profileKeywords) {
|