@lokascript/framework 2.11.1 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +120 -0
- package/dist/api/domain-registry.d.ts +2 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +7 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +7 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +7 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +7 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +7 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/prompts/prompt-generator.d.ts +1 -1
- package/package.json +2 -2
- package/src/api/domain-registry.ts +2 -2
- package/src/core/tokenization/apostrophe-possessive.test.ts +42 -0
- package/src/core/tokenization/base-tokenizer.ts +21 -3
- package/src/interfaces/value-extractor.ts +10 -0
- package/src/prompts/prompt-generator.ts +1 -1
package/dist/index.cjs
CHANGED
|
@@ -1953,6 +1953,9 @@ var StringLiteralExtractor = class {
|
|
|
1953
1953
|
}
|
|
1954
1954
|
canExtract(input, position) {
|
|
1955
1955
|
const char = input[position];
|
|
1956
|
+
if (char === "'" && position > 0 && /[\p{L}\p{N}_)\]]/u.test(input[position - 1])) {
|
|
1957
|
+
return false;
|
|
1958
|
+
}
|
|
1956
1959
|
return char === '"' || char === "'" || char === "`" || char === "\u201C" || // Chinese double quote open "
|
|
1957
1960
|
char === "\u2018";
|
|
1958
1961
|
}
|
|
@@ -5227,6 +5230,9 @@ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
|
5227
5230
|
"through",
|
|
5228
5231
|
"without"
|
|
5229
5232
|
]);
|
|
5233
|
+
function isHyphenatedWord(native) {
|
|
5234
|
+
return native.includes("-") && /^[\p{L}\p{N}_]+(-[\p{L}\p{N}_]+)+$/u.test(native);
|
|
5235
|
+
}
|
|
5230
5236
|
var ENGLISH_DOM_EVENT_NAMES = [
|
|
5231
5237
|
"click",
|
|
5232
5238
|
"dblclick",
|
|
@@ -5546,7 +5552,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5546
5552
|
(a, b) => b.native.length - a.native.length
|
|
5547
5553
|
);
|
|
5548
5554
|
this.multiWordKeywords = this.profileKeywords.filter(
|
|
5549
|
-
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5555
|
+
(k) => (k.native.includes(" ") || isHyphenatedWord(k.native)) && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5550
5556
|
);
|
|
5551
5557
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5552
5558
|
for (const keyword of this.profileKeywords) {
|