scoutline 0.24.0 → 0.24.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/archive.d.ts +18 -1
- package/dist/commands/archive.d.ts.map +1 -1
- package/dist/commands/archive.js +23 -12
- package/dist/commands/archive.js.map +1 -1
- package/dist/commands/science.d.ts +6 -0
- package/dist/commands/science.d.ts.map +1 -1
- package/dist/commands/science.js +15 -4
- package/dist/commands/science.js.map +1 -1
- package/dist/commands/search.d.ts +16 -0
- package/dist/commands/search.d.ts.map +1 -1
- package/dist/commands/search.js +10 -2
- package/dist/commands/search.js.map +1 -1
- package/dist/commands/watch.d.ts +13 -0
- package/dist/commands/watch.d.ts.map +1 -1
- package/dist/commands/watch.js +17 -4
- package/dist/commands/watch.js.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +35 -29
- package/dist/index.js.map +1 -1
- package/dist/lib/context-file.d.ts +21 -0
- package/dist/lib/context-file.d.ts.map +1 -1
- package/dist/lib/context-file.js +84 -6
- package/dist/lib/context-file.js.map +1 -1
- package/dist/lib/equals-form.d.ts +37 -0
- package/dist/lib/equals-form.d.ts.map +1 -0
- package/dist/lib/equals-form.js +52 -0
- package/dist/lib/equals-form.js.map +1 -0
- package/dist/lib/investigate-claims.d.ts +11 -3
- package/dist/lib/investigate-claims.d.ts.map +1 -1
- package/dist/lib/investigate-claims.js +28 -50
- package/dist/lib/investigate-claims.js.map +1 -1
- package/dist/lib/investigate-extract.d.ts +23 -0
- package/dist/lib/investigate-extract.d.ts.map +1 -1
- package/dist/lib/investigate-extract.js +32 -7
- package/dist/lib/investigate-extract.js.map +1 -1
- package/dist/lib/investigate-planner.d.ts +5 -3
- package/dist/lib/investigate-planner.d.ts.map +1 -1
- package/dist/lib/investigate-planner.js +17 -33
- package/dist/lib/investigate-planner.js.map +1 -1
- package/dist/lib/journal.d.ts.map +1 -1
- package/dist/lib/journal.js +10 -9
- package/dist/lib/journal.js.map +1 -1
- package/dist/providers/types.d.ts +16 -1
- package/dist/providers/types.d.ts.map +1 -1
- package/dist/providers/types.js.map +1 -1
- package/dist/providers/zai/adapter.d.ts.map +1 -1
- package/dist/providers/zai/adapter.js +38 -10
- package/dist/providers/zai/adapter.js.map +1 -1
- package/dist/providers/zai/layout-parsing.d.ts +2 -2
- package/dist/providers/zai/layout-parsing.d.ts.map +1 -1
- package/dist/providers/zai/layout-parsing.js.map +1 -1
- package/package.json +1 -1
|
@@ -27,6 +27,27 @@ export declare const MAX_TERMS = 12;
|
|
|
27
27
|
* `STOPWORD_SET`; the frozen array is the shipped constant.
|
|
28
28
|
*/
|
|
29
29
|
export declare const STOPWORDS: readonly string[];
|
|
30
|
+
/** D2.3: term length bounds. Exported: `investigate-planner` shares them. */
|
|
31
|
+
export declare const MIN_TERM_CHARS = 4;
|
|
32
|
+
export declare const MAX_TERM_CHARS = 40;
|
|
33
|
+
/**
|
|
34
|
+
* True for pure-ASCII tokens — callers use this to apply the
|
|
35
|
+
* MIN_TERM_CHARS floor to ASCII only (single non-ASCII chars are
|
|
36
|
+
* meaningful word units; issue #271).
|
|
37
|
+
*/
|
|
38
|
+
export declare function isAsciiToken(token: string): boolean;
|
|
39
|
+
/**
|
|
40
|
+
* Tokenize lowercased text into term tokens: word-like units with
|
|
41
|
+
* ASCII punctuation never inside a non-ASCII token. `keepApostrophe`
|
|
42
|
+
* selects claims' grammar ("don't" stays one token) and
|
|
43
|
+
* `keepUnderscore` journal recall's (snake_case identifiers stay
|
|
44
|
+
* whole tokens) — both on the ASCII path only.
|
|
45
|
+
* Segments longer than MAX_TERM_CHARS are dropped here (not by
|
|
46
|
+
* callers): the "en" segmenter splits long unspaced CJK runs into
|
|
47
|
+
* dictionary chunks (東東×…), so a caller-side length check could
|
|
48
|
+
* admit sub-cap chunks of an over-long word.
|
|
49
|
+
*/
|
|
50
|
+
export declare function tokenizeTerms(text: string, keepApostrophe?: boolean, keepUnderscore?: boolean): string[];
|
|
30
51
|
export interface ParsedContextText {
|
|
31
52
|
/** In document order, deduped exact-trim (case-sensitive). */
|
|
32
53
|
readonly headings: readonly string[];
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"context-file.d.ts","sourceRoot":"","sources":["../../src/lib/context-file.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAKH,sEAAsE;AACtE,eAAO,MAAM,iBAAiB,SAAU,CAAC;AAEzC,wEAAwE;AACxE,eAAO,MAAM,cAAc,IAAI,CAAC;AAEhC,wCAAwC;AACxC,eAAO,MAAM,SAAS,KAAK,CAAC;AAE5B;;;GAGG;AACH,eAAO,MAAM,SAAS,EAAE,SAAS,MAAM,EAQrC,CAAC;
|
|
1
|
+
{"version":3,"file":"context-file.d.ts","sourceRoot":"","sources":["../../src/lib/context-file.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAKH,sEAAsE;AACtE,eAAO,MAAM,iBAAiB,SAAU,CAAC;AAEzC,wEAAwE;AACxE,eAAO,MAAM,cAAc,IAAI,CAAC;AAEhC,wCAAwC;AACxC,eAAO,MAAM,SAAS,KAAK,CAAC;AAE5B;;;GAGG;AACH,eAAO,MAAM,SAAS,EAAE,SAAS,MAAM,EAQrC,CAAC;AAkBH,6EAA6E;AAC7E,eAAO,MAAM,cAAc,IAAI,CAAC;AAChC,eAAO,MAAM,cAAc,KAAK,CAAC;AAkCjC;;;;GAIG;AACH,wBAAgB,YAAY,CAAC,KAAK,EAAE,MAAM,GAAG,OAAO,CAEnD;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,MAAM,EACZ,cAAc,UAAQ,EACtB,cAAc,UAAQ,GACrB,MAAM,EAAE,CAsBV;AAKD,MAAM,WAAW,iBAAiB;IAChC,8DAA8D;IAC9D,QAAQ,CAAC,QAAQ,EAAE,SAAS,MAAM,EAAE,CAAC;IACrC,sDAAsD;IACtD,QAAQ,CAAC,SAAS,EAAE,SAAS,MAAM,EAAE,CAAC;IACtC,iDAAiD;IACjD,QAAQ,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,CAAC;IAClC,kEAAkE;IAClE,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;CACxC;AAOD;;;;;;;;;GASG;AACH,wBAAgB,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,iBAAiB,CAqChE;AA+DD;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS,MAAM,EAAE,CAEhE;AAED,gEAAgE;AAChE,wBAAgB,IAAI,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAK1C;AAMD;;;;;GAKG;AACH,wBAAgB,eAAe,CAAC,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAW/E;AAED,gCAAgC;AAChC,MAAM,MAAM,iBAAiB,GACzB;IAAE,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GACzB;IAAE,QAAQ,CAAC,KAAK,EAAE,IAAI,CAAA;CAAE,CAAC;AAE7B,sEAAsE;AACtE,MAAM,WAAW,eAAe;IAC9B,QAAQ,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC;IAC5C;;;;OAIG;IACH,SAAS,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC;CAC9C;AAED,MAAM,WAAW,oBAAoB;IACnC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;IACtB,qEAAqE;IACrE,QAAQ,CAAC,IAAI,CAAC,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO,CAAC;CACnC;AAED,wBAAsB,iBAAiB,CACrC,IAAI,EAAE,iBAAiB,EACvB,EAAE,EAAE,eAAe,GAClB,OAAO,CAAC,oBAAoB,CAAC,CAe/B"}
|
package/dist/lib/context-file.js
CHANGED
|
@@ -49,11 +49,80 @@ const HEADING_PATTERN = /^(#{1,6})\s+(.+)$/;
|
|
|
49
49
|
const MAX_QUESTION_CHARS = 200;
|
|
50
50
|
/** D2.4: headings longer than this are dropped from sub-queries. */
|
|
51
51
|
const MAX_SUBQUERY_HEADING_CHARS = 60;
|
|
52
|
-
/** D2.3: term length bounds. */
|
|
53
|
-
const MIN_TERM_CHARS = 4;
|
|
54
|
-
const MAX_TERM_CHARS = 40;
|
|
52
|
+
/** D2.3: term length bounds. Exported: `investigate-planner` shares them. */
|
|
53
|
+
export const MIN_TERM_CHARS = 4;
|
|
54
|
+
export const MAX_TERM_CHARS = 40;
|
|
55
55
|
/** D2.5: the appended bias segment fits within this many chars. */
|
|
56
56
|
const MAX_BIAS_APPEND_CHARS = 240;
|
|
57
|
+
// ---------------------------------------------------------------------------
|
|
58
|
+
// Issue #271: Unicode word segmentation at the shared term seam.
|
|
59
|
+
//
|
|
60
|
+
// Pure-ASCII input keeps the legacy split verbatim (byte-identical A/B —
|
|
61
|
+
// the segmenter would fuse ASCII punctuation into words like "3.14").
|
|
62
|
+
// Input containing non-ASCII goes through `Intl.Segmenter` (fixed "en"
|
|
63
|
+
// locale, granularity "word" — locale-independent for determinism; en/ja
|
|
64
|
+
// agree on CJK corpus). Non-ASCII word-like segments are kept WHOLE: a
|
|
65
|
+
// CJK word is a meaningful unit, so the 4-char ASCII minimum never
|
|
66
|
+
// applies to them (single CJK chars pass); the 40-char cap still does.
|
|
67
|
+
// Length bounds themselves are applied by the CALLERS, not here.
|
|
68
|
+
//
|
|
69
|
+
// OUT OF SCOPE (documented): negation cues stay English literals —
|
|
70
|
+
// non-Latin contradiction detection is not attempted.
|
|
71
|
+
// ---------------------------------------------------------------------------
|
|
72
|
+
/** The legacy ASCII split class (kept verbatim for the A/B pin). */
|
|
73
|
+
const ASCII_SPLIT = /[^a-z0-9]+/;
|
|
74
|
+
/** Claims' apostrophe-keeping variant of the legacy split. */
|
|
75
|
+
const ASCII_SPLIT_APOSTROPHE = /[^a-z0-9']+/;
|
|
76
|
+
/** Journal recall's underscore-keeping variant (identifiers stay whole). */
|
|
77
|
+
const ASCII_SPLIT_UNDERSCORE = /[^a-z0-9_]+/;
|
|
78
|
+
const SEGMENTER = new Intl.Segmenter("en", { granularity: "word" });
|
|
79
|
+
function isAscii(text) {
|
|
80
|
+
return /^[\x00-\x7f]*$/.test(text);
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* True for pure-ASCII tokens — callers use this to apply the
|
|
84
|
+
* MIN_TERM_CHARS floor to ASCII only (single non-ASCII chars are
|
|
85
|
+
* meaningful word units; issue #271).
|
|
86
|
+
*/
|
|
87
|
+
export function isAsciiToken(token) {
|
|
88
|
+
return isAscii(token);
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* Tokenize lowercased text into term tokens: word-like units with
|
|
92
|
+
* ASCII punctuation never inside a non-ASCII token. `keepApostrophe`
|
|
93
|
+
* selects claims' grammar ("don't" stays one token) and
|
|
94
|
+
* `keepUnderscore` journal recall's (snake_case identifiers stay
|
|
95
|
+
* whole tokens) — both on the ASCII path only.
|
|
96
|
+
* Segments longer than MAX_TERM_CHARS are dropped here (not by
|
|
97
|
+
* callers): the "en" segmenter splits long unspaced CJK runs into
|
|
98
|
+
* dictionary chunks (東東×…), so a caller-side length check could
|
|
99
|
+
* admit sub-cap chunks of an over-long word.
|
|
100
|
+
*/
|
|
101
|
+
export function tokenizeTerms(text, keepApostrophe = false, keepUnderscore = false) {
|
|
102
|
+
// ponytail: keepApostrophe (claims) and keepUnderscore (journal) are
|
|
103
|
+
// caller-exclusive today; a stacking caller adds the combined class.
|
|
104
|
+
const split = keepUnderscore
|
|
105
|
+
? ASCII_SPLIT_UNDERSCORE
|
|
106
|
+
: keepApostrophe
|
|
107
|
+
? ASCII_SPLIT_APOSTROPHE
|
|
108
|
+
: ASCII_SPLIT;
|
|
109
|
+
const lower = text.toLowerCase();
|
|
110
|
+
if (isAscii(lower)) {
|
|
111
|
+
return lower.split(split).filter((t) => t.length > 0);
|
|
112
|
+
}
|
|
113
|
+
const tokens = [];
|
|
114
|
+
for (const { segment, isWordLike } of SEGMENTER.segment(lower)) {
|
|
115
|
+
if (!isWordLike)
|
|
116
|
+
continue;
|
|
117
|
+
if (isAscii(segment)) {
|
|
118
|
+
tokens.push(...segment.split(split).filter((t) => t.length > 0));
|
|
119
|
+
}
|
|
120
|
+
else if (segment.length <= MAX_TERM_CHARS) {
|
|
121
|
+
tokens.push(segment);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return tokens;
|
|
125
|
+
}
|
|
57
126
|
/** D3 (G3): NUL byte anywhere in this prefix marks the source binary. */
|
|
58
127
|
const BINARY_SNIFF_BYTES = 8192;
|
|
59
128
|
/**
|
|
@@ -102,13 +171,22 @@ export function parseContextText(text) {
|
|
|
102
171
|
subQueries: deriveSubQueriesFromStream(stream),
|
|
103
172
|
};
|
|
104
173
|
}
|
|
105
|
-
/**
|
|
174
|
+
/**
|
|
175
|
+
* D2.3: lowercase tokens from the stream, stopword/length filtered,
|
|
176
|
+
* capped. Issue #271: tokens come from the shared Unicode tokenizer —
|
|
177
|
+
* non-ASCII segments are meaningful single units, so the ASCII
|
|
178
|
+
* MIN_TERM_CHARS floor never applies to them (single CJK chars pass);
|
|
179
|
+
* the MAX_TERM_CHARS cap applies to every token.
|
|
180
|
+
*/
|
|
106
181
|
function deriveTerms(stream) {
|
|
107
182
|
const terms = [];
|
|
108
183
|
const seen = new Set();
|
|
109
184
|
for (const item of stream) {
|
|
110
|
-
for (const token of item.value
|
|
111
|
-
if (token
|
|
185
|
+
for (const token of tokenizeTerms(item.value)) {
|
|
186
|
+
if (isAscii(token) && token.length < MIN_TERM_CHARS) {
|
|
187
|
+
continue;
|
|
188
|
+
}
|
|
189
|
+
if (token.length > MAX_TERM_CHARS) {
|
|
112
190
|
continue;
|
|
113
191
|
}
|
|
114
192
|
if (STOPWORD_SET.has(token) || seen.has(token)) {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"context-file.js","sourceRoot":"","sources":["../../src/lib/context-file.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AACzC,OAAO,EAAE,SAAS,EAAE,eAAe,EAAE,MAAM,aAAa,CAAC;AAEzD,sEAAsE;AACtE,MAAM,CAAC,MAAM,iBAAiB,GAAG,OAAO,CAAC,CAAC,UAAU;AAEpD,wEAAwE;AACxE,MAAM,CAAC,MAAM,cAAc,GAAG,CAAC,CAAC;AAEhC,wCAAwC;AACxC,MAAM,CAAC,MAAM,SAAS,GAAG,EAAE,CAAC;AAE5B;;;GAGG;AACH,MAAM,CAAC,MAAM,SAAS,GAAsB,MAAM,CAAC,MAAM,CAAC;IACxD,GAAG,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;IAClE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM;IACrE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,KAAK;IAClE,KAAK,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,OAAO;IAC7D,QAAQ,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,OAAO;IAClE,MAAM,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO,EAAE,SAAS,EAAE,SAAS,EAAE,QAAQ;IAChE,QAAQ,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO;CAC9D,CAAC,CAAC;AAEH,MAAM,YAAY,GAAwB,IAAI,GAAG,CAAC,SAAS,CAAC,CAAC;AAE7D;;;;;GAKG;AACH,MAAM,eAAe,GAAG,mBAAmB,CAAC;AAE5C,uEAAuE;AACvE,MAAM,kBAAkB,GAAG,GAAG,CAAC;AAE/B,oEAAoE;AACpE,MAAM,0BAA0B,GAAG,EAAE,CAAC;AAEtC,
|
|
1
|
+
{"version":3,"file":"context-file.js","sourceRoot":"","sources":["../../src/lib/context-file.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AACzC,OAAO,EAAE,SAAS,EAAE,eAAe,EAAE,MAAM,aAAa,CAAC;AAEzD,sEAAsE;AACtE,MAAM,CAAC,MAAM,iBAAiB,GAAG,OAAO,CAAC,CAAC,UAAU;AAEpD,wEAAwE;AACxE,MAAM,CAAC,MAAM,cAAc,GAAG,CAAC,CAAC;AAEhC,wCAAwC;AACxC,MAAM,CAAC,MAAM,SAAS,GAAG,EAAE,CAAC;AAE5B;;;GAGG;AACH,MAAM,CAAC,MAAM,SAAS,GAAsB,MAAM,CAAC,MAAM,CAAC;IACxD,GAAG,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM;IAClE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM;IACrE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,EAAE,KAAK;IAClE,KAAK,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,OAAO;IAC7D,QAAQ,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,OAAO;IAClE,MAAM,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO,EAAE,SAAS,EAAE,SAAS,EAAE,QAAQ;IAChE,QAAQ,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO;CAC9D,CAAC,CAAC;AAEH,MAAM,YAAY,GAAwB,IAAI,GAAG,CAAC,SAAS,CAAC,CAAC;AAE7D;;;;;GAKG;AACH,MAAM,eAAe,GAAG,mBAAmB,CAAC;AAE5C,uEAAuE;AACvE,MAAM,kBAAkB,GAAG,GAAG,CAAC;AAE/B,oEAAoE;AACpE,MAAM,0BAA0B,GAAG,EAAE,CAAC;AAEtC,6EAA6E;AAC7E,MAAM,CAAC,MAAM,cAAc,GAAG,CAAC,CAAC;AAChC,MAAM,CAAC,MAAM,cAAc,GAAG,EAAE,CAAC;AAEjC,mEAAmE;AACnE,MAAM,qBAAqB,GAAG,GAAG,CAAC;AAElC,8EAA8E;AAC9E,iEAAiE;AACjE,EAAE;AACF,yEAAyE;AACzE,sEAAsE;AACtE,uEAAuE;AACvE,yEAAyE;AACzE,uEAAuE;AACvE,mEAAmE;AACnE,uEAAuE;AACvE,iEAAiE;AACjE,EAAE;AACF,mEAAmE;AACnE,sDAAsD;AACtD,8EAA8E;AAE9E,oEAAoE;AACpE,MAAM,WAAW,GAAG,YAAY,CAAC;AACjC,8DAA8D;AAC9D,MAAM,sBAAsB,GAAG,aAAa,CAAC;AAC7C,4EAA4E;AAC5E,MAAM,sBAAsB,GAAG,aAAa,CAAC;AAE7C,MAAM,SAAS,GAAG,IAAI,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,EAAE,WAAW,EAAE,MAAM,EAAE,CAAC,CAAC;AAEpE,SAAS,OAAO,CAAC,IAAY;IAC3B,OAAO,gBAAgB,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACrC,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,YAAY,CAAC,KAAa;IACxC,OAAO,OAAO,CAAC,KAAK,CAAC,CAAC;AACxB,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,aAAa,CAC3B,IAAY,EACZ,cAAc,GAAG,KAAK,EACtB,cAAc,GAAG,KAAK;IAEtB,qEAAqE;IACrE,qEAAqE;IACrE,MAAM,KAAK,GAAG,cAAc;QAC1B,CAAC,CAAC,sBAAsB;QACxB,CAAC,CAAC,cAAc;YACd,CAAC,CAAC,sBAAsB;YACxB,CAAC,CAAC,WAAW,CAAC;IAClB,MAAM,KAAK,GAAG,IAAI,CAAC,WAAW,EAAE,CAAC;IACjC,IAAI,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC;QACnB,OAAO,KAAK,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACxD,CAAC;IACD,MAAM,MAAM,GAAa,EAAE,CAAC;IAC5B,KAAK,MAAM,EAAE,OAAO,EAAE,UAAU,EAAE,IAAI,SAAS,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC;QAC/D,IAAI,CAAC,UAAU;YAAE,SAAS;QAC1B,IAAI,OAAO,CAAC,OAAO,CAAC,EAAE,CAAC;YACrB,MAAM,CAAC,IAAI,CAAC,GAAG,OAAO,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC;QACnE,CAAC;aAAM,IAAI,OAAO,CAAC,MAAM,IAAI,cAAc,EAAE,CAAC;YAC5C,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;QACvB,CAAC;IACH,CAAC;IACD,OAAO,MAAM,CAAC;AAChB,CAAC;AAED,yEAAyE;AACzE,MAAM,kBAAkB,GAAG,IAAI,CAAC;AAkBhC;;;;;;;;;GASG;AACH,MAAM,UAAU,gBAAgB,CAAC,IAAY;IAC3C,MAAM,MAAM,GAAiB,EAAE,CAAC;IAChC,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,SAAS,GAAa,EAAE,CAAC;IAC/B,MAAM,YAAY,GAAG,IAAI,GAAG,EAAU,CAAC;IACvC,MAAM,aAAa,GAAG,IAAI,GAAG,EAAU,CAAC;IAExC,KAAK,MAAM,OAAO,IAAI,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC;QACvC,qEAAqE;QACrE,8DAA8D;QAC9D,MAAM,IAAI,GAAG,OAAO,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC;QACrE,MAAM,YAAY,GAAG,IAAI,CAAC,KAAK,CAAC,eAAe,CAAC,CAAC;QACjD,IAAI,YAAY,EAAE,CAAC;YACjB,MAAM,OAAO,GAAG,YAAY,CAAC,CAAC,CAAE,CAAC,IAAI,EAAE,CAAC;YACxC,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,IAAI,CAAC,YAAY,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,CAAC;gBACrD,YAAY,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC;gBAC1B,QAAQ,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;gBACvB,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,SAAS,EAAE,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC;YACnD,CAAC;YACD,SAAS;QACX,CAAC;QACD,IAAI,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,IAAI,CAAC,MAAM,IAAI,kBAAkB,EAAE,CAAC;YAC5D,MAAM,QAAQ,GAAG,IAAI,CAAC,OAAO,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;YACjD,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,IAAI,CAAC,aAAa,CAAC,GAAG,CAAC,QAAQ,CAAC,EAAE,CAAC;gBACxD,aAAa,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;gBAC5B,SAAS,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;gBACzB,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,UAAU,EAAE,KAAK,EAAE,QAAQ,EAAE,CAAC,CAAC;YACrD,CAAC;QACH,CAAC;IACH,CAAC;IAED,OAAO;QACL,QAAQ;QACR,SAAS;QACT,KAAK,EAAE,WAAW,CAAC,MAAM,CAAC;QAC1B,UAAU,EAAE,0BAA0B,CAAC,MAAM,CAAC;KAC/C,CAAC;AACJ,CAAC;AAED;;;;;;GAMG;AACH,SAAS,WAAW,CAAC,MAA6B;IAChD,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,IAAI,IAAI,MAAM,EAAE,CAAC;QAC1B,KAAK,MAAM,KAAK,IAAI,aAAa,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC;YAC9C,IAAI,OAAO,CAAC,KAAK,CAAC,IAAI,KAAK,CAAC,MAAM,GAAG,cAAc,EAAE,CAAC;gBACpD,SAAS;YACX,CAAC;YACD,IAAI,KAAK,CAAC,MAAM,GAAG,cAAc,EAAE,CAAC;gBAClC,SAAS;YACX,CAAC;YACD,IAAI,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,IAAI,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC;gBAC/C,SAAS;YACX,CAAC;YACD,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;YAChB,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;YAClB,IAAI,KAAK,CAAC,MAAM,KAAK,SAAS,EAAE,CAAC;gBAC/B,OAAO,KAAK,CAAC;YACf,CAAC;QACH,CAAC;IACH,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAED;;;;;;GAMG;AACH,SAAS,0BAA0B,CAAC,MAA6B;IAC/D,MAAM,UAAU,GAAa,EAAE,CAAC;IAChC,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,IAAI,IAAI,MAAM,EAAE,CAAC;QAC1B,IAAI,IAAI,CAAC,IAAI,KAAK,SAAS,IAAI,IAAI,CAAC,KAAK,CAAC,MAAM,GAAG,0BAA0B,EAAE,CAAC;YAC9E,SAAS;QACX,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC;QAChD,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YAC1B,SAAS;QACX,CAAC;QACD,IAAI,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,EAAE,CAAC;YACvB,SAAS;QACX,CAAC;QACD,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QACnB,UAAU,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;QAC1B,IAAI,UAAU,CAAC,MAAM,KAAK,cAAc,EAAE,CAAC;YACzC,MAAM;QACR,CAAC;IACH,CAAC;IACD,OAAO,UAAU,CAAC;AACpB,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAAC,IAAY;IAC3C,OAAO,gBAAgB,CAAC,IAAI,CAAC,CAAC,UAAU,CAAC;AAC3C,CAAC;AAED,gEAAgE;AAChE,MAAM,UAAU,IAAI,CAAC,KAAa;IAChC,OAAO,KAAK;SACT,WAAW,EAAE;SACb,OAAO,CAAC,aAAa,EAAE,GAAG,CAAC;SAC3B,OAAO,CAAC,UAAU,EAAE,EAAE,CAAC,CAAC;AAC7B,CAAC;AAED,SAAS,WAAW,CAAC,KAAwB;IAC3C,OAAO,YAAY,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,GAAG,CAAC;AACzC,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,eAAe,CAAC,KAAa,EAAE,KAAwB;IACrE,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACvB,OAAO,KAAK,CAAC;IACf,CAAC;IACD,MAAM,QAAQ,GAAG,CAAC,GAAG,KAAK,CAAC,CAAC;IAC5B,IAAI,OAAO,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;IACpC,OAAO,OAAO,CAAC,MAAM,GAAG,qBAAqB,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACrE,QAAQ,CAAC,GAAG,EAAE,CAAC;QACf,OAAO,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;IAClC,CAAC;IACD,OAAO,KAAK,GAAG,OAAO,CAAC;AACzB,CAAC;AA0BD,MAAM,CAAC,KAAK,UAAU,iBAAiB,CACrC,IAAuB,EACvB,EAAmB;IAEnB,IAAI,MAAM,IAAI,IAAI,EAAE,CAAC;QACnB,MAAM,MAAM,GAAG,MAAM,eAAe,CAAC,IAAI,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC;QACpD,cAAc,CAAC,MAAM,EAAE,QAAQ,IAAI,CAAC,IAAI,EAAE,CAAC,CAAC;QAC5C,OAAO,SAAS,CAAC,MAAM,EAAE,EAAE,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,CAAC,CAAC;IAChE,CAAC;IACD,qEAAqE;IACrE,kEAAkE;IAClE,8CAA8C;IAC9C,wEAAwE;IACxE,uEAAuE;IACvE,8BAA8B;IAC9B,MAAM,MAAM,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,EAAE,CAAC,SAAS,CAAC,iBAAiB,CAAC,EAAE,MAAM,CAAC,CAAC;IAC1E,cAAc,CAAC,MAAM,EAAE,gBAAgB,CAAC,CAAC;IACzC,OAAO,SAAS,CAAC,MAAM,EAAE,EAAE,MAAM,EAAE,OAAO,EAAE,CAAC,CAAC;AAChD,CAAC;AAED,KAAK,UAAU,eAAe,CAAC,QAAgB,EAAE,EAAmB;IAClE,IAAI,CAAC;QACH,OAAO,MAAM,EAAE,CAAC,QAAQ,CAAC,QAAQ,CAAC,CAAC;IACrC,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,MAAM,WAAW,CAAC,QAAQ,EAAE,KAAK,CAAC,CAAC;IACrC,CAAC;AACH,CAAC;AAED,yEAAyE;AACzE,SAAS,WAAW,CAAC,QAAgB,EAAE,KAAc;IACnD,MAAM,IAAI,GAAI,KAAsC,EAAE,IAAI,CAAC;IAC3D,IAAI,IAAI,KAAK,QAAQ,EAAE,CAAC;QACtB,OAAO,IAAI,SAAS,CAAC,2BAA2B,QAAQ,EAAE,EAAE,0BAA0B,CAAC,CAAC;IAC1F,CAAC;IACD,IAAI,IAAI,KAAK,QAAQ,EAAE,CAAC;QACtB,OAAO,IAAI,SAAS,CAClB,2CAA2C,QAAQ,EAAE,EACrD,kDAAkD,CACnD,CAAC;IACJ,CAAC;IACD,IAAI,IAAI,KAAK,QAAQ,EAAE,CAAC;QACtB,OAAO,IAAI,SAAS,CAAC,gCAAgC,QAAQ,EAAE,EAAE,+BAA+B,CAAC,CAAC;IACpG,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAED,SAAS,cAAc,CAAC,MAAc,EAAE,KAAa;IACnD,IAAI,MAAM,CAAC,MAAM,GAAG,iBAAiB,EAAE,CAAC;QACtC,MAAM,IAAI,eAAe,CACvB,8BAA8B,iBAAiB,gBAAgB,MAAM,CAAC,MAAM,SAAS,EACrF,oDAAoD,CACrD,CAAC;IACJ,CAAC;IACD,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,EAAE,kBAAkB,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,EAAE,CAAC;QACvD,MAAM,IAAI,SAAS,CACjB,mBAAmB,KAAK,wBAAwB,EAChD,2BAA2B,CAC5B,CAAC;IACJ,CAAC;AACH,CAAC;AAED,SAAS,SAAS,CAChB,MAAc,EACd,IAAiD;IAEjD,MAAM,IAAI,GAAG,MAAM,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC;IACrC,MAAM,MAAM,GAAG,UAAU,CAAC,QAAQ,CAAC,CAAC,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC;IACvE,OAAO,IAAI,CAAC,IAAI,KAAK,SAAS;QAC5B,CAAC,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,IAAI,CAAC,MAAM,EAAE;QACvC,CAAC,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,CAAC;AAC7D,CAAC"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Central `--flag=value` normalization (#263).
|
|
3
|
+
*
|
|
4
|
+
* `parseArgs` (index.ts) has no equals-form handling: a `--flag=value`
|
|
5
|
+
* token landed as a boolean flag under a garbage key (`flags["flag=value"]`)
|
|
6
|
+
* and was silently dropped — the accept-and-drop class, but at the parser
|
|
7
|
+
* layer. Two commands (archive, investigate) guarded locally with
|
|
8
|
+
* per-command rejections; this seam replaces both: `--flag=value` is
|
|
9
|
+
* EQUIVALENT to `--flag value` on every command.
|
|
10
|
+
*
|
|
11
|
+
* `main()` runs this over the full argv BEFORE extractGlobalOptions, the
|
|
12
|
+
* strict-flag gate, and every handler's parse, so all downstream token
|
|
13
|
+
* walks (parseArgs, findUnknownStrictFlag, isCommandHelpInvocation,
|
|
14
|
+
* collectLongFlagValues, isDryRunBatchInvocation) observe space-form
|
|
15
|
+
* tokens only. The exported raw-argv parsers (archive's
|
|
16
|
+
* parseArchiveArgs, watch's parseWatchArgs, science's parseScienceArgs)
|
|
17
|
+
* apply it directly because they are callable with raw argv (the tests
|
|
18
|
+
* do); their handlers consume already-normalized tokens via the
|
|
19
|
+
* parse<Tokens> inners, keeping the seam single-application.
|
|
20
|
+
*
|
|
21
|
+
* Semantics (deliberate rulings):
|
|
22
|
+
* - Splits on the FIRST '=' only, and only in `--`-prefixed tokens
|
|
23
|
+
* with a non-empty key: `--header=K:V=W` → `--header` `K:V=W` (a
|
|
24
|
+
* value's internal '='s are never split); `--=x` and bare `--` pass
|
|
25
|
+
* through untouched (a `=` at index 2 means an empty key).
|
|
26
|
+
* - `--flag=` → `--flag` `` — parseArgs treats the empty follower as
|
|
27
|
+
* no value, so the equals form with an empty value is the VALUELESS
|
|
28
|
+
* flag, exactly like the space form `--flag ""`.
|
|
29
|
+
* - `--no-cache=1` → `--no-cache` `1`: the no-branch consumes nothing,
|
|
30
|
+
* so `1` becomes a positional — identical to the space form.
|
|
31
|
+
* - Short flags (`-O=json`) and non-flag tokens (`a=b`) are untouched:
|
|
32
|
+
* the contract is the long-flag equals form ≡ space form, nothing
|
|
33
|
+
* more. Under SCOUTLINE_STRICT_FLAGS a `-O=json` token still rejects
|
|
34
|
+
* as unknown, same as before.
|
|
35
|
+
*/
|
|
36
|
+
export declare function normalizeEqualsFormFlags(args: readonly string[]): string[];
|
|
37
|
+
//# sourceMappingURL=equals-form.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"equals-form.d.ts","sourceRoot":"","sources":["../../src/lib/equals-form.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,wBAAgB,wBAAwB,CAAC,IAAI,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,EAAE,CAe1E"}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Central `--flag=value` normalization (#263).
|
|
3
|
+
*
|
|
4
|
+
* `parseArgs` (index.ts) has no equals-form handling: a `--flag=value`
|
|
5
|
+
* token landed as a boolean flag under a garbage key (`flags["flag=value"]`)
|
|
6
|
+
* and was silently dropped — the accept-and-drop class, but at the parser
|
|
7
|
+
* layer. Two commands (archive, investigate) guarded locally with
|
|
8
|
+
* per-command rejections; this seam replaces both: `--flag=value` is
|
|
9
|
+
* EQUIVALENT to `--flag value` on every command.
|
|
10
|
+
*
|
|
11
|
+
* `main()` runs this over the full argv BEFORE extractGlobalOptions, the
|
|
12
|
+
* strict-flag gate, and every handler's parse, so all downstream token
|
|
13
|
+
* walks (parseArgs, findUnknownStrictFlag, isCommandHelpInvocation,
|
|
14
|
+
* collectLongFlagValues, isDryRunBatchInvocation) observe space-form
|
|
15
|
+
* tokens only. The exported raw-argv parsers (archive's
|
|
16
|
+
* parseArchiveArgs, watch's parseWatchArgs, science's parseScienceArgs)
|
|
17
|
+
* apply it directly because they are callable with raw argv (the tests
|
|
18
|
+
* do); their handlers consume already-normalized tokens via the
|
|
19
|
+
* parse<Tokens> inners, keeping the seam single-application.
|
|
20
|
+
*
|
|
21
|
+
* Semantics (deliberate rulings):
|
|
22
|
+
* - Splits on the FIRST '=' only, and only in `--`-prefixed tokens
|
|
23
|
+
* with a non-empty key: `--header=K:V=W` → `--header` `K:V=W` (a
|
|
24
|
+
* value's internal '='s are never split); `--=x` and bare `--` pass
|
|
25
|
+
* through untouched (a `=` at index 2 means an empty key).
|
|
26
|
+
* - `--flag=` → `--flag` `` — parseArgs treats the empty follower as
|
|
27
|
+
* no value, so the equals form with an empty value is the VALUELESS
|
|
28
|
+
* flag, exactly like the space form `--flag ""`.
|
|
29
|
+
* - `--no-cache=1` → `--no-cache` `1`: the no-branch consumes nothing,
|
|
30
|
+
* so `1` becomes a positional — identical to the space form.
|
|
31
|
+
* - Short flags (`-O=json`) and non-flag tokens (`a=b`) are untouched:
|
|
32
|
+
* the contract is the long-flag equals form ≡ space form, nothing
|
|
33
|
+
* more. Under SCOUTLINE_STRICT_FLAGS a `-O=json` token still rejects
|
|
34
|
+
* as unknown, same as before.
|
|
35
|
+
*/
|
|
36
|
+
export function normalizeEqualsFormFlags(args) {
|
|
37
|
+
const out = [];
|
|
38
|
+
for (const arg of args) {
|
|
39
|
+
if (arg.startsWith("--")) {
|
|
40
|
+
const eq = arg.indexOf("=");
|
|
41
|
+
// eq > 2 ⇒ non-empty key between "--" and "=" (also excludes the
|
|
42
|
+
// -1 no-equals case and the `--=` empty-key case).
|
|
43
|
+
if (eq > 2) {
|
|
44
|
+
out.push(arg.slice(0, eq), arg.slice(eq + 1));
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
out.push(arg);
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
//# sourceMappingURL=equals-form.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"equals-form.js","sourceRoot":"","sources":["../../src/lib/equals-form.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,MAAM,UAAU,wBAAwB,CAAC,IAAuB;IAC9D,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACvB,IAAI,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,EAAE,CAAC;YACzB,MAAM,EAAE,GAAG,GAAG,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;YAC5B,iEAAiE;YACjE,mDAAmD;YACnD,IAAI,EAAE,GAAG,CAAC,EAAE,CAAC;gBACX,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,EAAE,GAAG,CAAC,KAAK,CAAC,EAAE,GAAG,CAAC,CAAC,CAAC,CAAC;gBAC9C,SAAS;YACX,CAAC;QACH,CAAC;QACD,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;IAChB,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC"}
|
|
@@ -3,9 +3,10 @@
|
|
|
3
3
|
* `investigate --verify` (investigate-verify lane, Ticket T1; DESIGN
|
|
4
4
|
* D2, PRD AC-2/AC-4/AC-5).
|
|
5
5
|
*
|
|
6
|
-
* splitClaims: the extract grammar's terminator rules
|
|
7
|
-
* (lib/investigate-extract.ts
|
|
8
|
-
* space (or end of input),
|
|
6
|
+
* splitClaims: the extract grammar's terminator rules — the SHARED
|
|
7
|
+
* splitWindows seam (lib/investigate-extract.ts, issue #276): `[.!?]`
|
|
8
|
+
* followed by a space (or end of input), the unconditional
|
|
9
|
+
* ideographic set 。!?, or a newline; the terminator char belongs
|
|
9
10
|
* to the window, the separator run (spaces/tabs) to neither. The
|
|
10
11
|
* splitter is PURE: the ≤ 8 claim cap is the CALLER's fail-loud
|
|
11
12
|
* validation (commands/investigate.ts, T3) — a 9-sentence statement
|
|
@@ -78,6 +79,13 @@ export interface MatchClaimsArgs {
|
|
|
78
79
|
readonly claims: readonly string[];
|
|
79
80
|
readonly sources: readonly EvidenceSource[];
|
|
80
81
|
}
|
|
82
|
+
/**
|
|
83
|
+
* Claim terms: lowercase tokens via the shared Unicode tokenizer
|
|
84
|
+
* (issue #271, apostrophe grammar — ASCII behavior byte-identical),
|
|
85
|
+
* stopword-filtered, normalizeTerms-dedupe.
|
|
86
|
+
*/
|
|
87
|
+
/** Exported for the #271 term-derivation pin (pure; test-observable). */
|
|
88
|
+
export declare function claimTerms(claim: string): string[];
|
|
81
89
|
/**
|
|
82
90
|
* Match claims to evidence: per claim, every (source, passage) whose
|
|
83
91
|
* quote contains ≥ 1 claim term whole-word case-folded becomes an
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"investigate-claims.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-claims.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"investigate-claims.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-claims.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAIH,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,kCAAkC,CAAC;AAQvE;;;;;GAKG;AACH,eAAO,MAAM,iBAAiB,IAAI,CAAC;AAMnC;;;;;GAKG;AACH,wBAAgB,WAAW,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,EAAE,CASvD;AAMD;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,aAAa,EAAE,SAAS,MAAM,EAgCzC,CAAC;AAQH,uDAAuD;AACvD,MAAM,MAAM,YAAY,GAAG,cAAc,GAAG,cAAc,GAAG,YAAY,CAAC;AAE1E,8DAA8D;AAC9D,MAAM,WAAW,oBAAoB;IACnC,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;CAC/B;AAED,kEAAkE;AAClE,MAAM,WAAW,QAAQ;IACvB,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;IACtB,QAAQ,CAAC,OAAO,EAAE,YAAY,CAAC;IAC/B,yEAAyE;IACzE,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,QAAQ,EAAE,SAAS,oBAAoB,EAAE,CAAC;CACpD;AAED,wEAAwE;AACxE,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,MAAM,EAAE,SAAS,QAAQ,EAAE,CAAC;CACtC;AAED,MAAM,WAAW,eAAe;IAC9B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,CAAC;IACnC,QAAQ,CAAC,OAAO,EAAE,SAAS,cAAc,EAAE,CAAC;CAC7C;AAwBD;;;;GAIG;AACH,yEAAyE;AACzE,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,EAAE,CAGlD;AAED;;;;;;;GAOG;AACH,wBAAgB,qBAAqB,CAAC,EAAE,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,EAAE,eAAe,GAAG,WAAW,CAqBlG"}
|
|
@@ -3,9 +3,10 @@
|
|
|
3
3
|
* `investigate --verify` (investigate-verify lane, Ticket T1; DESIGN
|
|
4
4
|
* D2, PRD AC-2/AC-4/AC-5).
|
|
5
5
|
*
|
|
6
|
-
* splitClaims: the extract grammar's terminator rules
|
|
7
|
-
* (lib/investigate-extract.ts
|
|
8
|
-
* space (or end of input),
|
|
6
|
+
* splitClaims: the extract grammar's terminator rules — the SHARED
|
|
7
|
+
* splitWindows seam (lib/investigate-extract.ts, issue #276): `[.!?]`
|
|
8
|
+
* followed by a space (or end of input), the unconditional
|
|
9
|
+
* ideographic set 。!?, or a newline; the terminator char belongs
|
|
9
10
|
* to the window, the separator run (spaces/tabs) to neither. The
|
|
10
11
|
* splitter is PURE: the ≤ 8 claim cap is the CALLER's fail-loud
|
|
11
12
|
* validation (commands/investigate.ts, T3) — a 9-sentence statement
|
|
@@ -23,8 +24,8 @@
|
|
|
23
24
|
* Byte-stable: no clocks, no randomness, no locale-dependent casing
|
|
24
25
|
* beyond toLowerCase.
|
|
25
26
|
*/
|
|
26
|
-
import { STOPWORDS } from "./context-file.js";
|
|
27
|
-
import { normalizeTerms } from "./investigate-extract.js";
|
|
27
|
+
import { STOPWORDS, tokenizeTerms } from "./context-file.js";
|
|
28
|
+
import { normalizeTerms, splitWindows } from "./investigate-extract.js";
|
|
28
29
|
// ---------------------------------------------------------------------------
|
|
29
30
|
// Claim splitting (extract-grammar terminators)
|
|
30
31
|
// ---------------------------------------------------------------------------
|
|
@@ -36,44 +37,9 @@ const STOPWORD_SET = new Set(STOPWORDS);
|
|
|
36
37
|
* from the single source of truth.
|
|
37
38
|
*/
|
|
38
39
|
export const MAX_VERIFY_CLAIMS = 8;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
* Pinned to lib/investigate-extract.ts splitWindows — a decimal like
|
|
43
|
-
* "3.14" has no space after the dot, so it never terminates.
|
|
44
|
-
*/
|
|
45
|
-
function splitWindows(content) {
|
|
46
|
-
const windows = [];
|
|
47
|
-
let start = 0;
|
|
48
|
-
for (let i = 0; i < content.length; i += 1) {
|
|
49
|
-
const ch = content.charAt(i);
|
|
50
|
-
const next = content.charAt(i + 1);
|
|
51
|
-
const isTerminator = (ch === "." || ch === "!" || ch === "?") &&
|
|
52
|
-
(i + 1 >= content.length || next === " ");
|
|
53
|
-
const isNewline = ch === "\n";
|
|
54
|
-
if (!isTerminator && !isNewline) {
|
|
55
|
-
continue;
|
|
56
|
-
}
|
|
57
|
-
const end = i + 1; // terminator char owned by the window
|
|
58
|
-
windows.push({ start, end });
|
|
59
|
-
// Skip the horizontal-whitespace separator run; the next window
|
|
60
|
-
// starts at the first character that is neither space nor tab.
|
|
61
|
-
let cursor = end;
|
|
62
|
-
while (cursor < content.length) {
|
|
63
|
-
const sep = content.charAt(cursor);
|
|
64
|
-
if (sep !== " " && sep !== "\t") {
|
|
65
|
-
break;
|
|
66
|
-
}
|
|
67
|
-
cursor += 1;
|
|
68
|
-
}
|
|
69
|
-
start = cursor;
|
|
70
|
-
i = cursor - 1; // loop increment lands on `cursor`
|
|
71
|
-
}
|
|
72
|
-
if (start < content.length) {
|
|
73
|
-
windows.push({ start, end: content.length });
|
|
74
|
-
}
|
|
75
|
-
return windows;
|
|
76
|
-
}
|
|
40
|
+
// splitWindows and its Window shape live in lib/investigate-extract.ts
|
|
41
|
+
// (issue #276: the terminator grammar is ONE shared seam — the ideographic
|
|
42
|
+
// set extends both extract and claims from the single definition).
|
|
77
43
|
/**
|
|
78
44
|
* Split a statement into sentence-claims: extract-grammar windows,
|
|
79
45
|
* whitespace-only sentences dropped, quotes trimmed. Deterministic —
|
|
@@ -144,16 +110,28 @@ const CUE_LITERALS = NEGATION_CUES;
|
|
|
144
110
|
function escapeRegExp(s) {
|
|
145
111
|
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
146
112
|
}
|
|
113
|
+
/**
|
|
114
|
+
* Whole-word containment. ASCII terms keep the boundary guard
|
|
115
|
+
* (`[^\p{L}\p{N}_]` on both sides). Non-ASCII terms match as
|
|
116
|
+
* substrings: unspaced scripts (CJK) have no word separators, so the
|
|
117
|
+
* boundary class can never fire between letters (issue #271).
|
|
118
|
+
*/
|
|
147
119
|
function containsWholeWord(content, term) {
|
|
148
|
-
const
|
|
120
|
+
const asciiTerm = /^[\x00-\x7f]*$/.test(term);
|
|
121
|
+
const body = escapeRegExp(term);
|
|
122
|
+
const re = new RegExp(asciiTerm
|
|
123
|
+
? `(^|[^\\p{L}\\p{N}_])${body}(?:[^\\p{L}\\p{N}_]|$)`
|
|
124
|
+
: body, "iu");
|
|
149
125
|
return re.test(content);
|
|
150
126
|
}
|
|
151
|
-
/**
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
127
|
+
/**
|
|
128
|
+
* Claim terms: lowercase tokens via the shared Unicode tokenizer
|
|
129
|
+
* (issue #271, apostrophe grammar — ASCII behavior byte-identical),
|
|
130
|
+
* stopword-filtered, normalizeTerms-dedupe.
|
|
131
|
+
*/
|
|
132
|
+
/** Exported for the #271 term-derivation pin (pure; test-observable). */
|
|
133
|
+
export function claimTerms(claim) {
|
|
134
|
+
const tokens = tokenizeTerms(claim, true).filter((t) => !STOPWORD_SET.has(t));
|
|
157
135
|
return normalizeTerms(tokens);
|
|
158
136
|
}
|
|
159
137
|
/**
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"investigate-claims.js","sourceRoot":"","sources":["../../src/lib/investigate-claims.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"investigate-claims.js","sourceRoot":"","sources":["../../src/lib/investigate-claims.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,EAAE,SAAS,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAC7D,OAAO,EAAE,cAAc,EAAE,YAAY,EAAE,MAAM,0BAA0B,CAAC;AAGxE,8EAA8E;AAC9E,gDAAgD;AAChD,8EAA8E;AAE9E,MAAM,YAAY,GAAwB,IAAI,GAAG,CAAC,SAAS,CAAC,CAAC;AAE7D;;;;;GAKG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC;AAEnC,uEAAuE;AACvE,2EAA2E;AAC3E,mEAAmE;AAEnE;;;;;GAKG;AACH,MAAM,UAAU,WAAW,CAAC,SAAiB;IAC3C,MAAM,MAAM,GAAa,EAAE,CAAC;IAC5B,KAAK,MAAM,MAAM,IAAI,YAAY,CAAC,SAAS,CAAC,EAAE,CAAC;QAC7C,MAAM,KAAK,GAAG,SAAS,CAAC,KAAK,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;QAC/D,IAAI,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACrB,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;QACrB,CAAC;IACH,CAAC;IACD,OAAO,MAAM,CAAC;AAChB,CAAC;AAED,8EAA8E;AAC9E,wDAAwD;AACxD,8EAA8E;AAE9E;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,MAAM,aAAa,GAAsB,MAAM,CAAC,MAAM,CAAC;IAC5D,KAAK;IACL,IAAI;IACJ,OAAO;IACP,MAAM;IACN,QAAQ;IACR,OAAO;IACP,OAAO;IACP,QAAQ;IACR,QAAQ;IACR,SAAS;IACT,SAAS;IACT,OAAO;IACP,QAAQ;IACR,OAAO;IACP,UAAU;IACV,SAAS;IACT,UAAU;IACV,WAAW;IACX,QAAQ;IACR,QAAQ;IACR,SAAS;IACT,SAAS;IACT,UAAU;IACV,UAAU;IACV,SAAS;IACT,OAAO;IACP,QAAQ;IACR,OAAO;IACP,QAAQ;IACR,QAAQ;IACR,QAAQ;CACT,CAAC,CAAC;AAEH,MAAM,YAAY,GAAsB,aAAa,CAAC;AAoCtD,SAAS,YAAY,CAAC,CAAS;IAC7B,OAAO,CAAC,CAAC,OAAO,CAAC,qBAAqB,EAAE,MAAM,CAAC,CAAC;AAClD,CAAC;AAED;;;;;GAKG;AACH,SAAS,iBAAiB,CAAC,OAAe,EAAE,IAAY;IACtD,MAAM,SAAS,GAAG,gBAAgB,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IAC9C,MAAM,IAAI,GAAG,YAAY,CAAC,IAAI,CAAC,CAAC;IAChC,MAAM,EAAE,GAAG,IAAI,MAAM,CACnB,SAAS;QACP,CAAC,CAAC,uBAAuB,IAAI,wBAAwB;QACrD,CAAC,CAAC,IAAI,EACR,IAAI,CACL,CAAC;IACF,OAAO,EAAE,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;AAC1B,CAAC;AAED;;;;GAIG;AACH,yEAAyE;AACzE,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,MAAM,MAAM,GAAG,aAAa,CAAC,KAAK,EAAE,IAAI,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,YAAY,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;IAC9E,OAAO,cAAc,CAAC,MAAM,CAAC,CAAC;AAChC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,qBAAqB,CAAC,EAAE,SAAS,EAAE,MAAM,EAAE,OAAO,EAAmB;IACnF,MAAM,IAAI,GAAe,MAAM,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE;QAC3C,MAAM,KAAK,GAAG,UAAU,CAAC,IAAI,CAAC,CAAC;QAC/B,MAAM,QAAQ,GAA2B,EAAE,CAAC;QAC5C,IAAI,UAAU,GAAG,CAAC,CAAC;QACnB,OAAO,CAAC,OAAO,CAAC,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE;YACtC,MAAM,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,OAAO,EAAE,YAAY,EAAE,EAAE;gBAChD,MAAM,OAAO,GAAG,KAAK,CAAC,IAAI,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,iBAAiB,CAAC,OAAO,CAAC,KAAK,EAAE,IAAI,CAAC,CAAC,CAAC;gBAC7E,IAAI,CAAC,OAAO;oBAAE,OAAO;gBACrB,QAAQ,CAAC,IAAI,CAAC,EAAE,WAAW,EAAE,YAAY,EAAE,CAAC,CAAC;gBAC7C,MAAM,UAAU,GAAG,YAAY,CAAC,IAAI,CAAC,CAAC,GAAG,EAAE,EAAE,CAC3C,iBAAiB,CAAC,OAAO,CAAC,KAAK,CAAC,WAAW,EAAE,EAAE,GAAG,CAAC,CACpD,CAAC;gBACF,IAAI,UAAU;oBAAE,UAAU,IAAI,CAAC,CAAC;YAClC,CAAC,CAAC,CAAC;QACL,CAAC,CAAC,CAAC;QACH,MAAM,OAAO,GACX,QAAQ,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,cAAc,CAAC,CAAC,CAAC,cAAc,CAAC;QAC1F,OAAO,EAAE,IAAI,EAAE,OAAO,EAAE,YAAY,EAAE,UAAU,EAAE,QAAQ,EAAE,CAAC;IAC/D,CAAC,CAAC,CAAC;IACH,OAAO,EAAE,SAAS,EAAE,MAAM,EAAE,IAAI,EAAE,CAAC;AACrC,CAAC"}
|
|
@@ -26,6 +26,29 @@ export interface ExtractPassagesArgs {
|
|
|
26
26
|
}
|
|
27
27
|
/** Normalize the term set: trim, case-fold, drop empties, dedupe. */
|
|
28
28
|
export declare function normalizeTerms(terms: string[]): string[];
|
|
29
|
+
/**
|
|
30
|
+
* Ideographic sentence terminators (issue #276): the ideographic full
|
|
31
|
+
* stop 。 (U+3002) and the fullwidth !?. They terminate UNCONDITIONALLY
|
|
32
|
+
* — no space lookahead — because unspaced scripts never satisfy the
|
|
33
|
+
* ASCII rule's lookahead. Clause commas (,、) are deliberately NOT
|
|
34
|
+
* terminators: they separate clauses inside a sentence, not sentences.
|
|
35
|
+
* Exported: investigate-claims.ts's splitClaims shares this grammar.
|
|
36
|
+
*/
|
|
37
|
+
export declare const IDEOGRAPHIC_TERMINATORS: Set<string>;
|
|
38
|
+
/** A window is [start, end) with the terminator char owned by the window. */
|
|
39
|
+
export interface Window {
|
|
40
|
+
start: number;
|
|
41
|
+
end: number;
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Split content into sentence windows on `[.!?]` terminators followed
|
|
45
|
+
* by a space (the literal `[.!?] +` grammar), the unconditional
|
|
46
|
+
* ideographic set (issue #276, see {@link IDEOGRAPHIC_TERMINATORS}),
|
|
47
|
+
* plus newline boundaries, tracking exact offsets into the original
|
|
48
|
+
* content. Newlines terminate but never separate: "here.\n" ends at
|
|
49
|
+
* the `\n`.
|
|
50
|
+
*/
|
|
51
|
+
export declare function splitWindows(content: string): Window[];
|
|
29
52
|
/**
|
|
30
53
|
* Extract passages: sentence windows of `content` containing >= 1 of
|
|
31
54
|
* `terms` as a whole word (case-folded), deduped by exact quote,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"investigate-extract.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,MAAM,WAAW,OAAO;IACtB,KAAK,EAAE,MAAM,CAAC;IACd,SAAS,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAC7B;AAED,MAAM,WAAW,mBAAmB;IAClC,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,EAAE,CAAC;CACjB;AAID,qEAAqE;AACrE,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,MAAM,EAAE,CASxD;
|
|
1
|
+
{"version":3,"file":"investigate-extract.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAEH,MAAM,WAAW,OAAO;IACtB,KAAK,EAAE,MAAM,CAAC;IACd,SAAS,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAC7B;AAED,MAAM,WAAW,mBAAmB;IAClC,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,EAAE,CAAC;CACjB;AAID,qEAAqE;AACrE,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,MAAM,EAAE,CASxD;AAED;;;;;;;GAOG;AACH,eAAO,MAAM,uBAAuB,aAA2B,CAAC;AAEhE,6EAA6E;AAC7E,MAAM,WAAW,MAAM;IACrB,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;CACb;AAED;;;;;;;GAOG;AACH,wBAAgB,YAAY,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,EAAE,CAoCtD;AAwBD;;;;GAIG;AACH,wBAAgB,eAAe,CAAC,EAAE,OAAO,EAAE,KAAK,EAAE,EAAE,mBAAmB,GAAG,OAAO,EAAE,CAqBlF"}
|
|
@@ -28,20 +28,35 @@ export function normalizeTerms(terms) {
|
|
|
28
28
|
}
|
|
29
29
|
return [...seen];
|
|
30
30
|
}
|
|
31
|
+
/**
|
|
32
|
+
* Ideographic sentence terminators (issue #276): the ideographic full
|
|
33
|
+
* stop 。 (U+3002) and the fullwidth !?. They terminate UNCONDITIONALLY
|
|
34
|
+
* — no space lookahead — because unspaced scripts never satisfy the
|
|
35
|
+
* ASCII rule's lookahead. Clause commas (,、) are deliberately NOT
|
|
36
|
+
* terminators: they separate clauses inside a sentence, not sentences.
|
|
37
|
+
* Exported: investigate-claims.ts's splitClaims shares this grammar.
|
|
38
|
+
*/
|
|
39
|
+
export const IDEOGRAPHIC_TERMINATORS = new Set(["。", "!", "?"]);
|
|
31
40
|
/**
|
|
32
41
|
* Split content into sentence windows on `[.!?]` terminators followed
|
|
33
|
-
* by a space (the literal `[.!?] +` grammar),
|
|
34
|
-
*
|
|
35
|
-
*
|
|
42
|
+
* by a space (the literal `[.!?] +` grammar), the unconditional
|
|
43
|
+
* ideographic set (issue #276, see {@link IDEOGRAPHIC_TERMINATORS}),
|
|
44
|
+
* plus newline boundaries, tracking exact offsets into the original
|
|
45
|
+
* content. Newlines terminate but never separate: "here.\n" ends at
|
|
46
|
+
* the `\n`.
|
|
36
47
|
*/
|
|
37
|
-
function splitWindows(content) {
|
|
48
|
+
export function splitWindows(content) {
|
|
38
49
|
const windows = [];
|
|
39
50
|
let start = 0;
|
|
40
51
|
for (let i = 0; i < content.length; i += 1) {
|
|
41
52
|
const ch = content.charAt(i);
|
|
42
53
|
const next = content.charAt(i + 1);
|
|
43
|
-
|
|
44
|
-
|
|
54
|
+
// ASCII rule unchanged: `.!?` need a following space (or end) so
|
|
55
|
+
// "3.14" never splits. Ideographic terminators (issue #276) fire
|
|
56
|
+
// unconditionally — unspaced scripts never satisfy a lookahead.
|
|
57
|
+
const isTerminator = ((ch === "." || ch === "!" || ch === "?") &&
|
|
58
|
+
(i + 1 >= content.length || next === " ")) ||
|
|
59
|
+
IDEOGRAPHIC_TERMINATORS.has(ch);
|
|
45
60
|
const isNewline = ch === "\n";
|
|
46
61
|
if (!isTerminator && !isNewline) {
|
|
47
62
|
continue;
|
|
@@ -66,8 +81,18 @@ function splitWindows(content) {
|
|
|
66
81
|
}
|
|
67
82
|
return windows;
|
|
68
83
|
}
|
|
84
|
+
/**
|
|
85
|
+
* Whole-word containment. ASCII terms keep the boundary guard;
|
|
86
|
+
* non-ASCII terms match as substrings — unspaced scripts (CJK) have
|
|
87
|
+
* no word separators, so the boundary class can never fire between
|
|
88
|
+
* letters (issue #271).
|
|
89
|
+
*/
|
|
69
90
|
function containsWholeWord(content, term) {
|
|
70
|
-
const
|
|
91
|
+
const asciiTerm = /^[\x00-\x7f]*$/.test(term);
|
|
92
|
+
const body = escapeRegExp(term);
|
|
93
|
+
const re = new RegExp(asciiTerm
|
|
94
|
+
? `(^|[^\\p{L}\\p{N}_])${body}(?:[^\\p{L}\\p{N}_]|$)`
|
|
95
|
+
: body, "iu");
|
|
71
96
|
return re.test(content);
|
|
72
97
|
}
|
|
73
98
|
function escapeRegExp(s) {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"investigate-extract.js","sourceRoot":"","sources":["../../src/lib/investigate-extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAYH,MAAM,YAAY,GAAG,CAAC,CAAC;AAEvB,qEAAqE;AACrE,MAAM,UAAU,cAAc,CAAC,KAAe;IAC5C,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,MAAM,UAAU,GAAG,IAAI,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC;QAC7C,IAAI,UAAU,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC1B,IAAI,CAAC,GAAG,CAAC,UAAU,CAAC,CAAC;QACvB,CAAC;IACH,CAAC;IACD,OAAO,CAAC,GAAG,IAAI,CAAC,CAAC;AACnB,CAAC;
|
|
1
|
+
{"version":3,"file":"investigate-extract.js","sourceRoot":"","sources":["../../src/lib/investigate-extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAYH,MAAM,YAAY,GAAG,CAAC,CAAC;AAEvB,qEAAqE;AACrE,MAAM,UAAU,cAAc,CAAC,KAAe;IAC5C,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,MAAM,UAAU,GAAG,IAAI,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC;QAC7C,IAAI,UAAU,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC1B,IAAI,CAAC,GAAG,CAAC,UAAU,CAAC,CAAC;QACvB,CAAC;IACH,CAAC;IACD,OAAO,CAAC,GAAG,IAAI,CAAC,CAAC;AACnB,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,IAAI,GAAG,CAAC,CAAC,GAAG,EAAE,GAAG,EAAE,GAAG,CAAC,CAAC,CAAC;AAQhE;;;;;;;GAOG;AACH,MAAM,UAAU,YAAY,CAAC,OAAe;IAC1C,MAAM,OAAO,GAAa,EAAE,CAAC;IAC7B,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC;QAC3C,MAAM,EAAE,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC;QAC7B,MAAM,IAAI,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;QACnC,iEAAiE;QACjE,iEAAiE;QACjE,gEAAgE;QAChE,MAAM,YAAY,GAChB,CAAC,CAAC,EAAE,KAAK,GAAG,IAAI,EAAE,KAAK,GAAG,IAAI,EAAE,KAAK,GAAG,CAAC;YACvC,CAAC,CAAC,GAAG,CAAC,IAAI,OAAO,CAAC,MAAM,IAAI,IAAI,KAAK,GAAG,CAAC,CAAC;YAC5C,uBAAuB,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;QAClC,MAAM,SAAS,GAAG,EAAE,KAAK,IAAI,CAAC;QAC9B,IAAI,CAAC,YAAY,IAAI,CAAC,SAAS,EAAE,CAAC;YAChC,SAAS;QACX,CAAC;QACD,MAAM,GAAG,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,sCAAsC;QACzD,OAAO,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,GAAG,EAAE,CAAC,CAAC;QAC7B,gEAAgE;QAChE,+DAA+D;QAC/D,IAAI,MAAM,GAAG,GAAG,CAAC;QACjB,OAAO,MAAM,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC;YAC/B,MAAM,GAAG,GAAG,OAAO,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC;YACnC,IAAI,GAAG,KAAK,GAAG,IAAI,GAAG,KAAK,IAAI,EAAE,CAAC;gBAChC,MAAM;YACR,CAAC;YACD,MAAM,IAAI,CAAC,CAAC;QACd,CAAC;QACD,KAAK,GAAG,MAAM,CAAC;QACf,CAAC,GAAG,MAAM,GAAG,CAAC,CAAC,CAAC,mCAAmC;IACrD,CAAC;IACD,IAAI,KAAK,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC;QAC3B,OAAO,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,GAAG,EAAE,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,OAAO,CAAC;AACjB,CAAC;AAED;;;;;GAKG;AACH,SAAS,iBAAiB,CAAC,OAAe,EAAE,IAAY;IACtD,MAAM,SAAS,GAAG,gBAAgB,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IAC9C,MAAM,IAAI,GAAG,YAAY,CAAC,IAAI,CAAC,CAAC;IAChC,MAAM,EAAE,GAAG,IAAI,MAAM,CACnB,SAAS;QACP,CAAC,CAAC,uBAAuB,IAAI,wBAAwB;QACrD,CAAC,CAAC,IAAI,EACR,IAAI,CACL,CAAC;IACF,OAAO,EAAE,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;AAC1B,CAAC;AAED,SAAS,YAAY,CAAC,CAAS;IAC7B,OAAO,CAAC,CAAC,OAAO,CAAC,qBAAqB,EAAE,MAAM,CAAC,CAAC;AAClD,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,eAAe,CAAC,EAAE,OAAO,EAAE,KAAK,EAAuB;IACrE,MAAM,UAAU,GAAG,cAAc,CAAC,KAAK,CAAC,CAAC;IACzC,IAAI,UAAU,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC5B,OAAO,EAAE,CAAC;IACZ,CAAC;IACD,MAAM,UAAU,GAAG,IAAI,GAAG,EAAU,CAAC;IACrC,MAAM,QAAQ,GAAc,EAAE,CAAC;IAC/B,KAAK,MAAM,MAAM,IAAI,YAAY,CAAC,OAAO,CAAC,EAAE,CAAC;QAC3C,MAAM,KAAK,GAAG,OAAO,CAAC,KAAK,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,CAAC,GAAG,CAAC,CAAC;QACtD,IACE,UAAU,CAAC,IAAI,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,iBAAiB,CAAC,KAAK,EAAE,IAAI,CAAC,CAAC;YACzD,CAAC,UAAU,CAAC,GAAG,CAAC,KAAK,CAAC,EACtB,CAAC;YACD,UAAU,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;YACtB,QAAQ,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,SAAS,EAAE,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;YAChE,IAAI,QAAQ,CAAC,MAAM,IAAI,YAAY,EAAE,CAAC;gBACpC,MAAM;YACR,CAAC;QACH,CAAC;IACH,CAAC;IACD,OAAO,QAAQ,CAAC;AAClB,CAAC"}
|
|
@@ -45,9 +45,11 @@ export interface SubQueryPlan {
|
|
|
45
45
|
}
|
|
46
46
|
/**
|
|
47
47
|
* Key-terms join for the template tier: lowercase the question,
|
|
48
|
-
* tokenize
|
|
49
|
-
*
|
|
50
|
-
*
|
|
48
|
+
* tokenize via the shared Unicode tokenizer (issue #271 — ASCII input
|
|
49
|
+
* keeps the legacy split byte-identically; non-ASCII segments are kept
|
|
50
|
+
* whole with no ASCII length floor), drop stopwords and tokens outside
|
|
51
|
+
* the context-file term bounds, preserve order, dedupe, join with
|
|
52
|
+
* spaces. Only `toLowerCase` is applied (byte-stable, no locale).
|
|
51
53
|
*/
|
|
52
54
|
export declare function deriveTemplateTopic(query: string): string;
|
|
53
55
|
/**
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"investigate-planner.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-planner.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;
|
|
1
|
+
{"version":3,"file":"investigate-planner.d.ts","sourceRoot":"","sources":["../../src/lib/investigate-planner.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAcH;;;;;GAKG;AACH,eAAO,MAAM,uBAAuB,IAAI,CAAC;AAMzC,MAAM,MAAM,WAAW,GAAG,UAAU,GAAG,SAAS,GAAG,UAAU,CAAC;AAE9D,MAAM,WAAW,qBAAqB;IACpC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,gEAAgE;IAChE,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CAC/B;AAED,MAAM,WAAW,kBAAkB;IACjC;;;OAGG;IACH,eAAe,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC;CACpD;AAED,MAAM,WAAW,YAAY;IAC3B,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;IACvC,QAAQ,CAAC,IAAI,EAAE,WAAW,CAAC;IAC3B,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;;;GAOG;AACH,wBAAgB,mBAAmB,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAiBzD;AAmBD;;;GAGG;AACH,wBAAsB,cAAc,CAClC,OAAO,EAAE,qBAAqB,EAC9B,IAAI,EAAE,kBAAkB,GACvB,OAAO,CAAC,YAAY,CAAC,CA6BvB"}
|