@mailwoman/kind-classifier 9.4.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/lib/classify.ts +35 -36
- package/lib/index.ts +4 -11
- package/lib/{intent-markers.ts → intent/markers.ts} +29 -26
- package/lib/intent/rules.ts +306 -0
- package/lib/poi.ts +96 -84
- package/lib/rules.ts +142 -105
- package/out/classify.d.ts +18 -20
- package/out/classify.d.ts.map +1 -1
- package/out/classify.js +30 -33
- package/out/classify.js.map +1 -1
- package/out/index.d.ts +4 -11
- package/out/index.d.ts.map +1 -1
- package/out/index.js +3 -10
- package/out/index.js.map +1 -1
- package/out/intent/markers.d.ts +45 -0
- package/out/intent/markers.d.ts.map +1 -0
- package/out/intent/markers.js +88 -0
- package/out/intent/markers.js.map +1 -0
- package/out/intent/rules.d.ts +55 -0
- package/out/intent/rules.d.ts.map +1 -0
- package/out/intent/rules.js +281 -0
- package/out/intent/rules.js.map +1 -0
- package/out/poi.d.ts +64 -56
- package/out/poi.d.ts.map +1 -1
- package/out/poi.js +60 -50
- package/out/poi.js.map +1 -1
- package/out/rules.d.ts +22 -44
- package/out/rules.d.ts.map +1 -1
- package/out/rules.js +118 -104
- package/out/rules.js.map +1 -1
- package/package.json +15 -50
- package/lib/intent-rules.ts +0 -307
- package/out/intent-markers.d.ts +0 -46
- package/out/intent-markers.d.ts.map +0 -1
- package/out/intent-markers.js +0 -87
- package/out/intent-markers.js.map +0 -1
- package/out/intent-rules.d.ts +0 -59
- package/out/intent-rules.d.ts.map +0 -1
- package/out/intent-rules.js +0 -282
- package/out/intent-rules.js.map +0 -1
package/lib/poi.ts
CHANGED
|
@@ -3,16 +3,18 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* POI subject detection for the `poi_query` kind. The lexicon is injected (`POIPhraseLookup`) so this
|
|
7
|
+
* package keeps its bitter-lesson invariant (no dictionaries in-tree); the phrase table lives in
|
|
8
|
+
* `@mailwoman/poi-taxonomy` and is wired in by `createRuntimePipeline` behind the `poiQueryKind` flag.
|
|
9
|
+
* Spec §3.1.
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
12
|
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
|
|
13
13
|
/**
|
|
14
|
-
* Comma-segment ceiling for a POI-led query.
|
|
15
|
-
*
|
|
14
|
+
* Comma-segment ceiling for a POI-led query.
|
|
15
|
+
*
|
|
16
|
+
* Past it the input is a venue plus a full address (`X, 350 5th Ave, New York, NY`),
|
|
17
|
+
* which the structured-address scorer should claim instead.
|
|
16
18
|
*/
|
|
17
19
|
const MAX_POI_SEGMENTS = 3
|
|
18
20
|
|
|
@@ -21,9 +23,9 @@ const MAX_POI_SEGMENTS = 3
|
|
|
21
23
|
*/
|
|
22
24
|
export interface POIPhraseMatch {
|
|
23
25
|
/**
|
|
24
|
-
* The matched subject's identifier
|
|
25
|
-
* `kind: "
|
|
26
|
-
*
|
|
26
|
+
* The matched subject's identifier: a `@mailwoman/poi-taxonomy` category id for
|
|
27
|
+
* `kind: "category"`, otherwise the canonical display name; `matchPOISubject`
|
|
28
|
+
* treats it opaquely and the caller interprets it per `kind`.
|
|
27
29
|
*/
|
|
28
30
|
categoryID: string
|
|
29
31
|
matchedPhrase: string
|
|
@@ -31,43 +33,46 @@ export interface POIPhraseMatch {
|
|
|
31
33
|
mechanism?: "exact" | "locale_normalized" | "typo"
|
|
32
34
|
inputPhrase?: string
|
|
33
35
|
/**
|
|
34
|
-
* Absent
|
|
35
|
-
* source-compatible.
|
|
36
|
+
* Absent means `"category"`; optional so existing `POIPhraseLookup` implementors stay source-compatible.
|
|
36
37
|
*/
|
|
37
38
|
kind?: "category" | "brand" | "name"
|
|
38
39
|
/**
|
|
39
|
-
* Wikidata QID
|
|
40
|
+
* Wikidata QID when known, `kind: "brand"` only.
|
|
41
|
+
* Absent when a brand resolved using its name only.
|
|
40
42
|
*/
|
|
41
43
|
wikidata?: string
|
|
42
44
|
/**
|
|
43
|
-
* Whether this hit is one member of a set the caller must search
|
|
44
|
-
* list.
|
|
45
|
+
* Whether this hit is one member of a set the caller must search together
|
|
46
|
+
* rather than one candidate in a preference list.
|
|
47
|
+
*
|
|
48
|
+
* A lookup returning several hits means two different things: a phrase index
|
|
49
|
+
* returns the categories one typed phrase could name, the curated reading first
|
|
50
|
+
* (`credit union` → the `bank` rollup its synonym redirects to), while an affordance
|
|
51
|
+
* rung returns every entity kind that affords one activity in a stable enumeration.
|
|
52
|
+
* The enumeration does not express a preference.
|
|
45
53
|
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
* `credit_union` category), and the first entry IS the subject. An affordance rung returns every entity kind that
|
|
50
|
-
* affords ONE activity, in a stable enumeration that is not a preference, and taking the first picks a winner nobody
|
|
51
|
-
* authored.
|
|
54
|
+
* The first result would impose an ordering that the source does not provide.
|
|
55
|
+
* Set on every member of such a set, so {@link matchPOISubject} returns them all
|
|
56
|
+
* and the POI branch searches their union.
|
|
52
57
|
*
|
|
53
|
-
*
|
|
54
|
-
* union, and the candidate ordering the resolver already owns decides the answer. Absent — the committed lexicon's
|
|
55
|
-
* shape — keeps the first-hit reading unchanged.
|
|
58
|
+
* Absent, the committed lexicon's shape, keeps the first-hit reading.
|
|
56
59
|
*/
|
|
57
60
|
searchAsSet?: boolean
|
|
58
61
|
/**
|
|
59
|
-
* ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to.
|
|
60
|
-
* everywhere.
|
|
62
|
+
* ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to.
|
|
63
|
+
* Absent means the condition is true everywhere.
|
|
61
64
|
*
|
|
62
|
-
* A scope is a statement about
|
|
63
|
-
* the caller's locale: the locale is the lens the
|
|
64
|
-
*
|
|
65
|
+
* A scope is a statement about establishments, so it is judged against the country of
|
|
66
|
+
* the place being searched rather than the caller's locale: the locale is the lens the
|
|
67
|
+
* phrase is read through and makes no statement about where the condition is true.
|
|
68
|
+
* `matchPOISubject` returns the value unchanged.
|
|
69
|
+
* The POI intent stage binds it once the anchor has resolved.
|
|
65
70
|
*/
|
|
66
71
|
countryScope?: readonly string[]
|
|
67
72
|
}
|
|
68
73
|
|
|
69
74
|
/**
|
|
70
|
-
* Injected phrase→category lookup
|
|
75
|
+
* Injected phrase→category lookup, exact-phrase and locale-aware, returning `[]` on miss.
|
|
71
76
|
*/
|
|
72
77
|
export type POIPhraseLookup = (phrase: string, locale?: string) => ReadonlyArray<POIPhraseMatch>
|
|
73
78
|
|
|
@@ -83,25 +88,24 @@ export interface POIQuerySpan {
|
|
|
83
88
|
}
|
|
84
89
|
|
|
85
90
|
/**
|
|
86
|
-
* Which lexicon this hit came from.
|
|
91
|
+
* Which lexicon this hit came from.
|
|
92
|
+
*
|
|
93
|
+
* Category lookups set `"category"` as the backward-compatible default.
|
|
87
94
|
*/
|
|
88
95
|
export interface POISubjectMatch {
|
|
89
96
|
/**
|
|
90
|
-
* The hit the subject
|
|
91
|
-
*
|
|
97
|
+
* The hit the subject scores under.
|
|
98
|
+
*
|
|
99
|
+
* Always `matches[0]`, built together in one place so they cannot disagree.
|
|
92
100
|
*/
|
|
93
101
|
match: POIPhraseMatch
|
|
94
102
|
/**
|
|
95
|
-
* Every category the subject reaches, `match` first
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
*
|
|
99
|
-
* nothing downstream may read position as rank.
|
|
103
|
+
* Every category the subject reaches, `match` first: one entry unless the lookup
|
|
104
|
+
* returned a {@link POIPhraseMatch.searchAsSet} set, in which case it holds the
|
|
105
|
+
* whole set and the POI branch searches their union.
|
|
106
|
+
* The order is the lookup's and states no preference.
|
|
100
107
|
*/
|
|
101
108
|
matches: POIPhraseMatch[]
|
|
102
|
-
/**
|
|
103
|
-
* The matched subject text as it appeared in the query.
|
|
104
|
-
*/
|
|
105
109
|
subject: string
|
|
106
110
|
subjectSpan: POIQuerySpan
|
|
107
111
|
/**
|
|
@@ -117,50 +121,52 @@ export interface POISubjectMatch {
|
|
|
117
121
|
}
|
|
118
122
|
|
|
119
123
|
/**
|
|
120
|
-
* Anchor separator between subject and place: comma, or near/in/at/around/to —
|
|
121
|
-
* hits the lexicon.
|
|
124
|
+
* Anchor separator between subject and place: comma, or near/in/at/around/to —
|
|
125
|
+
* scanned left-to-right until a prefix hits the lexicon.
|
|
126
|
+
*
|
|
127
|
+
* Linear by construction (no polynomial ReDoS): neither alternative places an unbounded
|
|
128
|
+
* whitespace quantifier before its required literal, the classic `js/polynomial-redos` shape.
|
|
129
|
+
* The comma alternative starts at the literal `,`.
|
|
122
130
|
*
|
|
123
|
-
*
|
|
124
|
-
*
|
|
125
|
-
*
|
|
126
|
-
* before a fixed anchor word. Every remaining quantifier (`,\s*`, `…\s+`) is _trailing_ — it runs only after the
|
|
127
|
-
* required literal has already matched and nothing follows it, so it never backtracks. Each start offset does O(1)
|
|
128
|
-
* work, making `matchAll` O(n).
|
|
131
|
+
* The anchor alternative starts at one `\s` before a fixed anchor word.
|
|
132
|
+
* Every remaining quantifier is trailing and runs only after the required literal matches.
|
|
133
|
+
* Each start offset does O(1) work, so `matchAll` is O(n).
|
|
129
134
|
*
|
|
130
|
-
*
|
|
131
|
-
*
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
* lastIndex) identical, preserving the exact subsequent-match sequence. Verified: 0 divergences across 22.7k inputs
|
|
135
|
-
* (systematic + fuzzed adversarial whitespace + shared-whitespace anchor/comma chains).
|
|
135
|
+
* Behavior is byte-identical to the previous `\s*,\s*|\s+(?:…)\s+`, because `matchPOISubject`
|
|
136
|
+
* trims both the subject and the remainder, so surrounding whitespace is redundant.
|
|
137
|
+
* The leading quantifier only shifted the match start within a whitespace run, while the retained
|
|
138
|
+
* trailing greedy quantifier keeps the match end and thus `matchAll`'s lastIndex identical.
|
|
136
139
|
*/
|
|
137
140
|
const ANCHOR_SEPARATOR = /,\s*|\s(near|in|at|around|to)\s+/gi
|
|
138
141
|
|
|
139
142
|
/**
|
|
140
|
-
* Longest subject
|
|
143
|
+
* Longest subject accepted, in tokens: eight covers compound taxonomy phrases
|
|
144
|
+
* while bounding lexicon probes.
|
|
141
145
|
*/
|
|
142
146
|
const MAX_SUBJECT_TOKENS = 8
|
|
143
147
|
|
|
144
148
|
/**
|
|
145
|
-
* The categories one candidate subject reaches
|
|
149
|
+
* The categories one candidate subject reaches: the whole array when the first hit
|
|
150
|
+
* declares {@link POIPhraseMatch.searchAsSet}, preserved as the lookup returned it
|
|
151
|
+
* and never filtered, so a rung that flagged only some members keeps every member.
|
|
146
152
|
*
|
|
147
|
-
* The
|
|
148
|
-
*
|
|
149
|
-
* visible to whoever reads the set. Otherwise the first hit alone, which is the preference-list reading the committed
|
|
150
|
-
* phrase index has always had.
|
|
153
|
+
* The inconsistency stays visible.
|
|
154
|
+
* Otherwise, use only the first hit.
|
|
151
155
|
*/
|
|
152
156
|
function reachedMatches(hits: ReadonlyArray<POIPhraseMatch>): POIPhraseMatch[] {
|
|
153
157
|
return hits[0]!.searchAsSet ? [...hits] : [hits[0]!]
|
|
154
158
|
}
|
|
155
159
|
|
|
156
160
|
/**
|
|
157
|
-
* Match a POI subject: the whole input, or the text before the
|
|
158
|
-
*
|
|
159
|
-
*
|
|
160
|
-
*
|
|
161
|
+
* Match a POI subject: the whole input, or the text before the first anchor
|
|
162
|
+
* separator whose prefix hits the lexicon (≤ 8 tokens).
|
|
163
|
+
*
|
|
164
|
+
* Scans separators left-to-right, because a lexicon phrase may itself contain a bare separator
|
|
165
|
+
* word ("walk in clinic") and the first separator is not necessarily the right split point.
|
|
166
|
+
* Returns null when the lexicon never fires, including comma-ridden full addresses
|
|
167
|
+
* whose leading segment is not a phrase.
|
|
161
168
|
*
|
|
162
|
-
* The winning candidate's hits are
|
|
163
|
-
* declared one.
|
|
169
|
+
* The winning candidate's hits are listed per {@link reachedMatches}.
|
|
164
170
|
*/
|
|
165
171
|
export function matchPOISubject(
|
|
166
172
|
text: string,
|
|
@@ -191,8 +197,8 @@ export function matchPOISubject(
|
|
|
191
197
|
|
|
192
198
|
const subject = trimmed.slice(0, separator.index).trim()
|
|
193
199
|
|
|
194
|
-
// Subjects only grow as the scan moves right
|
|
195
|
-
// split
|
|
200
|
+
// Subjects only grow as the scan moves right, so once over budget later splits are too.
|
|
201
|
+
// Whitespace-only split rather than `wordsOf`, because a comma inside a subject is real content.
|
|
196
202
|
if (subject.split(/\s+/).length > MAX_SUBJECT_TOKENS) break
|
|
197
203
|
|
|
198
204
|
const hits = lookup(subject, locale)
|
|
@@ -237,10 +243,12 @@ export function matchPOISubject(
|
|
|
237
243
|
}
|
|
238
244
|
|
|
239
245
|
/**
|
|
240
|
-
* `poi_query` scorer over an injected lexicon
|
|
241
|
-
* 0.88 ceiling
|
|
242
|
-
*
|
|
243
|
-
*
|
|
246
|
+
* `poi_query` scorer over an injected lexicon, with confidence bands: a whole-input lexicon hit
|
|
247
|
+
* scores 0.92 (above venue-landmark's 0.88 ceiling, because an exact phrase beats a shape heuristic)
|
|
248
|
+
* and a subject plus anchor 0.9.
|
|
249
|
+
*
|
|
250
|
+
* The guards keep venue-led full addresses (class 2) on the structured-address path:
|
|
251
|
+
* a remainder leading with a house number, or a 4+-segment input, scores 0.
|
|
244
252
|
*/
|
|
245
253
|
export function createScorePOIQuery(
|
|
246
254
|
lookup: POIPhraseLookup,
|
|
@@ -265,21 +273,24 @@ export function createScorePOIQuery(
|
|
|
265
273
|
}
|
|
266
274
|
|
|
267
275
|
/**
|
|
268
|
-
* Confidence band for a bare category
|
|
269
|
-
*
|
|
270
|
-
*
|
|
271
|
-
*
|
|
276
|
+
* Confidence band for a bare category, one notch above `poi_query`'s whole-input
|
|
277
|
+
* band (0.92) so only the anchorless subset takes the top slot and every anchored
|
|
278
|
+
* POI query keeps scoring `poi_query` as before.
|
|
279
|
+
*
|
|
280
|
+
* The coordinator's POI branch accepts both kinds, so the routing is identical either way.
|
|
281
|
+
* The split exists so the marker can say "you supplied a category and no place".
|
|
272
282
|
*/
|
|
273
283
|
const POI_CATEGORY_CONFIDENCE = 0.93
|
|
274
284
|
|
|
275
285
|
/**
|
|
276
|
-
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
277
|
-
* "drinking fountain".
|
|
286
|
+
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
287
|
+
* to search: "tacos", "grocery store", "drinking fountain".
|
|
288
|
+
*
|
|
289
|
+
* Fires only on a whole-input lexicon hit (`remainder === ""`) whose subject is a category.
|
|
290
|
+
* A brand (`kind: "brand"`) is excluded because `POIPhraseMatch.categoryID`
|
|
291
|
+
* then holds the brand's display name.
|
|
278
292
|
*
|
|
279
|
-
*
|
|
280
|
-
* is excluded: a bare "Starbucks" is a name lookup, not a category, and the taxonomy id a category marker promises to
|
|
281
|
-
* carry does not exist for it — `POIPhraseMatch.categoryID` holds the brand's display name in that case, which would
|
|
282
|
-
* make the marker's `categoryID` evidence a lie.
|
|
293
|
+
* That value would misrepresent the marker's `categoryID` evidence.
|
|
283
294
|
*/
|
|
284
295
|
export function createScorePOICategory(
|
|
285
296
|
lookup: POIPhraseLookup,
|
|
@@ -297,8 +308,9 @@ export function createScorePOICategory(
|
|
|
297
308
|
}
|
|
298
309
|
|
|
299
310
|
/**
|
|
300
|
-
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence
|
|
301
|
-
*
|
|
311
|
+
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence;
|
|
312
|
+
* `null` when the input is not a bare category, under the same conditions as
|
|
313
|
+
* {@link createScorePOICategory} so the two cannot disagree.
|
|
302
314
|
*/
|
|
303
315
|
export function matchPOICategory(
|
|
304
316
|
text: string,
|