@mailwoman/kind-classifier 10.0.0 → 10.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/poi.ts CHANGED
@@ -3,16 +3,18 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * POI subject detection for the `poi_query` kind. The lexicon is INJECTED (`POIPhraseLookup`) —
7
- * this package keeps its bitter-lesson invariant (no dictionaries in-tree); the phrase table
8
- * lives in `@mailwoman/poi-taxonomy` and is wired in by `createRuntimePipeline` behind the
9
- * `poiQueryKind` flag (default-ON since 2026-07-20). Spec §3.1.
6
+ * POI subject detection for the `poi_query` kind. The lexicon is injected (`POIPhraseLookup`) so this
7
+ * package keeps its bitter-lesson invariant (no dictionaries in-tree); the phrase table lives in
8
+ * `@mailwoman/poi-taxonomy` and is wired in by `createRuntimePipeline` behind the `poiQueryKind` flag.
9
+ * Spec §3.1.
10
10
  */
11
11
 
12
12
  import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
13
13
  /**
14
- * Comma-segment ceiling for a POI-led query. Past it the input is a venue plus a full address (`X, 350 5th Ave, New
15
- * York, NY`), which the structured-address scorer should claim instead.
14
+ * Comma-segment ceiling for a POI-led query.
15
+ *
16
+ * Past it the input is a venue plus a full address (`X, 350 5th Ave, New York, NY`),
17
+ * which the structured-address scorer should claim instead.
16
18
  */
17
19
  const MAX_POI_SEGMENTS = 3
18
20
 
@@ -21,9 +23,9 @@ const MAX_POI_SEGMENTS = 3
21
23
  */
22
24
  export interface POIPhraseMatch {
23
25
  /**
24
- * The matched subject's identifier string. For `kind: "category"`, a `@mailwoman/poi-taxonomy` category id; for
25
- * `kind: "brand"` or `kind: "name"`, the canonical display name. `matchPOISubject` treats it opaquely; the caller
26
- * (`mailwoman`'s `poi-intent.ts`) interprets it per `kind`.
26
+ * The matched subject's identifier: a `@mailwoman/poi-taxonomy` category id for
27
+ * `kind: "category"`, otherwise the canonical display name; `matchPOISubject`
28
+ * treats it opaquely and the caller interprets it per `kind`.
27
29
  */
28
30
  categoryID: string
29
31
  matchedPhrase: string
@@ -31,43 +33,46 @@ export interface POIPhraseMatch {
31
33
  mechanism?: "exact" | "locale_normalized" | "typo"
32
34
  inputPhrase?: string
33
35
  /**
34
- * Absent = "category" (the pre-brand shape) — optional so pre-7.3 POIPhraseLookup implementors stay
35
- * source-compatible.
36
+ * Absent means `"category"`; optional so existing `POIPhraseLookup` implementors stay source-compatible.
36
37
  */
37
38
  kind?: "category" | "brand" | "name"
38
39
  /**
39
- * Wikidata QID, when known. `kind: "brand"` only — absent when a brand resolved by name alone (no QID match).
40
+ * Wikidata QID when known, `kind: "brand"` only.
41
+ * Absent when a brand resolved using its name only.
40
42
  */
41
43
  wikidata?: string
42
44
  /**
43
- * Whether this hit is one member of a set the caller must search TOGETHER, rather than one candidate in a preference
44
- * list.
45
+ * Whether this hit is one member of a set the caller must search together
46
+ * rather than one candidate in a preference list.
47
+ *
48
+ * A lookup returning several hits means two different things: a phrase index
49
+ * returns the categories one typed phrase could name, the curated reading first
50
+ * (`credit union` → the `bank` rollup its synonym redirects to), while an affordance
51
+ * rung returns every entity kind that affords one activity in a stable enumeration.
52
+ * The enumeration does not express a preference.
45
53
  *
46
- * A lookup returning several hits means two different things, and the difference decides whether narrowing to the
47
- * first is an answer or an invented ordering. A phrase index returns the categories one typed phrase could name, the
48
- * curated reading first (`credit union` → the `bank` rollup its synonym redirects to, then the standalone
49
- * `credit_union` category), and the first entry IS the subject. An affordance rung returns every entity kind that
50
- * affords ONE activity, in a stable enumeration that is not a preference, and taking the first picks a winner nobody
51
- * authored.
54
+ * The first result would impose an ordering that the source does not provide.
55
+ * Set on every member of such a set, so {@link matchPOISubject} returns them all
56
+ * and the POI branch searches their union.
52
57
  *
53
- * Set on every member of such a set. {@link matchPOISubject} then carries them all, the POI branch searches their
54
- * union, and the candidate ordering the resolver already owns decides the answer. Absent — the committed lexicon's
55
- * shape — keeps the first-hit reading unchanged.
58
+ * Absent, the committed lexicon's shape, keeps the first-hit reading.
56
59
  */
57
60
  searchAsSet?: boolean
58
61
  /**
59
- * ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to. Absent = the claim holds
60
- * everywhere.
62
+ * ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to.
63
+ * Absent means the condition is true everywhere.
61
64
  *
62
- * A scope is a statement about ESTABLISHMENTS, so it is judged against the country of the place being searched, not
63
- * the caller's locale: the locale is the lens the phrase is read through, and it says nothing about where the claim
64
- * holds. `matchPOISubject` carries the value untouched; the POI intent stage binds it once the anchor has resolved.
65
+ * A scope is a statement about establishments, so it is judged against the country of
66
+ * the place being searched rather than the caller's locale: the locale is the lens the
67
+ * phrase is read through and makes no statement about where the condition is true.
68
+ * `matchPOISubject` returns the value unchanged.
69
+ * The POI intent stage binds it once the anchor has resolved.
65
70
  */
66
71
  countryScope?: readonly string[]
67
72
  }
68
73
 
69
74
  /**
70
- * Injected phrase→category lookup. Exact-phrase, locale-aware; returns [] on miss.
75
+ * Injected phrase→category lookup, exact-phrase and locale-aware, returning `[]` on miss.
71
76
  */
72
77
  export type POIPhraseLookup = (phrase: string, locale?: string) => ReadonlyArray<POIPhraseMatch>
73
78
 
@@ -83,25 +88,24 @@ export interface POIQuerySpan {
83
88
  }
84
89
 
85
90
  /**
86
- * Which lexicon this hit came from. Existing category lookups set `"category"` (backward-compatible default).
91
+ * Which lexicon this hit came from.
92
+ *
93
+ * Category lookups set `"category"` as the backward-compatible default.
87
94
  */
88
95
  export interface POISubjectMatch {
89
96
  /**
90
- * The hit the subject SCORES under — its kind and its confidence. Always `matches[0]`; the two are built together in
91
- * one place so they cannot disagree.
97
+ * The hit the subject scores under.
98
+ *
99
+ * Always `matches[0]`, built together in one place so they cannot disagree.
92
100
  */
93
101
  match: POIPhraseMatch
94
102
  /**
95
- * Every category the subject reaches, `match` first.
96
- *
97
- * One entry unless the lookup returned a {@link POIPhraseMatch.searchAsSet} set, in which case it holds the whole set
98
- * and the POI branch searches their union. The order is the order the lookup returned and states no preference:
99
- * nothing downstream may read position as rank.
103
+ * Every category the subject reaches, `match` first: one entry unless the lookup
104
+ * returned a {@link POIPhraseMatch.searchAsSet} set, in which case it holds the
105
+ * whole set and the POI branch searches their union.
106
+ * The order is the lookup's and states no preference.
100
107
  */
101
108
  matches: POIPhraseMatch[]
102
- /**
103
- * The matched subject text as it appeared in the query.
104
- */
105
109
  subject: string
106
110
  subjectSpan: POIQuerySpan
107
111
  /**
@@ -117,50 +121,52 @@ export interface POISubjectMatch {
117
121
  }
118
122
 
119
123
  /**
120
- * Anchor separator between subject and place: comma, or near/in/at/around/to — scanned left-to-right until a prefix
121
- * hits the lexicon.
124
+ * Anchor separator between subject and place: comma, or near/in/at/around/to —
125
+ * scanned left-to-right until a prefix hits the lexicon.
126
+ *
127
+ * Linear by construction (no polynomial ReDoS): neither alternative places an unbounded
128
+ * whitespace quantifier before its required literal, the classic `js/polynomial-redos` shape.
129
+ * The comma alternative starts at the literal `,`.
122
130
  *
123
- * Linear by construction (no polynomial ReDoS): neither alternative places an unbounded whitespace quantifier _before_
124
- * its required literal — the classic `\s*`/`\s+`-then-literal backtracking shape that CodeQL's `js/polynomial-redos`
125
- * flags. The comma alternative starts at the literal `,`; the anchor alternative starts at a single `\s` immediately
126
- * before a fixed anchor word. Every remaining quantifier (`,\s*`, `…\s+`) is _trailing_ — it runs only after the
127
- * required literal has already matched and nothing follows it, so it never backtracks. Each start offset does O(1)
128
- * work, making `matchAll` O(n).
131
+ * The anchor alternative starts at one `\s` before a fixed anchor word.
132
+ * Every remaining quantifier is trailing and runs only after the required literal matches.
133
+ * Each start offset does O(1) work, so `matchAll` is O(n).
129
134
  *
130
- * Behaviour is byte-identical to the previous `\s*,\s*|\s+(?:…)\s+` because `matchPOISubject` `.trim()`s both the
131
- * subject (text before `.index`) and the remainder (text after the match), so surrounding whitespace on either side of
132
- * the separator is redundant. The leading `\s*`/`\s+` only shifted the match _start_ within a whitespace run — trim
133
- * absorbs that — while the retained _trailing_ greedy quantifier keeps the match _end_ (and thus `matchAll`'s
134
- * lastIndex) identical, preserving the exact subsequent-match sequence. Verified: 0 divergences across 22.7k inputs
135
- * (systematic + fuzzed adversarial whitespace + shared-whitespace anchor/comma chains).
135
+ * Behavior is byte-identical to the previous `\s*,\s*|\s+(?:…)\s+`, because `matchPOISubject`
136
+ * trims both the subject and the remainder, so surrounding whitespace is redundant.
137
+ * The leading quantifier only shifted the match start within a whitespace run, while the retained
138
+ * trailing greedy quantifier keeps the match end and thus `matchAll`'s lastIndex identical.
136
139
  */
137
140
  const ANCHOR_SEPARATOR = /,\s*|\s(near|in|at|around|to)\s+/gi
138
141
 
139
142
  /**
140
- * Longest subject we accept, in tokens. Eight covers compound taxonomy phrases while bounding lexicon probes.
143
+ * Longest subject accepted, in tokens: eight covers compound taxonomy phrases
144
+ * while bounding lexicon probes.
141
145
  */
142
146
  const MAX_SUBJECT_TOKENS = 8
143
147
 
144
148
  /**
145
- * The categories one candidate subject reaches, from the hits the lookup returned for it.
149
+ * The categories one candidate subject reaches: the whole array when the first hit
150
+ * declares {@link POIPhraseMatch.searchAsSet}, preserved as the lookup returned it
151
+ * and never filtered, so a rung that flagged only some members keeps every member.
146
152
  *
147
- * The whole array when the first hit declares {@link POIPhraseMatch.searchAsSet} — carried as the lookup returned it,
148
- * never filtered, so a rung that flagged only some of its members loses nothing here and the inconsistency stays
149
- * visible to whoever reads the set. Otherwise the first hit alone, which is the preference-list reading the committed
150
- * phrase index has always had.
153
+ * The inconsistency stays visible.
154
+ * Otherwise, use only the first hit.
151
155
  */
152
156
  function reachedMatches(hits: ReadonlyArray<POIPhraseMatch>): POIPhraseMatch[] {
153
157
  return hits[0]!.searchAsSet ? [...hits] : [hits[0]!]
154
158
  }
155
159
 
156
160
  /**
157
- * Match a POI subject: the whole input, or the text before the FIRST anchor separator WHOSE PREFIX HITS THE LEXICON (≤
158
- * 8 tokens). Scans separator occurrences left-to-right — a lexicon phrase may itself contain a bare separator word
159
- * (e.g. "walk in clinic"), so the first separator isn't necessarily the right split point. Returns null when the
160
- * lexicon never fires — including comma-ridden full addresses whose leading segment isn't a lexicon phrase.
161
+ * Match a POI subject: the whole input, or the text before the first anchor
162
+ * separator whose prefix hits the lexicon (≤ 8 tokens).
163
+ *
164
+ * Scans separators left-to-right, because a lexicon phrase may itself contain a bare separator
165
+ * word ("walk in clinic") and the first separator is not necessarily the right split point.
166
+ * Returns null when the lexicon never fires, including comma-ridden full addresses
167
+ * whose leading segment is not a phrase.
161
168
  *
162
- * The winning candidate's hits are carried per {@link reachedMatches}: the first hit, or the whole set when the lookup
163
- * declared one.
169
+ * The winning candidate's hits are listed per {@link reachedMatches}.
164
170
  */
165
171
  export function matchPOISubject(
166
172
  text: string,
@@ -191,8 +197,8 @@ export function matchPOISubject(
191
197
 
192
198
  const subject = trimmed.slice(0, separator.index).trim()
193
199
 
194
- // Subjects only grow as the scan moves right — once over budget, later splits are too. Whitespace-only
195
- // split, not `wordsOf`: a comma inside a subject is real content here, not a separator to erase.
200
+ // Subjects only grow as the scan moves right, so once over budget later splits are too.
201
+ // Whitespace-only split rather than `wordsOf`, because a comma inside a subject is real content.
196
202
  if (subject.split(/\s+/).length > MAX_SUBJECT_TOKENS) break
197
203
 
198
204
  const hits = lookup(subject, locale)
@@ -237,10 +243,12 @@ export function matchPOISubject(
237
243
  }
238
244
 
239
245
  /**
240
- * `poi_query` scorer over an injected lexicon. Confidence bands: whole-input lexicon hit 0.92 (above venue-landmark's
241
- * 0.88 ceiling — an exact lexicon phrase beats a shape heuristic); subject + anchor 0.9. Guards below keep venue-led
242
- * FULL addresses (class 2) on the structured-address path: a remainder that leads with a house number, or a 4+-segment
243
- * input, scores 0 here.
246
+ * `poi_query` scorer over an injected lexicon, with confidence bands: a whole-input lexicon hit
247
+ * scores 0.92 (above venue-landmark's 0.88 ceiling, because an exact phrase beats a shape heuristic)
248
+ * and a subject plus anchor 0.9.
249
+ *
250
+ * The guards keep venue-led full addresses (class 2) on the structured-address path:
251
+ * a remainder leading with a house number, or a 4+-segment input, scores 0.
244
252
  */
245
253
  export function createScorePOIQuery(
246
254
  lookup: POIPhraseLookup,
@@ -265,21 +273,24 @@ export function createScorePOIQuery(
265
273
  }
266
274
 
267
275
  /**
268
- * Confidence band for a bare category. One notch above `poi_query`'s whole-input band (0.92) so the anchorless subset
269
- * takes the top slot from it, and only from it — every anchored POI query keeps scoring `poi_query` exactly as before.
270
- * The coordinator's POI branch accepts both kinds, so the routing is identical either way; the split exists so the
271
- * marker can say "you named a category and no place", which is a different thing to tell a caller.
276
+ * Confidence band for a bare category, one notch above `poi_query`'s whole-input
277
+ * band (0.92) so only the anchorless subset takes the top slot and every anchored
278
+ * POI query keeps scoring `poi_query` as before.
279
+ *
280
+ * The coordinator's POI branch accepts both kinds, so the routing is identical either way.
281
+ * The split exists so the marker can say "you supplied a category and no place".
272
282
  */
273
283
  const POI_CATEGORY_CONFIDENCE = 0.93
274
284
 
275
285
  /**
276
- * `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere to search: "tacos", "grocery store",
277
- * "drinking fountain".
286
+ * `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
287
+ * to search: "tacos", "grocery store", "drinking fountain".
288
+ *
289
+ * Fires only on a whole-input lexicon hit (`remainder === ""`) whose subject is a category.
290
+ * A brand (`kind: "brand"`) is excluded because `POIPhraseMatch.categoryID`
291
+ * then holds the brand's display name.
278
292
  *
279
- * Fires ONLY on a whole-input lexicon hit (`remainder === ""`) whose subject is a CATEGORY. A brand (`kind: "brand"`)
280
- * is excluded: a bare "Starbucks" is a name lookup, not a category, and the taxonomy id a category marker promises to
281
- * carry does not exist for it — `POIPhraseMatch.categoryID` holds the brand's display name in that case, which would
282
- * make the marker's `categoryID` evidence a lie.
293
+ * That value would misrepresent the marker's `categoryID` evidence.
283
294
  */
284
295
  export function createScorePOICategory(
285
296
  lookup: POIPhraseLookup,
@@ -297,8 +308,9 @@ export function createScorePOICategory(
297
308
  }
298
309
 
299
310
  /**
300
- * The whole-input category hit behind a `poi_category` verdict, for the marker's evidence. `null` when the input is not
301
- * a bare category — same conditions as {@link createScorePOICategory}, so the two cannot disagree.
311
+ * The whole-input category hit behind a `poi_category` verdict, for the marker's evidence;
312
+ * `null` when the input is not a bare category, under the same conditions as
313
+ * {@link createScorePOICategory} so the two cannot disagree.
302
314
  */
303
315
  export function matchPOICategory(
304
316
  text: string,