@mailwoman/kind-classifier 10.0.0 → 10.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,11 +1,11 @@
1
1
  # @mailwoman/kind-classifier
2
2
 
3
- **Stage 2.5 of the Mailwoman runtime pipeline** — query kind classification.
3
+ This package is **stage 2.5 of the Mailwoman runtime pipeline**, which classifies
4
+ the query kind.
4
5
 
5
- Categorizes an input into one of seven `QueryKind`s by composing rule-based
6
- scorers over the `QueryShape` output. Returns possibilities (alternatives)
7
- alongside the top pick so the coordinator can fall back when the winning kind
8
- isn't actionable.
6
+ It assigns an input to one of seven `QueryKind`s by combining rule-based scorers
7
+ over the `QueryShape` output. It returns alternatives alongside the top pick, so
8
+ the coordinator can fall back when the winning kind is not actionable.
9
9
 
10
10
  ```ts
11
11
  import { classifyKind } from "@mailwoman/kind-classifier"
@@ -51,18 +51,19 @@ locale-hint → kind-classifier → phrase-grouper → classifier → ...
51
51
 
52
52
  ## Design
53
53
 
54
- - **Pure functions, no ML.** Rule-based v1; a trained classifier is deferred.
54
+ - **Pure functions without ML.** Version 1 is rule-based, and a trained
55
+ classifier is deferred.
55
56
  - **Returns alternatives.** The coordinator might skip a `locality_only` parse
56
- and fall back to a `vague` handler — the alternatives list makes that possible.
57
+ and fall back to a `vague` handler, which the alternatives list makes possible.
57
58
  - **Consumes `QueryShape` + `LocaleHint`** from the two preceding stages.
58
59
 
59
60
  ## Related
60
61
 
61
- - [`@mailwoman/query-shape`](../query-shape) — feeds structural data into this stage
62
- - [`@mailwoman/locale-hint`](../locale-hint) — feeds locale context
63
- - [`@mailwoman/phrase-grouper`](../phrase-grouper) — Stage 2.7, next in the pipeline
64
- - [Staged Pipeline Contract](https://github.com/sister-software/mailwoman/blob/main/docs/engineering/reference/STAGES.mdx)
62
+ - [`@mailwoman/query-shape`](../query-shape): supplies structural data to this stage.
63
+ - [`@mailwoman/locale-hint`](../locale-hint): supplies locale context.
64
+ - [`@mailwoman/phrase-grouper`](../phrase-grouper): stage 2.7, next in the pipeline.
65
+ - [Staged Pipeline Interface](https://github.com/sister-software/mailwoman/blob/main/docs/records/plan/reference/STAGES.mdx)
65
66
 
66
67
  ## License
67
68
 
68
- [AGPL-3.0-only](https://www.gnu.org/licenses/agpl-3.0.html)
69
+ [AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
package/lib/classify.ts CHANGED
@@ -3,17 +3,13 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * `classifyKind` — entry point for Stage 2.5 (kind classification).
6
+ * `classifyKind` is the entry point for Stage 2.5 (kind classification), composing the per-kind rules
7
+ * from `rules.ts` and `intent-rules.ts` and returning alternatives sorted by confidence.
7
8
  *
8
- * Composes the per-kind rules from `rules.ts` and `intent-rules.ts` and picks the winner. Returns
9
- * alternatives sorted by confidence so the coordinator can offer fallback paths when the top kind
10
- * isn't actionable.
11
- *
12
- * Per the project's "possibilities not constraints" principle, every kind that fires above 0
13
- * surfaces in `alternatives` — the caller decides whether to act on the top kind only or consider
14
- * runner-ups. The ROAD_TO_V9 §4 intent vocabulary leans on that: `bare_toponym` and `route_pair`
15
- * are scored below their structural incumbent precisely so they land in `alternatives`, where they
16
- * inform the markers without moving the routing decision.
9
+ * Per the project's "possibilities not constraints" principle, every kind that fires above 0 surfaces
10
+ * in `alternatives` and the caller decides whether to act on the top kind only: `bare_toponym` and
11
+ * `route_pair` are scored below their structural incumbent precisely so they land in `alternatives`,
12
+ * where they inform the markers without moving the routing decision.
17
13
  */
18
14
 
19
15
  import type { LocaleHint, QueryIntentMarker, QueryKind, QueryKindResult } from "@mailwoman/core/pipeline"
@@ -45,9 +41,9 @@ const SCORERS: ReadonlyArray<KindScorer> = [
45
41
  { kind: "postcode_only", score: scorePostcodeOnly },
46
42
  { kind: "locality_only", score: scoreLocalityOnly },
47
43
  { kind: "structured_address", score: scoreStructuredAddress },
48
- // ROAD_TO_V9 §4. Ordinary members of the same list — intent is vocabulary, not a stage. `bare_toponym` and
49
- // `route_pair` are scored under `locality_only` on purpose (see `intent-rules.ts`), so their position here is
50
- // cosmetic; the sort below is what decides.
44
+ // Intent is vocabulary rather than a stage.
45
+ // The sort below decides, so `bare_toponym` and `route_pair` sit here cosmetically
46
+ // and are scored under `locality_only` on purpose (see `intent-rules.ts`).
51
47
  { kind: "bare_toponym", score: scoreBareToponym },
52
48
  { kind: "route_pair", score: scoreRoutePair },
53
49
  { kind: "near_me", score: scoreNearMe },
@@ -55,8 +51,8 @@ const SCORERS: ReadonlyArray<KindScorer> = [
55
51
  ]
56
52
 
57
53
  /**
58
- * Rank a scored list and shape it into a verdict. Shared by the lexicon-free and lexicon-wired paths so the two cannot
59
- * drift in how they break ties or build `alternatives`.
54
+ * Rank a scored list into a verdict, shared by the lexicon-free and lexicon-wired paths
55
+ * so the two cannot drift in how they break ties or build `alternatives`.
60
56
  */
61
57
  function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKindResult {
62
58
  scored.sort((a, b) => b.confidence - a.confidence)
@@ -71,14 +67,14 @@ function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKind
71
67
  }
72
68
 
73
69
  /**
74
- * Every kind whose verdict carries `intentMarkers`. Checked before the marker builder runs so the hot path — a
75
- * structured address, where none of these fire — pays one set membership test per kind and nothing else.
70
+ * Every kind whose verdict includes `intentMarkers`, checked before the marker builder
71
+ * so the hot path pays one set membership test per kind.
76
72
  */
77
- const MARKER_BEARING_KINDS: ReadonlySet<QueryKind> = new Set<QueryKind>(["route_pair", "near_me", "poi_category"])
73
+ const MARKER_KINDS: ReadonlySet<QueryKind> = new Set<QueryKind>(["route_pair", "near_me", "poi_category"])
78
74
 
79
75
  /**
80
- * Attach markers to a verdict, or return it untouched. Separate from {@link rank} because the lexicon-wired path needs
81
- * to merge `poi_query`/`poi_category` in first.
76
+ * Attach markers to a verdict or return it untouched, separate from {@link rank}
77
+ * because the lexicon-wired path merges `poi_query`/`poi_category` in first.
82
78
  */
83
79
  function withIntentMarkers(
84
80
  verdict: QueryKindResult,
@@ -88,7 +84,7 @@ function withIntentMarkers(
88
84
  ): QueryKindResult {
89
85
  const kinds = [{ kind: verdict.kind, confidence: verdict.confidence }, ...verdict.alternatives]
90
86
 
91
- if (!kinds.some((k) => MARKER_BEARING_KINDS.has(k.kind))) return verdict
87
+ if (!kinds.some((k) => MARKER_KINDS.has(k.kind))) return verdict
92
88
 
93
89
  const intentMarkers: QueryIntentMarker[] = deriveIntentMarkers(kinds, { input, poiLexicon, locale })
94
90
 
@@ -98,8 +94,8 @@ function withIntentMarkers(
98
94
  }
99
95
 
100
96
  /**
101
- * Classify the query shape into a `QueryKind`. Synchronous + pure — produces the same result for the same `(input,
102
- * shape)` pair.
97
+ * Classify the query shape into a `QueryKind`, synchronously and purely,
98
+ * producing the same result for the same `(input, shape)` pair.
103
99
  */
104
100
  export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLike): QueryKindResult {
105
101
  const scored = SCORERS.map((s) => ({ kind: s.kind, confidence: s.score(input, shape) })).filter(
@@ -110,9 +106,8 @@ export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLi
110
106
  }
111
107
 
112
108
  /**
113
- * Async variant matching the runtime-pipeline's `classifyKind` contract.
114
- *
115
- * The locale parameter is accepted for future locale-aware rules (Japanese honorifics, etc.) but not currently used.
109
+ * Async variant matching the runtime pipeline's `classifyKind` interface;
110
+ * `_locale` is accepted for future locale-aware rules but currently unused.
116
111
  */
117
112
  export async function classifyKind(
118
113
  input: NormalizedInputLite,
@@ -127,16 +122,19 @@ export async function classifyKind(
127
122
  */
128
123
  export interface KindClassifierOpts {
129
124
  /**
130
- * POI phrase lexicon (spec §3.1). When present, `poi_query` and `poi_category` scorers join the rule set — injected,
131
- * never imported, so this package stays dictionary-free. Absent → the returned classifier is behaviorally identical
132
- * to {@link classifyKind}.
125
+ * POI phrase lexicon (spec §3.1); when present the `poi_query`
126
+ * and `poi_category` scorers join the rule set.
127
+ *
128
+ * They are injected rather than imported so this package stays dictionary-free.
129
+ * When absent, the returned classifier behaves identically to {@link classifyKind}.
133
130
  */
134
131
  poiLexicon?: POIPhraseLookup
135
132
  }
136
133
 
137
134
  /**
138
- * Build a kind classifier. Without opts this is exactly the default {@link classifyKind}; with a `poiLexicon` it
139
- * additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4) and merges them into the ranked result.
135
+ * Build a kind classifier that is exactly {@link classifyKind} without options,
136
+ * or that additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4)
137
+ * and merges them into the ranked result when given a `poiLexicon`.
140
138
  */
141
139
  export function createKindClassifier(
142
140
  opts: KindClassifierOpts = {}
@@ -153,9 +151,10 @@ export function createKindClassifier(
153
151
 
154
152
  if (poiConfidence <= 0 && categoryConfidence <= 0) return base
155
153
 
156
- // Re-rank over the union rather than special-casing "did POI beat the base?". The base verdict's own
157
- // alternatives are preserved, which is what keeps `bare_toponym` / `route_pair` visible to the marker builder
158
- // even when a POI kind takes the top slot.
154
+ // Re-ranking over the union rather than special-casing whether POI beat the
155
+ // base preserves the base's own alternatives.
156
+ // Those keep `bare_toponym`/`route_pair` visible to the marker builder even
157
+ // when a POI kind takes the top slot.
159
158
  const merged: Array<{ kind: QueryKind; confidence: number }> = [
160
159
  { kind: base.kind, confidence: base.confidence },
161
160
  ...base.alternatives,
package/lib/index.ts CHANGED
@@ -3,14 +3,7 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * `@mailwoman/kind-classifier` — Stage 2.5 of the runtime pipeline.
7
- *
8
- * Categorize inputs into one of eight `QueryKind`s by composing rule-based scorers over the
9
- * QueryShape sub-system's output. Pure functions, no ML, no place-name dictionaries. Returns
10
- * possibilities (alternatives) alongside the top pick so the coordinator can fall back when the
11
- * winning kind isn't actionable.
12
- *
13
- * See `docs/engineering/reference/STAGES.md` § Stage 2.5 for the contract.
6
+ * Stage 2.5 classifier: compose rule-based scorers over QueryShape and return a top kind with alternatives.
14
7
  */
15
8
 
16
9
  export { classifyKind, classifyKindSync, createKindClassifier } from "#classify"
@@ -3,16 +3,15 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * Marker derivation for the ROAD_TO_V9 §4 intent vocabulary. Pure, synchronous, and the ONLY place
7
- * the classifier turns a fired rule into something a caller reads.
6
+ * Marker derivation for the ROAD_TO_V9 §4 intent vocabulary. This code is pure and synchronous. It is the only place the
7
+ * classifier turns a fired rule into something a caller reads.
8
8
  *
9
- * Three of the four intent kinds can raise their marker here, from the string alone. The fourth —
10
- * `bare_toponym`'s `declared_ambiguity` — cannot: its trigger is the dominance margin of the
11
- * RESOLVED candidate list, which does not exist yet at Stage 2.5. That one is raised by
12
- * `mailwoman/query-intent.ts` after the resolve, against the measured 0.5-log10 threshold. The split is
13
- * deliberate and it is why this module never emits `declared_ambiguity`: a marker that asserted
14
- * ambiguity from the string alone would be declaring that every bare city name is ambiguous, which
15
- * is false 89.1% of the time (the measured table behind `DECISIVE_MARGIN_LOG10`).
9
+ * Three of the four intent kinds can raise their marker here from the string by itself. The fourth,
10
+ * Stage 2.5 cannot raise `bare_toponym`'s `declared_ambiguity` from the string by itself.
11
+ * Its trigger is the dominance margin of the resolved candidate list. `mailwoman/query-intent.ts` raises it
12
+ * after the resolve against `DECISIVE_MARGIN_LOG10`. This module therefore never emits
13
+ * `declared_ambiguity`, since a marker asserting ambiguity from the string by itself would declare every
14
+ * bare city name ambiguous.
16
15
  */
17
16
 
18
17
  import type { QueryIntentMarker, QueryKind } from "@mailwoman/core/pipeline"
@@ -27,22 +26,22 @@ import { matchPOICategory, type POIPhraseLookup } from "#poi"
27
26
  export interface IntentMarkerContext {
28
27
  input: NormalizedInputLite
29
28
  /**
30
- * The injected POI lexicon, when one was wired. Absent → no `poi_category` marker can be built, which is consistent
31
- * because the kind cannot fire without it either.
29
+ * The injected POI lexicon when one was wired.
30
+ *
31
+ * Absent means no `poi_category` marker can be built, consistent with the
32
+ * kind not firing without it either.
32
33
  */
33
34
  poiLexicon?: POIPhraseLookup
34
35
  locale?: string
35
36
  }
36
37
 
37
38
  /**
38
- * Build the advisories for one classified query.
39
+ * Build the advisories for one classified query, taking the full verdict — top plus alternatives —
40
+ * because two of the four intent kinds live in `alternatives` by design (see `intent-rules.ts`).
39
41
  *
40
- * `kinds` is the FULL verdict — top plus alternatives — because two of the four intent kinds live in `alternatives` by
41
- * design (see `intent-rules.ts`). Reading only the top kind would make them invisible, which is the mistake this
42
- * signature exists to prevent.
43
- *
44
- * Returns `[]` when no intent kind fired. Callers surface that empty array rather than dropping the field: an empty
45
- * array is the classifier stating it looked.
42
+ * @returns `[]` when no intent kind fired.
43
+ * Callers surface that empty array rather than dropping the field, because an
44
+ * empty array is the classifier stating it looked.
46
45
  */
47
46
  export function deriveIntentMarkers(
48
47
  kinds: ReadonlyArray<{ kind: QueryKind; confidence: number }>,
@@ -52,8 +51,8 @@ export function deriveIntentMarkers(
52
51
  const markers: QueryIntentMarker[] = []
53
52
 
54
53
  if (fired.has("route_pair")) {
55
- // Whitespace-only split, not `wordsOf`: `route_pair` inputs are comma-free by construction (a comma
56
- // disqualifies the kind), and the tokens are re-joined verbatim into the message.
54
+ // Whitespace-only split rather than `wordsOf`, because `route_pair` inputs are
55
+ // comma-free by construction and the tokens are re-joined verbatim into the message.
57
56
  const tokens = ctx.input.normalized.trim().split(/\s+/)
58
57
 
59
58
  markers.push({
@@ -64,8 +63,11 @@ export function deriveIntentMarkers(
64
63
  evidence: {
65
64
  tokens,
66
65
  /**
67
- * Both readings, named. The order is stable (pair first, then the admin reading) so a consumer can index it; it
68
- * is NOT a ranking, and nothing downstream reads it as one.
66
+ * Both readings, listed.
67
+ *
68
+ * The order is stable: pair first, then the admin reading.
69
+ * A consumer can index it.
70
+ * This order does not rank the entries.
69
71
  */
70
72
  interpretations: ["two_toponyms", "locality_with_admin_context"],
71
73
  },
@@ -83,9 +85,10 @@ export function deriveIntentMarkers(
83
85
  evidence: {
84
86
  subject,
85
87
  /**
86
- * THE PLUG POINT, named but not wired (ROAD_TO_V9 §4.4 scopes v9 to classification). Photon's `/api` already
87
- * accepts `lat`/`lon` location-bias params — `photon/` is the eventual consumer of this marker, and this string
88
- * is the note that says where it plugs in.
88
+ * The plug point is documented but not wired.
89
+ *
90
+ * `photon/` is the eventual consumer.
91
+ * Its `/api` already accepts `lat`/`lon` location-bias params.
89
92
  */
90
93
  focusParameter: "photon:lat/lon",
91
94
  },
@@ -3,63 +3,59 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * ROAD_TO_V9 §4 — the query-INTENT rules. Same contract as `rules.ts` (a `(input, shape) => number`
7
- * in [0, 1], 0 when the rule does not fire) and the same bitter-lesson invariant: universal
8
- * structural patterns and bounded linguistic categories only, never a place-name dictionary. The
9
- * one lexicon these kinds consult — the POI synonym table — is INJECTED, exactly as `poi.ts`
10
- * already does it.
6
+ * ROAD_TO_V9 §4 — the query-intent rules, with the same `(input, shape) => number` interface as
7
+ * `rules.ts` and the same bitter-lesson invariant: universal structural patterns and bounded linguistic
8
+ * categories only, never a place-name dictionary, with the POI synonym table injected exactly as
9
+ * `poi.ts` does it.
11
10
  *
12
- * ## Why two of these three deliberately lose
11
+ * `bare_toponym` and `route_pair` score below the structural kind that already owns their population
12
+ * (`locality_only`, 0.85), so they surface in `QueryKindResult.alternatives` and never as the top kind.
13
+ * The top kind is the only thing the coordinator routes on, so pinning it is what makes these additions
14
+ * answer-neutral on the bare-city-name register. Their intent travels on the marker.
13
15
  *
14
- * `bare_toponym` and `route_pair` are scored BELOW the structural kind that already owns their
15
- * population (`locality_only`, 0.85). They therefore surface in `QueryKindResult.alternatives` and
16
- * never as the top kind. That is not timidity — it is the D-rule discharge. The top kind is the
17
- * only thing the coordinator routes on (`deriveInputMode`, `canShortCircuit`, the POI branch), so
18
- * pinning it is what makes these additions provably answer-neutral on the bare-city-name register,
19
- * which is the single largest population in map search. The intent they carry travels on the
20
- * marker instead, where it is advisory by construction.
21
- *
22
- * `near_me` DOES win its top slot (0.91), because there is no incumbent worth preserving: a query
23
- * ending "near me" is not a locality and answering it as one is the bug.
16
+ * `near_me` does win its top slot (0.91), because there is no incumbent worth preserving: a query
17
+ * ending "near me" is not a locality.
24
18
  */
25
19
 
26
20
  import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
27
21
 
28
- import { isDisqualifyingStreetSuffix, MAX_LOCALITY_ONLY_LENGTH, wordsOf } from "#rules"
22
+ import { carriesLetter, isDisqualifyingStreetSuffix, MAX_LOCALITY_ONLY_LENGTH, wordsOf } from "#rules"
29
23
  /**
30
- * `locality_only` scores 0.85. Both refinement kinds sit under it by a whole confidence step so no float-comparison
31
- * accident can flip the top slot, and so the gap reads as deliberate to the next person.
24
+ * Both refinement kinds sit a whole confidence step below `locality_only`'s 0.85
25
+ * so no float-comparison accident can flip the top slot.
32
26
  */
33
27
  const BARE_TOPONYM_CONFIDENCE = 0.84
34
28
 
35
29
  /**
36
- * Lower still, and for a second reason on top of the ranking discipline: a route pair is a HYPOTHESIS about a query
37
- * whose competing reading (locality + region) is more common in this corpus. The number states that.
30
+ * Lower still for a second reason beyond the ranking discipline.
31
+ *
32
+ * A route pair is a hypothesis whose competing reading (locality + region) is more common in this corpus.
38
33
  */
39
34
  const ROUTE_PAIR_CONFIDENCE = 0.55
40
35
 
41
36
  /**
42
- * Above `landmark`'s venue ceiling (0.88) and above `poi_query`'s anchored band (0.90), because a deictic tail is a
43
- * stronger signal than either shape heuristic: nothing else in the vocabulary explains why "me" is at the end of the
44
- * string.
37
+ * Above `landmark`'s venue ceiling (0.88) and above `poi_query`'s anchored band (0.90),
38
+ * because a deictic tail is a stronger signal than either shape heuristic:
39
+ * no other rule explains why "me" ends the string.
45
40
  */
46
41
  const NEAR_ME_CONFIDENCE = 0.91
47
42
 
48
43
  /**
49
- * Word ceiling for a single bare toponym. Four covers the long tail that actually exists as one place name ("Newcastle
50
- * upon Tyne", "Sault Sainte Marie", "Las Palmas de Gran Canaria"); past it the input is carrying more than a name.
44
+ * Word ceiling for a single bare toponym: four covers the long tail that exists as one
45
+ * place name ("Newcastle upon Tyne", "Sault Sainte Marie", "Las Palmas de Gran Canaria"),
46
+ * and past it the input contains more than a name.
51
47
  */
52
48
  const MAX_BARE_TOPONYM_WORDS = 4
53
49
 
54
50
  /**
55
- * Toponymic HEAD particles — the bounded linguistic category that makes a multi-token string ONE place name.
56
- *
57
- * Same justification, and the same boundary, as `@mailwoman/phrase-grouper`'s `PLACE_NAME_PARTICLES` (which covers the
58
- * INFIX glue: `de`, `am`, `aan den`). This set covers the PREFIX heads, and it exists for exactly one job: keeping
59
- * `route_pair` off "New York", "San Francisco", "Fort Worth" and their kin. It is a closed morphological class, not a
60
- * gazetteer — growing it with actual place names is the wrong move, and the pressure for that belongs on the resolver.
51
+ * Toponymic head particles — the bounded linguistic category that makes a multi-token
52
+ * string one place name, with the same boundary as `@mailwoman/phrase-grouper`'s
53
+ * `PLACE_NAME_PARTICLES` (which covers the infix glue `de`, `am`, `aan den`) and the one
54
+ * job of keeping `route_pair` off "New York", "San Francisco", "Fort Worth" and their kin.
61
55
  *
62
- * Case-folded on read, because lowercase is the primary user register and "new york" is the same query.
56
+ * It is a closed morphological class rather than a gazetteer, so growing it
57
+ * with actual place names is the wrong move.
58
+ * It is case-folded on read, because "new york" is the same query.
63
59
  */
64
60
  const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
65
61
  // English
@@ -122,9 +118,10 @@ const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
122
118
  "sint",
123
119
  // Definite article as a head — "The Valley" (Anguilla), "The Hague", "The Bottom".
124
120
  "the",
125
- // Generic toponymic heads outside the Latin/Germanic families, added because the 306-case corpus MEASURED them
126
- // (see `mailwoman/test/kind-intent-invariance.test.ts`): each is a common noun in its own language — Semitic "tel"
127
- // (mound), Malay "kuala" (confluence), Khmer "phnom" (hill) — that heads a place name the way "mount" does.
121
+ // Generic toponymic heads outside the Latin/Germanic families, each a common
122
+ // noun in its own language — Semitic "tel" (mound), Malay "kuala" (confluence),
123
+ // Khmer "phnom" (hill) — that heads a place name the way "mount" does.
124
+ // `mailwoman/test/kind-intent-invariance.test.ts` covers them.
128
125
  "tel",
129
126
  "kuala",
130
127
  "phnom",
@@ -135,11 +132,9 @@ const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
135
132
  ])
136
133
 
137
134
  /**
138
- * Generic toponymic TAIL nouns — the other half of the same bounded morphological class. "Belize City", "George Town",
139
- * "Cape Town", "Palm Springs": a place name whose last token is a settlement/landform generic is ONE name, not two.
140
- *
141
- * Measured additions, same as the heads above: `city`, `town` and `valley` each came off a real corpus row that was
142
- * forking wrongly.
135
+ * Generic toponymic tail nouns — the other half of the same bounded morphological
136
+ * class, so a place name whose last token is a settlement/landform generic
137
+ * ("Belize City", "George Town", "Palm Springs") is one name rather than two.
143
138
  */
144
139
  const TOPONYM_TAIL_NOUNS: ReadonlySet<string> = new Set([
145
140
  "city",
@@ -167,53 +162,54 @@ const TOPONYM_TAIL_NOUNS: ReadonlySet<string> = new Set([
167
162
  ])
168
163
 
169
164
  /**
170
- * Deictic locator tails — "near me", "nearby", "around here", "in my area".
165
+ * Deictic locator tails — "near me", "nearby", "around here", "in my area" —
166
+ * the bounded class `preposition + a reference to the asker`, where `me`, `here`,
167
+ * `my <noun>` are function words rather than places.
171
168
  *
172
- * The class is `preposition + a reference to the ASKER`, which is why it is bounded and why it is safe: `me`, `here`,
173
- * `my <noun>` are function words, not places. Anchored to the END of the string (`$`) on purpose — the whole point of
174
- * the kind is that the query names no anchor, so anything AFTER the locator is an anchor and disqualifies it.
169
+ * Anchored to the end of the string (`$`) on purpose: the query names no anchor,
170
+ * so anything after the locator is an anchor and disqualifies it.
175
171
  *
176
- * Linear by construction: every alternative begins with a required literal, and the only quantifiers are bounded `\s+`
177
- * runs BETWEEN two required literals or trailing before `$`. No unbounded-whitespace-then-literal prefix, which is the
178
- * `js/polynomial-redos` shape (see the `ANCHOR_SEPARATOR` docstring in `poi.ts` for the same analysis).
172
+ * Linear by construction: every alternative begins with a required literal and the only
173
+ * quantifiers are bounded `\s+` runs between two required literals or trailing before `$`,
174
+ * with no unbounded-whitespace-then-literal prefix (the `js/polynomial-redos` shape).
175
+ * `ANCHOR_SEPARATOR` in `poi.ts` uses the same analysis.
179
176
  */
180
177
  const DEICTIC_LOCATOR_TAIL =
181
178
  /\b(?:near|close\s+to|next\s+to|around|by|closest\s+to|nearest\s+to)\s+(?:me|us|here|my\s+(?:location|position|area|place|house|home))\s*$/
182
179
 
183
180
  /**
184
- * The adverbial half of the same class — no preposition, the deixis is baked into the word.
181
+ * The adverbial half of the same class, where the deixis is baked into the word
182
+ * rather than introduced by a preposition.
185
183
  */
186
184
  const DEICTIC_ADVERB_TAIL =
187
185
  /\b(?:nearby|near\s?by|close\s+by|around\s+here|in\s+my\s+(?:area|neighborhood|neighbourhood))\s*$/
188
186
 
189
- /**
190
- * True when the input carries a deictic locator tail in EITHER form.
191
- */
192
187
  function hasDeicticTail(lowercased: string): boolean {
193
188
  return DEICTIC_LOCATOR_TAIL.test(lowercased) || DEICTIC_ADVERB_TAIL.test(lowercased)
194
189
  }
195
190
 
196
191
  /**
197
- * The conditions `bare_toponym` and `route_pair` share: no address grammar of any kind, one segment, alpha throughout.
192
+ * The conditions `bare_toponym` and `route_pair` share: no address grammar of any kind,
193
+ * one segment, alpha throughout.
198
194
  *
199
- * Returns the word list when the input clears them, `null` when it does not. Deliberately a SUPERSET of
200
- * `scoreLocalityOnly`'s conditions (which admit two segments), so `bare_toponym` is a strict refinement of
201
- * `locality_only` and can never fire where `locality_only` did not — the property `intent-rules.test.ts` asserts and
202
- * the reason the ranking discipline above is enough to keep the top kind pinned.
195
+ * @returns the word list when the input clears them, `null` when it does not.
196
+ * The conditions are deliberately a superset of `scoreLocalityOnly`'s, so `bare_toponym` is
197
+ * a strict refinement of `locality_only` and can never fire where `locality_only` did not.
203
198
  */
204
199
  function bareNameWords(input: NormalizedInputLite, shape: QueryShapeLike): string[] | null {
205
200
  const text = input.normalized.trim()
206
201
 
207
202
  if (!text || text.length > MAX_LOCALITY_ONLY_LENGTH) return null
208
203
 
209
- // A recognized postcode/known format IS address grammar. Nothing bare survives this.
204
+ // A recognized postcode/known format is address grammar, so no bare toponym survives it.
210
205
  if (shape.knownFormats.length) return null
211
206
 
212
- // `alpha` excludes every house number and every postcode by construction — the cheapest available statement of
213
- // "no address grammar", and it costs no lexicon.
214
- if (shape.characterClass !== "alpha") return null
207
+ // `alpha` excludes every house number and postcode by construction, the cheapest statement of
208
+ // "no address grammar"; it is silent about whether a name is present, so the letter test stands
209
+ // beside it because `foldInputClass` answers `alpha` for input carrying no classified token.
210
+ if (shape.characterClass !== "alpha" || !carriesLetter(text)) return null
215
211
 
216
- // A comma is the admin-context marker ("Paris, FR"). One segment, or the name is not bare.
212
+ // A comma is the admin-context marker ("Paris, FR"), so one segment or the name is not bare.
217
213
  if ((shape.segments?.length ?? 1) !== 1) return null
218
214
 
219
215
  const lowercased = text.toLowerCase()
@@ -232,27 +228,28 @@ function bareNameWords(input: NormalizedInputLite, shape: QueryShapeLike): strin
232
228
  }
233
229
 
234
230
  /**
235
- * `bare_toponym` rule: a single coherent place-name carrying no address grammar.
236
- *
237
- * Feeds the declared-ambiguity path. The rule itself asserts nothing about WHICH place — that is the resolver's
238
- * question, and `mailwoman/query-intent.ts` is where the answer's dominance margin decides whether the ambiguity gets
239
- * declared.
231
+ * `bare_toponym` rule: a single coherent place-name carrying no address grammar, feeding the
232
+ * declared-ambiguity path without asserting which place — that is the resolver's question,
233
+ * decided by the answer's dominance margin in `mailwoman/query-intent.ts`.
240
234
  */
241
235
  export function scoreBareToponym(input: NormalizedInputLite, shape: QueryShapeLike): number {
242
236
  return bareNameWords(input, shape) ? BARE_TOPONYM_CONFIDENCE : 0
243
237
  }
244
238
 
245
239
  /**
246
- * `route_pair` rule: exactly two toponym-shaped tokens with nothing between them.
240
+ * `route_pair` rule: exactly two toponym-shaped tokens with no token between them.
241
+ *
242
+ * The known confound is structural and unfixable here.
243
+ * "Paris London" and "Moscow Idaho" have the same string shape.
247
244
  *
248
- * **The known confound is structural and unfixable here.** "Paris London" and "Moscow Idaho" are the same string shape
249
- * — two bare capitalized words — and separating them needs to know that Idaho is a region, which is a gazetteer fact,
250
- * not a structural one. The hard-slice board's 18 `comma_free` rows are that population, and they fire this rule. That
251
- * is the reason ROAD_TO_V9 §4.3 specifies **classification + a declared fork, never a router**: both readings are named
252
- * in the marker, neither wins, and the resolver keeps answering exactly as it did.
245
+ * A classifier needs gazetteer knowledge that Idaho is a region to distinguish them.
246
+ * ROAD_TO_V9 §4.3 therefore specifies classification plus a declared fork.
253
247
  *
254
- * The one class that IS separable structurally is the two-token SINGLE name — "New York", "Fort Worth", "San Francisco"
255
- * — because those carry a toponymic head particle. That guard is what keeps the fork off the common case.
248
+ * Both readings appear in the marker, neither wins and the resolver keeps its existing answer.
249
+ *
250
+ * The structurally separable class is a two-token single name ("New York", "Fort Worth")
251
+ * with a toponymic head particle.
252
+ * That guard keeps the fork off the common case.
256
253
  */
257
254
  export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike): number {
258
255
  const words = bareNameWords(input, shape)
@@ -261,8 +258,8 @@ export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike
261
258
 
262
259
  const [first, second] = [words[0]!.toLowerCase(), words[1]!.toLowerCase()]
263
260
 
264
- // Reduplication — "Pago Pago", "Baden-Baden", "Walla Walla", "Bora Bora". Nobody travels from a place to itself,
265
- // so a repeated token is a universal single-name signal and needs no lexicon at all.
261
+ // Reduplication — "Pago Pago", "Baden-Baden", "Walla Walla" — is a universal single-name
262
+ // signal that needs no lexicon, because a route from a place to itself is invalid.
266
263
  if (first === second) return 0
267
264
 
268
265
  if (TOPONYM_HEAD_PARTICLES.has(first) || TOPONYM_HEAD_PARTICLES.has(second)) return 0
@@ -275,16 +272,16 @@ export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike
275
272
  /**
276
273
  * `near_me` rule: a subject plus a deictic locator, with no anchor.
277
274
  *
278
- * Requires a non-empty subject before the locator, so a bare "near me" stays with the `landmark` leaders rule rather
279
- * than claiming to be a category search with a missing focus point.
275
+ * A non-empty subject is required, so a bare "near me" stays with the `landmark` leaders rule.
280
276
  */
281
277
  export function scoreNearMe(input: NormalizedInputLite, _shape: QueryShapeLike): number {
282
278
  const lowercased = input.normalized.trim().toLowerCase()
283
279
 
284
280
  if (!hasDeicticTail(lowercased)) return 0
285
281
 
286
- // The subject is everything before the locator. `hasDeicticTail` already anchored the match to the end, so the
287
- // first match index is where the subject stops.
282
+ // The subject is everything before the locator.
283
+ // `hasDeicticTail` already anchored the match to the end, so the first match
284
+ // index marks where the subject stops.
288
285
  const match = DEICTIC_LOCATOR_TAIL.exec(lowercased) ?? DEICTIC_ADVERB_TAIL.exec(lowercased)
289
286
 
290
287
  if (!match) return 0
@@ -293,8 +290,10 @@ export function scoreNearMe(input: NormalizedInputLite, _shape: QueryShapeLike):
293
290
  }
294
291
 
295
292
  /**
296
- * The subject of a `near_me` query — the category or thing the asker wants, with the locator stripped. Empty string
297
- * when the rule would not have fired. Used to build the marker's evidence, never to route.
293
+ * The subject of a `near_me` query is the requested category or thing with the locator stripped.
294
+ *
295
+ * It is empty when the rule would not have fired.
296
+ * The subject builds marker evidence and never a route.
298
297
  */
299
298
  export function nearMeSubject(input: NormalizedInputLite): string {
300
299
  const trimmed = input.normalized.trim()