@mailwoman/kind-classifier 10.0.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/lib/classify.ts +33 -34
- package/lib/index.ts +1 -8
- package/lib/intent/markers.ts +28 -25
- package/lib/intent/rules.ts +82 -83
- package/lib/poi.ts +96 -84
- package/lib/rules.ts +141 -79
- package/out/classify.d.ts +18 -20
- package/out/classify.d.ts.map +1 -1
- package/out/classify.js +28 -31
- package/out/classify.js.map +1 -1
- package/out/index.d.ts +1 -8
- package/out/index.d.ts.map +1 -1
- package/out/index.js +1 -8
- package/out/index.js.map +1 -1
- package/out/intent/markers.d.ts +17 -18
- package/out/intent/markers.d.ts.map +1 -1
- package/out/intent/markers.js +24 -23
- package/out/intent/markers.js.map +1 -1
- package/out/intent/rules.d.ts +31 -35
- package/out/intent/rules.d.ts.map +1 -1
- package/out/intent/rules.js +82 -83
- package/out/intent/rules.js.map +1 -1
- package/out/poi.d.ts +64 -56
- package/out/poi.d.ts.map +1 -1
- package/out/poi.js +60 -50
- package/out/poi.js.map +1 -1
- package/out/rules.d.ts +23 -40
- package/out/rules.d.ts.map +1 -1
- package/out/rules.js +117 -80
- package/out/rules.js.map +1 -1
- package/package.json +15 -50
package/README.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
# @mailwoman/kind-classifier
|
|
2
2
|
|
|
3
|
-
**
|
|
3
|
+
This package is **stage 2.5 of the Mailwoman runtime pipeline**, which classifies
|
|
4
|
+
the query kind.
|
|
4
5
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
isn't actionable.
|
|
6
|
+
It assigns an input to one of seven `QueryKind`s by combining rule-based scorers
|
|
7
|
+
over the `QueryShape` output. It returns alternatives alongside the top pick, so
|
|
8
|
+
the coordinator can fall back when the winning kind is not actionable.
|
|
9
9
|
|
|
10
10
|
```ts
|
|
11
11
|
import { classifyKind } from "@mailwoman/kind-classifier"
|
|
@@ -51,18 +51,19 @@ locale-hint → kind-classifier → phrase-grouper → classifier → ...
|
|
|
51
51
|
|
|
52
52
|
## Design
|
|
53
53
|
|
|
54
|
-
- **Pure functions
|
|
54
|
+
- **Pure functions without ML.** Version 1 is rule-based, and a trained
|
|
55
|
+
classifier is deferred.
|
|
55
56
|
- **Returns alternatives.** The coordinator might skip a `locality_only` parse
|
|
56
|
-
and fall back to a `vague` handler
|
|
57
|
+
and fall back to a `vague` handler, which the alternatives list makes possible.
|
|
57
58
|
- **Consumes `QueryShape` + `LocaleHint`** from the two preceding stages.
|
|
58
59
|
|
|
59
60
|
## Related
|
|
60
61
|
|
|
61
|
-
- [`@mailwoman/query-shape`](../query-shape)
|
|
62
|
-
- [`@mailwoman/locale-hint`](../locale-hint)
|
|
63
|
-
- [`@mailwoman/phrase-grouper`](../phrase-grouper)
|
|
64
|
-
- [Staged Pipeline
|
|
62
|
+
- [`@mailwoman/query-shape`](../query-shape): supplies structural data to this stage.
|
|
63
|
+
- [`@mailwoman/locale-hint`](../locale-hint): supplies locale context.
|
|
64
|
+
- [`@mailwoman/phrase-grouper`](../phrase-grouper): stage 2.7, next in the pipeline.
|
|
65
|
+
- [Staged Pipeline Interface](https://github.com/sister-software/mailwoman/blob/main/docs/records/plan/reference/STAGES.mdx)
|
|
65
66
|
|
|
66
67
|
## License
|
|
67
68
|
|
|
68
|
-
[AGPL-3.0-only](https://www.gnu.org/licenses/
|
|
69
|
+
[AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
|
package/lib/classify.ts
CHANGED
|
@@ -3,17 +3,13 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* `classifyKind` is the entry point for Stage 2.5 (kind classification), composing the per-kind rules
|
|
7
|
+
* from `rules.ts` and `intent-rules.ts` and returning alternatives sorted by confidence.
|
|
7
8
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* Per the project's "possibilities not constraints" principle, every kind that fires above 0
|
|
13
|
-
* surfaces in `alternatives` — the caller decides whether to act on the top kind only or consider
|
|
14
|
-
* runner-ups. The ROAD_TO_V9 §4 intent vocabulary leans on that: `bare_toponym` and `route_pair`
|
|
15
|
-
* are scored below their structural incumbent precisely so they land in `alternatives`, where they
|
|
16
|
-
* inform the markers without moving the routing decision.
|
|
9
|
+
* Per the project's "possibilities not constraints" principle, every kind that fires above 0 surfaces
|
|
10
|
+
* in `alternatives` and the caller decides whether to act on the top kind only: `bare_toponym` and
|
|
11
|
+
* `route_pair` are scored below their structural incumbent precisely so they land in `alternatives`,
|
|
12
|
+
* where they inform the markers without moving the routing decision.
|
|
17
13
|
*/
|
|
18
14
|
|
|
19
15
|
import type { LocaleHint, QueryIntentMarker, QueryKind, QueryKindResult } from "@mailwoman/core/pipeline"
|
|
@@ -45,9 +41,9 @@ const SCORERS: ReadonlyArray<KindScorer> = [
|
|
|
45
41
|
{ kind: "postcode_only", score: scorePostcodeOnly },
|
|
46
42
|
{ kind: "locality_only", score: scoreLocalityOnly },
|
|
47
43
|
{ kind: "structured_address", score: scoreStructuredAddress },
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
//
|
|
44
|
+
// Intent is vocabulary rather than a stage.
|
|
45
|
+
// The sort below decides, so `bare_toponym` and `route_pair` sit here cosmetically
|
|
46
|
+
// and are scored under `locality_only` on purpose (see `intent-rules.ts`).
|
|
51
47
|
{ kind: "bare_toponym", score: scoreBareToponym },
|
|
52
48
|
{ kind: "route_pair", score: scoreRoutePair },
|
|
53
49
|
{ kind: "near_me", score: scoreNearMe },
|
|
@@ -55,8 +51,8 @@ const SCORERS: ReadonlyArray<KindScorer> = [
|
|
|
55
51
|
]
|
|
56
52
|
|
|
57
53
|
/**
|
|
58
|
-
* Rank a scored list
|
|
59
|
-
* drift in how they break ties or build `alternatives`.
|
|
54
|
+
* Rank a scored list into a verdict, shared by the lexicon-free and lexicon-wired paths
|
|
55
|
+
* so the two cannot drift in how they break ties or build `alternatives`.
|
|
60
56
|
*/
|
|
61
57
|
function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKindResult {
|
|
62
58
|
scored.sort((a, b) => b.confidence - a.confidence)
|
|
@@ -71,14 +67,14 @@ function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKind
|
|
|
71
67
|
}
|
|
72
68
|
|
|
73
69
|
/**
|
|
74
|
-
* Every kind whose verdict
|
|
75
|
-
*
|
|
70
|
+
* Every kind whose verdict includes `intentMarkers`, checked before the marker builder
|
|
71
|
+
* so the hot path pays one set membership test per kind.
|
|
76
72
|
*/
|
|
77
|
-
const
|
|
73
|
+
const MARKER_KINDS: ReadonlySet<QueryKind> = new Set<QueryKind>(["route_pair", "near_me", "poi_category"])
|
|
78
74
|
|
|
79
75
|
/**
|
|
80
|
-
* Attach markers to a verdict
|
|
81
|
-
*
|
|
76
|
+
* Attach markers to a verdict or return it untouched, separate from {@link rank}
|
|
77
|
+
* because the lexicon-wired path merges `poi_query`/`poi_category` in first.
|
|
82
78
|
*/
|
|
83
79
|
function withIntentMarkers(
|
|
84
80
|
verdict: QueryKindResult,
|
|
@@ -88,7 +84,7 @@ function withIntentMarkers(
|
|
|
88
84
|
): QueryKindResult {
|
|
89
85
|
const kinds = [{ kind: verdict.kind, confidence: verdict.confidence }, ...verdict.alternatives]
|
|
90
86
|
|
|
91
|
-
if (!kinds.some((k) =>
|
|
87
|
+
if (!kinds.some((k) => MARKER_KINDS.has(k.kind))) return verdict
|
|
92
88
|
|
|
93
89
|
const intentMarkers: QueryIntentMarker[] = deriveIntentMarkers(kinds, { input, poiLexicon, locale })
|
|
94
90
|
|
|
@@ -98,8 +94,8 @@ function withIntentMarkers(
|
|
|
98
94
|
}
|
|
99
95
|
|
|
100
96
|
/**
|
|
101
|
-
* Classify the query shape into a `QueryKind
|
|
102
|
-
* shape)` pair.
|
|
97
|
+
* Classify the query shape into a `QueryKind`, synchronously and purely,
|
|
98
|
+
* producing the same result for the same `(input, shape)` pair.
|
|
103
99
|
*/
|
|
104
100
|
export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLike): QueryKindResult {
|
|
105
101
|
const scored = SCORERS.map((s) => ({ kind: s.kind, confidence: s.score(input, shape) })).filter(
|
|
@@ -110,9 +106,8 @@ export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLi
|
|
|
110
106
|
}
|
|
111
107
|
|
|
112
108
|
/**
|
|
113
|
-
* Async variant matching the runtime
|
|
114
|
-
*
|
|
115
|
-
* The locale parameter is accepted for future locale-aware rules (Japanese honorifics, etc.) but not currently used.
|
|
109
|
+
* Async variant matching the runtime pipeline's `classifyKind` interface;
|
|
110
|
+
* `_locale` is accepted for future locale-aware rules but currently unused.
|
|
116
111
|
*/
|
|
117
112
|
export async function classifyKind(
|
|
118
113
|
input: NormalizedInputLite,
|
|
@@ -127,16 +122,19 @@ export async function classifyKind(
|
|
|
127
122
|
*/
|
|
128
123
|
export interface KindClassifierOpts {
|
|
129
124
|
/**
|
|
130
|
-
* POI phrase lexicon (spec §3.1)
|
|
131
|
-
*
|
|
132
|
-
*
|
|
125
|
+
* POI phrase lexicon (spec §3.1); when present the `poi_query`
|
|
126
|
+
* and `poi_category` scorers join the rule set.
|
|
127
|
+
*
|
|
128
|
+
* They are injected rather than imported so this package stays dictionary-free.
|
|
129
|
+
* When absent, the returned classifier behaves identically to {@link classifyKind}.
|
|
133
130
|
*/
|
|
134
131
|
poiLexicon?: POIPhraseLookup
|
|
135
132
|
}
|
|
136
133
|
|
|
137
134
|
/**
|
|
138
|
-
* Build a kind classifier
|
|
139
|
-
* additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4)
|
|
135
|
+
* Build a kind classifier that is exactly {@link classifyKind} without options,
|
|
136
|
+
* or that additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4)
|
|
137
|
+
* and merges them into the ranked result when given a `poiLexicon`.
|
|
140
138
|
*/
|
|
141
139
|
export function createKindClassifier(
|
|
142
140
|
opts: KindClassifierOpts = {}
|
|
@@ -153,9 +151,10 @@ export function createKindClassifier(
|
|
|
153
151
|
|
|
154
152
|
if (poiConfidence <= 0 && categoryConfidence <= 0) return base
|
|
155
153
|
|
|
156
|
-
// Re-
|
|
157
|
-
//
|
|
158
|
-
//
|
|
154
|
+
// Re-ranking over the union rather than special-casing whether POI beat the
|
|
155
|
+
// base preserves the base's own alternatives.
|
|
156
|
+
// Those keep `bare_toponym`/`route_pair` visible to the marker builder even
|
|
157
|
+
// when a POI kind takes the top slot.
|
|
159
158
|
const merged: Array<{ kind: QueryKind; confidence: number }> = [
|
|
160
159
|
{ kind: base.kind, confidence: base.confidence },
|
|
161
160
|
...base.alternatives,
|
package/lib/index.ts
CHANGED
|
@@ -3,14 +3,7 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* Categorize inputs into one of eight `QueryKind`s by composing rule-based scorers over the
|
|
9
|
-
* QueryShape sub-system's output. Pure functions, no ML, no place-name dictionaries. Returns
|
|
10
|
-
* possibilities (alternatives) alongside the top pick so the coordinator can fall back when the
|
|
11
|
-
* winning kind isn't actionable.
|
|
12
|
-
*
|
|
13
|
-
* See `docs/engineering/reference/STAGES.md` § Stage 2.5 for the contract.
|
|
6
|
+
* Stage 2.5 classifier: compose rule-based scorers over QueryShape and return a top kind with alternatives.
|
|
14
7
|
*/
|
|
15
8
|
|
|
16
9
|
export { classifyKind, classifyKindSync, createKindClassifier } from "#classify"
|
package/lib/intent/markers.ts
CHANGED
|
@@ -3,16 +3,15 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
6
|
+
* Marker derivation for the ROAD_TO_V9 §4 intent vocabulary. This code is pure and synchronous. It is the only place the
|
|
7
|
+
* classifier turns a fired rule into something a caller reads.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
* is false 89.1% of the time (the measured table behind `DECISIVE_MARGIN_LOG10`).
|
|
9
|
+
* Three of the four intent kinds can raise their marker here from the string by itself. The fourth,
|
|
10
|
+
* Stage 2.5 cannot raise `bare_toponym`'s `declared_ambiguity` from the string by itself.
|
|
11
|
+
* Its trigger is the dominance margin of the resolved candidate list. `mailwoman/query-intent.ts` raises it
|
|
12
|
+
* after the resolve against `DECISIVE_MARGIN_LOG10`. This module therefore never emits
|
|
13
|
+
* `declared_ambiguity`, since a marker asserting ambiguity from the string by itself would declare every
|
|
14
|
+
* bare city name ambiguous.
|
|
16
15
|
*/
|
|
17
16
|
|
|
18
17
|
import type { QueryIntentMarker, QueryKind } from "@mailwoman/core/pipeline"
|
|
@@ -27,22 +26,22 @@ import { matchPOICategory, type POIPhraseLookup } from "#poi"
|
|
|
27
26
|
export interface IntentMarkerContext {
|
|
28
27
|
input: NormalizedInputLite
|
|
29
28
|
/**
|
|
30
|
-
* The injected POI lexicon
|
|
31
|
-
*
|
|
29
|
+
* The injected POI lexicon when one was wired.
|
|
30
|
+
*
|
|
31
|
+
* Absent means no `poi_category` marker can be built, consistent with the
|
|
32
|
+
* kind not firing without it either.
|
|
32
33
|
*/
|
|
33
34
|
poiLexicon?: POIPhraseLookup
|
|
34
35
|
locale?: string
|
|
35
36
|
}
|
|
36
37
|
|
|
37
38
|
/**
|
|
38
|
-
* Build the advisories for one classified query
|
|
39
|
+
* Build the advisories for one classified query, taking the full verdict — top plus alternatives —
|
|
40
|
+
* because two of the four intent kinds live in `alternatives` by design (see `intent-rules.ts`).
|
|
39
41
|
*
|
|
40
|
-
* `
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
* Returns `[]` when no intent kind fired. Callers surface that empty array rather than dropping the field: an empty
|
|
45
|
-
* array is the classifier stating it looked.
|
|
42
|
+
* @returns `[]` when no intent kind fired.
|
|
43
|
+
* Callers surface that empty array rather than dropping the field, because an
|
|
44
|
+
* empty array is the classifier stating it looked.
|
|
46
45
|
*/
|
|
47
46
|
export function deriveIntentMarkers(
|
|
48
47
|
kinds: ReadonlyArray<{ kind: QueryKind; confidence: number }>,
|
|
@@ -52,8 +51,8 @@ export function deriveIntentMarkers(
|
|
|
52
51
|
const markers: QueryIntentMarker[] = []
|
|
53
52
|
|
|
54
53
|
if (fired.has("route_pair")) {
|
|
55
|
-
// Whitespace-only split
|
|
56
|
-
//
|
|
54
|
+
// Whitespace-only split rather than `wordsOf`, because `route_pair` inputs are
|
|
55
|
+
// comma-free by construction and the tokens are re-joined verbatim into the message.
|
|
57
56
|
const tokens = ctx.input.normalized.trim().split(/\s+/)
|
|
58
57
|
|
|
59
58
|
markers.push({
|
|
@@ -64,8 +63,11 @@ export function deriveIntentMarkers(
|
|
|
64
63
|
evidence: {
|
|
65
64
|
tokens,
|
|
66
65
|
/**
|
|
67
|
-
* Both readings,
|
|
68
|
-
*
|
|
66
|
+
* Both readings, listed.
|
|
67
|
+
*
|
|
68
|
+
* The order is stable: pair first, then the admin reading.
|
|
69
|
+
* A consumer can index it.
|
|
70
|
+
* This order does not rank the entries.
|
|
69
71
|
*/
|
|
70
72
|
interpretations: ["two_toponyms", "locality_with_admin_context"],
|
|
71
73
|
},
|
|
@@ -83,9 +85,10 @@ export function deriveIntentMarkers(
|
|
|
83
85
|
evidence: {
|
|
84
86
|
subject,
|
|
85
87
|
/**
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
* is the
|
|
88
|
+
* The plug point is documented but not wired.
|
|
89
|
+
*
|
|
90
|
+
* `photon/` is the eventual consumer.
|
|
91
|
+
* Its `/api` already accepts `lat`/`lon` location-bias params.
|
|
89
92
|
*/
|
|
90
93
|
focusParameter: "photon:lat/lon",
|
|
91
94
|
},
|
package/lib/intent/rules.ts
CHANGED
|
@@ -3,63 +3,59 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* already does it.
|
|
6
|
+
* ROAD_TO_V9 §4 — the query-intent rules, with the same `(input, shape) => number` interface as
|
|
7
|
+
* `rules.ts` and the same bitter-lesson invariant: universal structural patterns and bounded linguistic
|
|
8
|
+
* categories only, never a place-name dictionary, with the POI synonym table injected exactly as
|
|
9
|
+
* `poi.ts` does it.
|
|
11
10
|
*
|
|
12
|
-
*
|
|
11
|
+
* `bare_toponym` and `route_pair` score below the structural kind that already owns their population
|
|
12
|
+
* (`locality_only`, 0.85), so they surface in `QueryKindResult.alternatives` and never as the top kind.
|
|
13
|
+
* The top kind is the only thing the coordinator routes on, so pinning it is what makes these additions
|
|
14
|
+
* answer-neutral on the bare-city-name register. Their intent travels on the marker.
|
|
13
15
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* never as the top kind. That is not timidity — it is the D-rule discharge. The top kind is the
|
|
17
|
-
* only thing the coordinator routes on (`deriveInputMode`, `canShortCircuit`, the POI branch), so
|
|
18
|
-
* pinning it is what makes these additions provably answer-neutral on the bare-city-name register,
|
|
19
|
-
* which is the single largest population in map search. The intent they carry travels on the
|
|
20
|
-
* marker instead, where it is advisory by construction.
|
|
21
|
-
*
|
|
22
|
-
* `near_me` DOES win its top slot (0.91), because there is no incumbent worth preserving: a query
|
|
23
|
-
* ending "near me" is not a locality and answering it as one is the bug.
|
|
16
|
+
* `near_me` does win its top slot (0.91), because there is no incumbent worth preserving: a query
|
|
17
|
+
* ending "near me" is not a locality.
|
|
24
18
|
*/
|
|
25
19
|
|
|
26
20
|
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
|
|
27
21
|
|
|
28
|
-
import { isDisqualifyingStreetSuffix, MAX_LOCALITY_ONLY_LENGTH, wordsOf } from "#rules"
|
|
22
|
+
import { carriesLetter, isDisqualifyingStreetSuffix, MAX_LOCALITY_ONLY_LENGTH, wordsOf } from "#rules"
|
|
29
23
|
/**
|
|
30
|
-
*
|
|
31
|
-
* accident can flip the top slot
|
|
24
|
+
* Both refinement kinds sit a whole confidence step below `locality_only`'s 0.85
|
|
25
|
+
* so no float-comparison accident can flip the top slot.
|
|
32
26
|
*/
|
|
33
27
|
const BARE_TOPONYM_CONFIDENCE = 0.84
|
|
34
28
|
|
|
35
29
|
/**
|
|
36
|
-
* Lower still
|
|
37
|
-
*
|
|
30
|
+
* Lower still for a second reason beyond the ranking discipline.
|
|
31
|
+
*
|
|
32
|
+
* A route pair is a hypothesis whose competing reading (locality + region) is more common in this corpus.
|
|
38
33
|
*/
|
|
39
34
|
const ROUTE_PAIR_CONFIDENCE = 0.55
|
|
40
35
|
|
|
41
36
|
/**
|
|
42
|
-
* Above `landmark`'s venue ceiling (0.88) and above `poi_query`'s anchored band (0.90),
|
|
43
|
-
* stronger signal than either shape heuristic:
|
|
44
|
-
* string.
|
|
37
|
+
* Above `landmark`'s venue ceiling (0.88) and above `poi_query`'s anchored band (0.90),
|
|
38
|
+
* because a deictic tail is a stronger signal than either shape heuristic:
|
|
39
|
+
* no other rule explains why "me" ends the string.
|
|
45
40
|
*/
|
|
46
41
|
const NEAR_ME_CONFIDENCE = 0.91
|
|
47
42
|
|
|
48
43
|
/**
|
|
49
|
-
* Word ceiling for a single bare toponym
|
|
50
|
-
* upon Tyne", "Sault Sainte Marie", "Las Palmas de Gran Canaria")
|
|
44
|
+
* Word ceiling for a single bare toponym: four covers the long tail that exists as one
|
|
45
|
+
* place name ("Newcastle upon Tyne", "Sault Sainte Marie", "Las Palmas de Gran Canaria"),
|
|
46
|
+
* and past it the input contains more than a name.
|
|
51
47
|
*/
|
|
52
48
|
const MAX_BARE_TOPONYM_WORDS = 4
|
|
53
49
|
|
|
54
50
|
/**
|
|
55
|
-
* Toponymic
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* `route_pair` off "New York", "San Francisco", "Fort Worth" and their kin. It is a closed morphological class, not a
|
|
60
|
-
* gazetteer — growing it with actual place names is the wrong move, and the pressure for that belongs on the resolver.
|
|
51
|
+
* Toponymic head particles — the bounded linguistic category that makes a multi-token
|
|
52
|
+
* string one place name, with the same boundary as `@mailwoman/phrase-grouper`'s
|
|
53
|
+
* `PLACE_NAME_PARTICLES` (which covers the infix glue `de`, `am`, `aan den`) and the one
|
|
54
|
+
* job of keeping `route_pair` off "New York", "San Francisco", "Fort Worth" and their kin.
|
|
61
55
|
*
|
|
62
|
-
*
|
|
56
|
+
* It is a closed morphological class rather than a gazetteer, so growing it
|
|
57
|
+
* with actual place names is the wrong move.
|
|
58
|
+
* It is case-folded on read, because "new york" is the same query.
|
|
63
59
|
*/
|
|
64
60
|
const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
|
|
65
61
|
// English
|
|
@@ -122,9 +118,10 @@ const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
|
|
|
122
118
|
"sint",
|
|
123
119
|
// Definite article as a head — "The Valley" (Anguilla), "The Hague", "The Bottom".
|
|
124
120
|
"the",
|
|
125
|
-
// Generic toponymic heads outside the Latin/Germanic families,
|
|
126
|
-
//
|
|
127
|
-
//
|
|
121
|
+
// Generic toponymic heads outside the Latin/Germanic families, each a common
|
|
122
|
+
// noun in its own language — Semitic "tel" (mound), Malay "kuala" (confluence),
|
|
123
|
+
// Khmer "phnom" (hill) — that heads a place name the way "mount" does.
|
|
124
|
+
// `mailwoman/test/kind-intent-invariance.test.ts` covers them.
|
|
128
125
|
"tel",
|
|
129
126
|
"kuala",
|
|
130
127
|
"phnom",
|
|
@@ -135,11 +132,9 @@ const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
|
|
|
135
132
|
])
|
|
136
133
|
|
|
137
134
|
/**
|
|
138
|
-
* Generic toponymic
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
* Measured additions, same as the heads above: `city`, `town` and `valley` each came off a real corpus row that was
|
|
142
|
-
* forking wrongly.
|
|
135
|
+
* Generic toponymic tail nouns — the other half of the same bounded morphological
|
|
136
|
+
* class, so a place name whose last token is a settlement/landform generic
|
|
137
|
+
* ("Belize City", "George Town", "Palm Springs") is one name rather than two.
|
|
143
138
|
*/
|
|
144
139
|
const TOPONYM_TAIL_NOUNS: ReadonlySet<string> = new Set([
|
|
145
140
|
"city",
|
|
@@ -167,53 +162,54 @@ const TOPONYM_TAIL_NOUNS: ReadonlySet<string> = new Set([
|
|
|
167
162
|
])
|
|
168
163
|
|
|
169
164
|
/**
|
|
170
|
-
* Deictic locator tails — "near me", "nearby", "around here", "in my area"
|
|
165
|
+
* Deictic locator tails — "near me", "nearby", "around here", "in my area" —
|
|
166
|
+
* the bounded class `preposition + a reference to the asker`, where `me`, `here`,
|
|
167
|
+
* `my <noun>` are function words rather than places.
|
|
171
168
|
*
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
* the kind is that the query names no anchor, so anything AFTER the locator is an anchor and disqualifies it.
|
|
169
|
+
* Anchored to the end of the string (`$`) on purpose: the query names no anchor,
|
|
170
|
+
* so anything after the locator is an anchor and disqualifies it.
|
|
175
171
|
*
|
|
176
|
-
* Linear by construction: every alternative begins with a required literal
|
|
177
|
-
* runs
|
|
178
|
-
* `js/polynomial-redos` shape
|
|
172
|
+
* Linear by construction: every alternative begins with a required literal and the only
|
|
173
|
+
* quantifiers are bounded `\s+` runs between two required literals or trailing before `$`,
|
|
174
|
+
* with no unbounded-whitespace-then-literal prefix (the `js/polynomial-redos` shape).
|
|
175
|
+
* `ANCHOR_SEPARATOR` in `poi.ts` uses the same analysis.
|
|
179
176
|
*/
|
|
180
177
|
const DEICTIC_LOCATOR_TAIL =
|
|
181
178
|
/\b(?:near|close\s+to|next\s+to|around|by|closest\s+to|nearest\s+to)\s+(?:me|us|here|my\s+(?:location|position|area|place|house|home))\s*$/
|
|
182
179
|
|
|
183
180
|
/**
|
|
184
|
-
* The adverbial half of the same class
|
|
181
|
+
* The adverbial half of the same class, where the deixis is baked into the word
|
|
182
|
+
* rather than introduced by a preposition.
|
|
185
183
|
*/
|
|
186
184
|
const DEICTIC_ADVERB_TAIL =
|
|
187
185
|
/\b(?:nearby|near\s?by|close\s+by|around\s+here|in\s+my\s+(?:area|neighborhood|neighbourhood))\s*$/
|
|
188
186
|
|
|
189
|
-
/**
|
|
190
|
-
* True when the input carries a deictic locator tail in EITHER form.
|
|
191
|
-
*/
|
|
192
187
|
function hasDeicticTail(lowercased: string): boolean {
|
|
193
188
|
return DEICTIC_LOCATOR_TAIL.test(lowercased) || DEICTIC_ADVERB_TAIL.test(lowercased)
|
|
194
189
|
}
|
|
195
190
|
|
|
196
191
|
/**
|
|
197
|
-
* The conditions `bare_toponym` and `route_pair` share: no address grammar of any kind,
|
|
192
|
+
* The conditions `bare_toponym` and `route_pair` share: no address grammar of any kind,
|
|
193
|
+
* one segment, alpha throughout.
|
|
198
194
|
*
|
|
199
|
-
*
|
|
200
|
-
*
|
|
201
|
-
* `locality_only` and can never fire where `locality_only` did not
|
|
202
|
-
* the reason the ranking discipline above is enough to keep the top kind pinned.
|
|
195
|
+
* @returns the word list when the input clears them, `null` when it does not.
|
|
196
|
+
* The conditions are deliberately a superset of `scoreLocalityOnly`'s, so `bare_toponym` is
|
|
197
|
+
* a strict refinement of `locality_only` and can never fire where `locality_only` did not.
|
|
203
198
|
*/
|
|
204
199
|
function bareNameWords(input: NormalizedInputLite, shape: QueryShapeLike): string[] | null {
|
|
205
200
|
const text = input.normalized.trim()
|
|
206
201
|
|
|
207
202
|
if (!text || text.length > MAX_LOCALITY_ONLY_LENGTH) return null
|
|
208
203
|
|
|
209
|
-
// A recognized postcode/known format
|
|
204
|
+
// A recognized postcode/known format is address grammar, so no bare toponym survives it.
|
|
210
205
|
if (shape.knownFormats.length) return null
|
|
211
206
|
|
|
212
|
-
// `alpha` excludes every house number and
|
|
213
|
-
// "no address grammar",
|
|
214
|
-
|
|
207
|
+
// `alpha` excludes every house number and postcode by construction, the cheapest statement of
|
|
208
|
+
// "no address grammar"; it is silent about whether a name is present, so the letter test stands
|
|
209
|
+
// beside it because `foldInputClass` answers `alpha` for input carrying no classified token.
|
|
210
|
+
if (shape.characterClass !== "alpha" || !carriesLetter(text)) return null
|
|
215
211
|
|
|
216
|
-
// A comma is the admin-context marker ("Paris, FR")
|
|
212
|
+
// A comma is the admin-context marker ("Paris, FR"), so one segment or the name is not bare.
|
|
217
213
|
if ((shape.segments?.length ?? 1) !== 1) return null
|
|
218
214
|
|
|
219
215
|
const lowercased = text.toLowerCase()
|
|
@@ -232,27 +228,28 @@ function bareNameWords(input: NormalizedInputLite, shape: QueryShapeLike): strin
|
|
|
232
228
|
}
|
|
233
229
|
|
|
234
230
|
/**
|
|
235
|
-
* `bare_toponym` rule: a single coherent place-name carrying no address grammar
|
|
236
|
-
*
|
|
237
|
-
*
|
|
238
|
-
* question, and `mailwoman/query-intent.ts` is where the answer's dominance margin decides whether the ambiguity gets
|
|
239
|
-
* declared.
|
|
231
|
+
* `bare_toponym` rule: a single coherent place-name carrying no address grammar, feeding the
|
|
232
|
+
* declared-ambiguity path without asserting which place — that is the resolver's question,
|
|
233
|
+
* decided by the answer's dominance margin in `mailwoman/query-intent.ts`.
|
|
240
234
|
*/
|
|
241
235
|
export function scoreBareToponym(input: NormalizedInputLite, shape: QueryShapeLike): number {
|
|
242
236
|
return bareNameWords(input, shape) ? BARE_TOPONYM_CONFIDENCE : 0
|
|
243
237
|
}
|
|
244
238
|
|
|
245
239
|
/**
|
|
246
|
-
* `route_pair` rule: exactly two toponym-shaped tokens with
|
|
240
|
+
* `route_pair` rule: exactly two toponym-shaped tokens with no token between them.
|
|
241
|
+
*
|
|
242
|
+
* The known confound is structural and unfixable here.
|
|
243
|
+
* "Paris London" and "Moscow Idaho" have the same string shape.
|
|
247
244
|
*
|
|
248
|
-
*
|
|
249
|
-
*
|
|
250
|
-
* not a structural one. The hard-slice board's 18 `comma_free` rows are that population, and they fire this rule. That
|
|
251
|
-
* is the reason ROAD_TO_V9 §4.3 specifies **classification + a declared fork, never a router**: both readings are named
|
|
252
|
-
* in the marker, neither wins, and the resolver keeps answering exactly as it did.
|
|
245
|
+
* A classifier needs gazetteer knowledge that Idaho is a region to distinguish them.
|
|
246
|
+
* ROAD_TO_V9 §4.3 therefore specifies classification plus a declared fork.
|
|
253
247
|
*
|
|
254
|
-
*
|
|
255
|
-
*
|
|
248
|
+
* Both readings appear in the marker, neither wins and the resolver keeps its existing answer.
|
|
249
|
+
*
|
|
250
|
+
* The structurally separable class is a two-token single name ("New York", "Fort Worth")
|
|
251
|
+
* with a toponymic head particle.
|
|
252
|
+
* That guard keeps the fork off the common case.
|
|
256
253
|
*/
|
|
257
254
|
export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike): number {
|
|
258
255
|
const words = bareNameWords(input, shape)
|
|
@@ -261,8 +258,8 @@ export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike
|
|
|
261
258
|
|
|
262
259
|
const [first, second] = [words[0]!.toLowerCase(), words[1]!.toLowerCase()]
|
|
263
260
|
|
|
264
|
-
// Reduplication — "Pago Pago", "Baden-Baden", "Walla Walla"
|
|
265
|
-
//
|
|
261
|
+
// Reduplication — "Pago Pago", "Baden-Baden", "Walla Walla" — is a universal single-name
|
|
262
|
+
// signal that needs no lexicon, because a route from a place to itself is invalid.
|
|
266
263
|
if (first === second) return 0
|
|
267
264
|
|
|
268
265
|
if (TOPONYM_HEAD_PARTICLES.has(first) || TOPONYM_HEAD_PARTICLES.has(second)) return 0
|
|
@@ -275,16 +272,16 @@ export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike
|
|
|
275
272
|
/**
|
|
276
273
|
* `near_me` rule: a subject plus a deictic locator, with no anchor.
|
|
277
274
|
*
|
|
278
|
-
*
|
|
279
|
-
* than claiming to be a category search with a missing focus point.
|
|
275
|
+
* A non-empty subject is required, so a bare "near me" stays with the `landmark` leaders rule.
|
|
280
276
|
*/
|
|
281
277
|
export function scoreNearMe(input: NormalizedInputLite, _shape: QueryShapeLike): number {
|
|
282
278
|
const lowercased = input.normalized.trim().toLowerCase()
|
|
283
279
|
|
|
284
280
|
if (!hasDeicticTail(lowercased)) return 0
|
|
285
281
|
|
|
286
|
-
// The subject is everything before the locator.
|
|
287
|
-
//
|
|
282
|
+
// The subject is everything before the locator.
|
|
283
|
+
// `hasDeicticTail` already anchored the match to the end, so the first match
|
|
284
|
+
// index marks where the subject stops.
|
|
288
285
|
const match = DEICTIC_LOCATOR_TAIL.exec(lowercased) ?? DEICTIC_ADVERB_TAIL.exec(lowercased)
|
|
289
286
|
|
|
290
287
|
if (!match) return 0
|
|
@@ -293,8 +290,10 @@ export function scoreNearMe(input: NormalizedInputLite, _shape: QueryShapeLike):
|
|
|
293
290
|
}
|
|
294
291
|
|
|
295
292
|
/**
|
|
296
|
-
* The subject of a `near_me` query
|
|
297
|
-
*
|
|
293
|
+
* The subject of a `near_me` query is the requested category or thing with the locator stripped.
|
|
294
|
+
*
|
|
295
|
+
* It is empty when the rule would not have fired.
|
|
296
|
+
* The subject builds marker evidence and never a route.
|
|
298
297
|
*/
|
|
299
298
|
export function nearMeSubject(input: NormalizedInputLite): string {
|
|
300
299
|
const trimmed = input.normalized.trim()
|