@mailwoman/kind-classifier 9.4.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/lib/classify.ts +35 -36
- package/lib/index.ts +4 -11
- package/lib/{intent-markers.ts → intent/markers.ts} +29 -26
- package/lib/intent/rules.ts +306 -0
- package/lib/poi.ts +96 -84
- package/lib/rules.ts +142 -105
- package/out/classify.d.ts +18 -20
- package/out/classify.d.ts.map +1 -1
- package/out/classify.js +30 -33
- package/out/classify.js.map +1 -1
- package/out/index.d.ts +4 -11
- package/out/index.d.ts.map +1 -1
- package/out/index.js +3 -10
- package/out/index.js.map +1 -1
- package/out/intent/markers.d.ts +45 -0
- package/out/intent/markers.d.ts.map +1 -0
- package/out/intent/markers.js +88 -0
- package/out/intent/markers.js.map +1 -0
- package/out/intent/rules.d.ts +55 -0
- package/out/intent/rules.d.ts.map +1 -0
- package/out/intent/rules.js +281 -0
- package/out/intent/rules.js.map +1 -0
- package/out/poi.d.ts +64 -56
- package/out/poi.d.ts.map +1 -1
- package/out/poi.js +60 -50
- package/out/poi.js.map +1 -1
- package/out/rules.d.ts +22 -44
- package/out/rules.d.ts.map +1 -1
- package/out/rules.js +118 -104
- package/out/rules.js.map +1 -1
- package/package.json +15 -50
- package/lib/intent-rules.ts +0 -307
- package/out/intent-markers.d.ts +0 -46
- package/out/intent-markers.d.ts.map +0 -1
- package/out/intent-markers.js +0 -87
- package/out/intent-markers.js.map +0 -1
- package/out/intent-rules.d.ts +0 -59
- package/out/intent-rules.d.ts.map +0 -1
- package/out/intent-rules.js +0 -282
- package/out/intent-rules.js.map +0 -1
package/README.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
# @mailwoman/kind-classifier
|
|
2
2
|
|
|
3
|
-
**
|
|
3
|
+
This package is **stage 2.5 of the Mailwoman runtime pipeline**, which classifies
|
|
4
|
+
the query kind.
|
|
4
5
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
isn't actionable.
|
|
6
|
+
It assigns an input to one of seven `QueryKind`s by combining rule-based scorers
|
|
7
|
+
over the `QueryShape` output. It returns alternatives alongside the top pick, so
|
|
8
|
+
the coordinator can fall back when the winning kind is not actionable.
|
|
9
9
|
|
|
10
10
|
```ts
|
|
11
11
|
import { classifyKind } from "@mailwoman/kind-classifier"
|
|
@@ -51,18 +51,19 @@ locale-hint → kind-classifier → phrase-grouper → classifier → ...
|
|
|
51
51
|
|
|
52
52
|
## Design
|
|
53
53
|
|
|
54
|
-
- **Pure functions
|
|
54
|
+
- **Pure functions without ML.** Version 1 is rule-based, and a trained
|
|
55
|
+
classifier is deferred.
|
|
55
56
|
- **Returns alternatives.** The coordinator might skip a `locality_only` parse
|
|
56
|
-
and fall back to a `vague` handler
|
|
57
|
+
and fall back to a `vague` handler, which the alternatives list makes possible.
|
|
57
58
|
- **Consumes `QueryShape` + `LocaleHint`** from the two preceding stages.
|
|
58
59
|
|
|
59
60
|
## Related
|
|
60
61
|
|
|
61
|
-
- [`@mailwoman/query-shape`](../query-shape)
|
|
62
|
-
- [`@mailwoman/locale-hint`](../locale-hint)
|
|
63
|
-
- [`@mailwoman/phrase-grouper`](../phrase-grouper)
|
|
64
|
-
- [Staged Pipeline
|
|
62
|
+
- [`@mailwoman/query-shape`](../query-shape): supplies structural data to this stage.
|
|
63
|
+
- [`@mailwoman/locale-hint`](../locale-hint): supplies locale context.
|
|
64
|
+
- [`@mailwoman/phrase-grouper`](../phrase-grouper): stage 2.7, next in the pipeline.
|
|
65
|
+
- [Staged Pipeline Interface](https://github.com/sister-software/mailwoman/blob/main/docs/records/plan/reference/STAGES.mdx)
|
|
65
66
|
|
|
66
67
|
## License
|
|
67
68
|
|
|
68
|
-
[AGPL-3.0-only](https://www.gnu.org/licenses/
|
|
69
|
+
[AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
|
package/lib/classify.ts
CHANGED
|
@@ -3,24 +3,20 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* `classifyKind` is the entry point for Stage 2.5 (kind classification), composing the per-kind rules
|
|
7
|
+
* from `rules.ts` and `intent-rules.ts` and returning alternatives sorted by confidence.
|
|
7
8
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* Per the project's "possibilities not constraints" principle, every kind that fires above 0
|
|
13
|
-
* surfaces in `alternatives` — the caller decides whether to act on the top kind only or consider
|
|
14
|
-
* runner-ups. The ROAD_TO_V9 §4 intent vocabulary leans on that: `bare_toponym` and `route_pair`
|
|
15
|
-
* are scored below their structural incumbent precisely so they land in `alternatives`, where they
|
|
16
|
-
* inform the markers without moving the routing decision.
|
|
9
|
+
* Per the project's "possibilities not constraints" principle, every kind that fires above 0 surfaces
|
|
10
|
+
* in `alternatives` and the caller decides whether to act on the top kind only: `bare_toponym` and
|
|
11
|
+
* `route_pair` are scored below their structural incumbent precisely so they land in `alternatives`,
|
|
12
|
+
* where they inform the markers without moving the routing decision.
|
|
17
13
|
*/
|
|
18
14
|
|
|
19
15
|
import type { LocaleHint, QueryIntentMarker, QueryKind, QueryKindResult } from "@mailwoman/core/pipeline"
|
|
20
16
|
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
|
|
21
17
|
|
|
22
|
-
import { deriveIntentMarkers } from "#intent
|
|
23
|
-
import { scoreBareToponym, scoreNearMe, scoreRoutePair } from "#intent
|
|
18
|
+
import { deriveIntentMarkers } from "#intent/markers"
|
|
19
|
+
import { scoreBareToponym, scoreNearMe, scoreRoutePair } from "#intent/rules"
|
|
24
20
|
import { createScorePOICategory, createScorePOIQuery, type POIPhraseLookup } from "#poi"
|
|
25
21
|
import {
|
|
26
22
|
scoreIntersection,
|
|
@@ -45,9 +41,9 @@ const SCORERS: ReadonlyArray<KindScorer> = [
|
|
|
45
41
|
{ kind: "postcode_only", score: scorePostcodeOnly },
|
|
46
42
|
{ kind: "locality_only", score: scoreLocalityOnly },
|
|
47
43
|
{ kind: "structured_address", score: scoreStructuredAddress },
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
//
|
|
44
|
+
// Intent is vocabulary rather than a stage.
|
|
45
|
+
// The sort below decides, so `bare_toponym` and `route_pair` sit here cosmetically
|
|
46
|
+
// and are scored under `locality_only` on purpose (see `intent-rules.ts`).
|
|
51
47
|
{ kind: "bare_toponym", score: scoreBareToponym },
|
|
52
48
|
{ kind: "route_pair", score: scoreRoutePair },
|
|
53
49
|
{ kind: "near_me", score: scoreNearMe },
|
|
@@ -55,8 +51,8 @@ const SCORERS: ReadonlyArray<KindScorer> = [
|
|
|
55
51
|
]
|
|
56
52
|
|
|
57
53
|
/**
|
|
58
|
-
* Rank a scored list
|
|
59
|
-
* drift in how they break ties or build `alternatives`.
|
|
54
|
+
* Rank a scored list into a verdict, shared by the lexicon-free and lexicon-wired paths
|
|
55
|
+
* so the two cannot drift in how they break ties or build `alternatives`.
|
|
60
56
|
*/
|
|
61
57
|
function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKindResult {
|
|
62
58
|
scored.sort((a, b) => b.confidence - a.confidence)
|
|
@@ -71,14 +67,14 @@ function rank(scored: Array<{ kind: QueryKind; confidence: number }>): QueryKind
|
|
|
71
67
|
}
|
|
72
68
|
|
|
73
69
|
/**
|
|
74
|
-
* Every kind whose verdict
|
|
75
|
-
*
|
|
70
|
+
* Every kind whose verdict includes `intentMarkers`, checked before the marker builder
|
|
71
|
+
* so the hot path pays one set membership test per kind.
|
|
76
72
|
*/
|
|
77
|
-
const
|
|
73
|
+
const MARKER_KINDS: ReadonlySet<QueryKind> = new Set<QueryKind>(["route_pair", "near_me", "poi_category"])
|
|
78
74
|
|
|
79
75
|
/**
|
|
80
|
-
* Attach markers to a verdict
|
|
81
|
-
*
|
|
76
|
+
* Attach markers to a verdict or return it untouched, separate from {@link rank}
|
|
77
|
+
* because the lexicon-wired path merges `poi_query`/`poi_category` in first.
|
|
82
78
|
*/
|
|
83
79
|
function withIntentMarkers(
|
|
84
80
|
verdict: QueryKindResult,
|
|
@@ -88,7 +84,7 @@ function withIntentMarkers(
|
|
|
88
84
|
): QueryKindResult {
|
|
89
85
|
const kinds = [{ kind: verdict.kind, confidence: verdict.confidence }, ...verdict.alternatives]
|
|
90
86
|
|
|
91
|
-
if (!kinds.some((k) =>
|
|
87
|
+
if (!kinds.some((k) => MARKER_KINDS.has(k.kind))) return verdict
|
|
92
88
|
|
|
93
89
|
const intentMarkers: QueryIntentMarker[] = deriveIntentMarkers(kinds, { input, poiLexicon, locale })
|
|
94
90
|
|
|
@@ -98,8 +94,8 @@ function withIntentMarkers(
|
|
|
98
94
|
}
|
|
99
95
|
|
|
100
96
|
/**
|
|
101
|
-
* Classify the query shape into a `QueryKind
|
|
102
|
-
* shape)` pair.
|
|
97
|
+
* Classify the query shape into a `QueryKind`, synchronously and purely,
|
|
98
|
+
* producing the same result for the same `(input, shape)` pair.
|
|
103
99
|
*/
|
|
104
100
|
export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLike): QueryKindResult {
|
|
105
101
|
const scored = SCORERS.map((s) => ({ kind: s.kind, confidence: s.score(input, shape) })).filter(
|
|
@@ -110,9 +106,8 @@ export function classifyKindSync(input: NormalizedInputLite, shape: QueryShapeLi
|
|
|
110
106
|
}
|
|
111
107
|
|
|
112
108
|
/**
|
|
113
|
-
* Async variant matching the runtime
|
|
114
|
-
*
|
|
115
|
-
* The locale parameter is accepted for future locale-aware rules (Japanese honorifics, etc.) but not currently used.
|
|
109
|
+
* Async variant matching the runtime pipeline's `classifyKind` interface;
|
|
110
|
+
* `_locale` is accepted for future locale-aware rules but currently unused.
|
|
116
111
|
*/
|
|
117
112
|
export async function classifyKind(
|
|
118
113
|
input: NormalizedInputLite,
|
|
@@ -127,16 +122,19 @@ export async function classifyKind(
|
|
|
127
122
|
*/
|
|
128
123
|
export interface KindClassifierOpts {
|
|
129
124
|
/**
|
|
130
|
-
* POI phrase lexicon (spec §3.1)
|
|
131
|
-
*
|
|
132
|
-
*
|
|
125
|
+
* POI phrase lexicon (spec §3.1); when present the `poi_query`
|
|
126
|
+
* and `poi_category` scorers join the rule set.
|
|
127
|
+
*
|
|
128
|
+
* They are injected rather than imported so this package stays dictionary-free.
|
|
129
|
+
* When absent, the returned classifier behaves identically to {@link classifyKind}.
|
|
133
130
|
*/
|
|
134
131
|
poiLexicon?: POIPhraseLookup
|
|
135
132
|
}
|
|
136
133
|
|
|
137
134
|
/**
|
|
138
|
-
* Build a kind classifier
|
|
139
|
-
* additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4)
|
|
135
|
+
* Build a kind classifier that is exactly {@link classifyKind} without options,
|
|
136
|
+
* or that additionally scores `poi_query` + `poi_category` (ROAD_TO_V9 §4.4)
|
|
137
|
+
* and merges them into the ranked result when given a `poiLexicon`.
|
|
140
138
|
*/
|
|
141
139
|
export function createKindClassifier(
|
|
142
140
|
opts: KindClassifierOpts = {}
|
|
@@ -153,9 +151,10 @@ export function createKindClassifier(
|
|
|
153
151
|
|
|
154
152
|
if (poiConfidence <= 0 && categoryConfidence <= 0) return base
|
|
155
153
|
|
|
156
|
-
// Re-
|
|
157
|
-
//
|
|
158
|
-
//
|
|
154
|
+
// Re-ranking over the union rather than special-casing whether POI beat the
|
|
155
|
+
// base preserves the base's own alternatives.
|
|
156
|
+
// Those keep `bare_toponym`/`route_pair` visible to the marker builder even
|
|
157
|
+
// when a POI kind takes the top slot.
|
|
159
158
|
const merged: Array<{ kind: QueryKind; confidence: number }> = [
|
|
160
159
|
{ kind: base.kind, confidence: base.confidence },
|
|
161
160
|
...base.alternatives,
|
package/lib/index.ts
CHANGED
|
@@ -3,21 +3,14 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* Categorize inputs into one of eight `QueryKind`s by composing rule-based scorers over the
|
|
9
|
-
* QueryShape sub-system's output. Pure functions, no ML, no place-name dictionaries. Returns
|
|
10
|
-
* possibilities (alternatives) alongside the top pick so the coordinator can fall back when the
|
|
11
|
-
* winning kind isn't actionable.
|
|
12
|
-
*
|
|
13
|
-
* See `docs/engineering/reference/STAGES.md` § Stage 2.5 for the contract.
|
|
6
|
+
* Stage 2.5 classifier: compose rule-based scorers over QueryShape and return a top kind with alternatives.
|
|
14
7
|
*/
|
|
15
8
|
|
|
16
9
|
export { classifyKind, classifyKindSync, createKindClassifier } from "#classify"
|
|
17
10
|
export type { KindClassifierOpts } from "#classify"
|
|
18
|
-
export { deriveIntentMarkers } from "#intent
|
|
19
|
-
export type { IntentMarkerContext } from "#intent
|
|
20
|
-
export { nearMeSubject, scoreBareToponym, scoreNearMe, scoreRoutePair } from "#intent
|
|
11
|
+
export { deriveIntentMarkers } from "#intent/markers"
|
|
12
|
+
export type { IntentMarkerContext } from "#intent/markers"
|
|
13
|
+
export { nearMeSubject, scoreBareToponym, scoreNearMe, scoreRoutePair } from "#intent/rules"
|
|
21
14
|
export { matchPOICategory, matchPOISubject } from "#poi"
|
|
22
15
|
export type { POIPhraseMatch, POIPhraseLookup, POIQuerySpan, POISpatialRelation, POISubjectMatch } from "#poi"
|
|
23
16
|
|
|
@@ -3,22 +3,21 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
6
|
+
* Marker derivation for the ROAD_TO_V9 §4 intent vocabulary. This code is pure and synchronous. It is the only place the
|
|
7
|
+
* classifier turns a fired rule into something a caller reads.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
* is false 89.1% of the time (the measured table behind `DECISIVE_MARGIN_LOG10`).
|
|
9
|
+
* Three of the four intent kinds can raise their marker here from the string by itself. The fourth,
|
|
10
|
+
* Stage 2.5 cannot raise `bare_toponym`'s `declared_ambiguity` from the string by itself.
|
|
11
|
+
* Its trigger is the dominance margin of the resolved candidate list. `mailwoman/query-intent.ts` raises it
|
|
12
|
+
* after the resolve against `DECISIVE_MARGIN_LOG10`. This module therefore never emits
|
|
13
|
+
* `declared_ambiguity`, since a marker asserting ambiguity from the string by itself would declare every
|
|
14
|
+
* bare city name ambiguous.
|
|
16
15
|
*/
|
|
17
16
|
|
|
18
17
|
import type { QueryIntentMarker, QueryKind } from "@mailwoman/core/pipeline"
|
|
19
18
|
import type { NormalizedInputLite } from "@mailwoman/query-shape"
|
|
20
19
|
|
|
21
|
-
import { nearMeSubject } from "#intent
|
|
20
|
+
import { nearMeSubject } from "#intent/rules"
|
|
22
21
|
import { matchPOICategory, type POIPhraseLookup } from "#poi"
|
|
23
22
|
|
|
24
23
|
/**
|
|
@@ -27,22 +26,22 @@ import { matchPOICategory, type POIPhraseLookup } from "#poi"
|
|
|
27
26
|
export interface IntentMarkerContext {
|
|
28
27
|
input: NormalizedInputLite
|
|
29
28
|
/**
|
|
30
|
-
* The injected POI lexicon
|
|
31
|
-
*
|
|
29
|
+
* The injected POI lexicon when one was wired.
|
|
30
|
+
*
|
|
31
|
+
* Absent means no `poi_category` marker can be built, consistent with the
|
|
32
|
+
* kind not firing without it either.
|
|
32
33
|
*/
|
|
33
34
|
poiLexicon?: POIPhraseLookup
|
|
34
35
|
locale?: string
|
|
35
36
|
}
|
|
36
37
|
|
|
37
38
|
/**
|
|
38
|
-
* Build the advisories for one classified query
|
|
39
|
+
* Build the advisories for one classified query, taking the full verdict — top plus alternatives —
|
|
40
|
+
* because two of the four intent kinds live in `alternatives` by design (see `intent-rules.ts`).
|
|
39
41
|
*
|
|
40
|
-
* `
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
* Returns `[]` when no intent kind fired. Callers surface that empty array rather than dropping the field: an empty
|
|
45
|
-
* array is the classifier stating it looked.
|
|
42
|
+
* @returns `[]` when no intent kind fired.
|
|
43
|
+
* Callers surface that empty array rather than dropping the field, because an
|
|
44
|
+
* empty array is the classifier stating it looked.
|
|
46
45
|
*/
|
|
47
46
|
export function deriveIntentMarkers(
|
|
48
47
|
kinds: ReadonlyArray<{ kind: QueryKind; confidence: number }>,
|
|
@@ -52,8 +51,8 @@ export function deriveIntentMarkers(
|
|
|
52
51
|
const markers: QueryIntentMarker[] = []
|
|
53
52
|
|
|
54
53
|
if (fired.has("route_pair")) {
|
|
55
|
-
// Whitespace-only split
|
|
56
|
-
//
|
|
54
|
+
// Whitespace-only split rather than `wordsOf`, because `route_pair` inputs are
|
|
55
|
+
// comma-free by construction and the tokens are re-joined verbatim into the message.
|
|
57
56
|
const tokens = ctx.input.normalized.trim().split(/\s+/)
|
|
58
57
|
|
|
59
58
|
markers.push({
|
|
@@ -64,8 +63,11 @@ export function deriveIntentMarkers(
|
|
|
64
63
|
evidence: {
|
|
65
64
|
tokens,
|
|
66
65
|
/**
|
|
67
|
-
* Both readings,
|
|
68
|
-
*
|
|
66
|
+
* Both readings, listed.
|
|
67
|
+
*
|
|
68
|
+
* The order is stable: pair first, then the admin reading.
|
|
69
|
+
* A consumer can index it.
|
|
70
|
+
* This order does not rank the entries.
|
|
69
71
|
*/
|
|
70
72
|
interpretations: ["two_toponyms", "locality_with_admin_context"],
|
|
71
73
|
},
|
|
@@ -83,9 +85,10 @@ export function deriveIntentMarkers(
|
|
|
83
85
|
evidence: {
|
|
84
86
|
subject,
|
|
85
87
|
/**
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
* is the
|
|
88
|
+
* The plug point is documented but not wired.
|
|
89
|
+
*
|
|
90
|
+
* `photon/` is the eventual consumer.
|
|
91
|
+
* Its `/api` already accepts `lat`/`lon` location-bias params.
|
|
89
92
|
*/
|
|
90
93
|
focusParameter: "photon:lat/lon",
|
|
91
94
|
},
|
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* ROAD_TO_V9 §4 — the query-intent rules, with the same `(input, shape) => number` interface as
|
|
7
|
+
* `rules.ts` and the same bitter-lesson invariant: universal structural patterns and bounded linguistic
|
|
8
|
+
* categories only, never a place-name dictionary, with the POI synonym table injected exactly as
|
|
9
|
+
* `poi.ts` does it.
|
|
10
|
+
*
|
|
11
|
+
* `bare_toponym` and `route_pair` score below the structural kind that already owns their population
|
|
12
|
+
* (`locality_only`, 0.85), so they surface in `QueryKindResult.alternatives` and never as the top kind.
|
|
13
|
+
* The top kind is the only thing the coordinator routes on, so pinning it is what makes these additions
|
|
14
|
+
* answer-neutral on the bare-city-name register. Their intent travels on the marker.
|
|
15
|
+
*
|
|
16
|
+
* `near_me` does win its top slot (0.91), because there is no incumbent worth preserving: a query
|
|
17
|
+
* ending "near me" is not a locality.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape"
|
|
21
|
+
|
|
22
|
+
import { carriesLetter, isDisqualifyingStreetSuffix, MAX_LOCALITY_ONLY_LENGTH, wordsOf } from "#rules"
|
|
23
|
+
/**
|
|
24
|
+
* Both refinement kinds sit a whole confidence step below `locality_only`'s 0.85
|
|
25
|
+
* so no float-comparison accident can flip the top slot.
|
|
26
|
+
*/
|
|
27
|
+
const BARE_TOPONYM_CONFIDENCE = 0.84
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Lower still for a second reason beyond the ranking discipline.
|
|
31
|
+
*
|
|
32
|
+
* A route pair is a hypothesis whose competing reading (locality + region) is more common in this corpus.
|
|
33
|
+
*/
|
|
34
|
+
const ROUTE_PAIR_CONFIDENCE = 0.55
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Above `landmark`'s venue ceiling (0.88) and above `poi_query`'s anchored band (0.90),
|
|
38
|
+
* because a deictic tail is a stronger signal than either shape heuristic:
|
|
39
|
+
* no other rule explains why "me" ends the string.
|
|
40
|
+
*/
|
|
41
|
+
const NEAR_ME_CONFIDENCE = 0.91
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Word ceiling for a single bare toponym: four covers the long tail that exists as one
|
|
45
|
+
* place name ("Newcastle upon Tyne", "Sault Sainte Marie", "Las Palmas de Gran Canaria"),
|
|
46
|
+
* and past it the input contains more than a name.
|
|
47
|
+
*/
|
|
48
|
+
const MAX_BARE_TOPONYM_WORDS = 4
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Toponymic head particles — the bounded linguistic category that makes a multi-token
|
|
52
|
+
* string one place name, with the same boundary as `@mailwoman/phrase-grouper`'s
|
|
53
|
+
* `PLACE_NAME_PARTICLES` (which covers the infix glue `de`, `am`, `aan den`) and the one
|
|
54
|
+
* job of keeping `route_pair` off "New York", "San Francisco", "Fort Worth" and their kin.
|
|
55
|
+
*
|
|
56
|
+
* It is a closed morphological class rather than a gazetteer, so growing it
|
|
57
|
+
* with actual place names is the wrong move.
|
|
58
|
+
* It is case-folded on read, because "new york" is the same query.
|
|
59
|
+
*/
|
|
60
|
+
const TOPONYM_HEAD_PARTICLES: ReadonlySet<string> = new Set([
|
|
61
|
+
// English
|
|
62
|
+
"new",
|
|
63
|
+
"old",
|
|
64
|
+
"fort",
|
|
65
|
+
"ft",
|
|
66
|
+
"port",
|
|
67
|
+
"lake",
|
|
68
|
+
"mount",
|
|
69
|
+
"mt",
|
|
70
|
+
"north",
|
|
71
|
+
"south",
|
|
72
|
+
"east",
|
|
73
|
+
"west",
|
|
74
|
+
"upper",
|
|
75
|
+
"lower",
|
|
76
|
+
"great",
|
|
77
|
+
"little",
|
|
78
|
+
"saint",
|
|
79
|
+
"st",
|
|
80
|
+
"st.",
|
|
81
|
+
// Romance
|
|
82
|
+
"san",
|
|
83
|
+
"santa",
|
|
84
|
+
"santo",
|
|
85
|
+
"são",
|
|
86
|
+
"sao",
|
|
87
|
+
"los",
|
|
88
|
+
"las",
|
|
89
|
+
"el",
|
|
90
|
+
"la",
|
|
91
|
+
"le",
|
|
92
|
+
"les",
|
|
93
|
+
"villa",
|
|
94
|
+
"rio",
|
|
95
|
+
"nueva",
|
|
96
|
+
"nuevo",
|
|
97
|
+
"puerto",
|
|
98
|
+
"ciudad",
|
|
99
|
+
"campo",
|
|
100
|
+
"monte",
|
|
101
|
+
"castel",
|
|
102
|
+
"borgo",
|
|
103
|
+
// Germanic / Nordic
|
|
104
|
+
"bad",
|
|
105
|
+
"sankt",
|
|
106
|
+
"neu",
|
|
107
|
+
"alt",
|
|
108
|
+
"groß",
|
|
109
|
+
"gross",
|
|
110
|
+
"klein",
|
|
111
|
+
"ober",
|
|
112
|
+
"unter",
|
|
113
|
+
"nieuw",
|
|
114
|
+
"oud",
|
|
115
|
+
"ny",
|
|
116
|
+
"stor",
|
|
117
|
+
"lille",
|
|
118
|
+
"sint",
|
|
119
|
+
// Definite article as a head — "The Valley" (Anguilla), "The Hague", "The Bottom".
|
|
120
|
+
"the",
|
|
121
|
+
// Generic toponymic heads outside the Latin/Germanic families, each a common
|
|
122
|
+
// noun in its own language — Semitic "tel" (mound), Malay "kuala" (confluence),
|
|
123
|
+
// Khmer "phnom" (hill) — that heads a place name the way "mount" does.
|
|
124
|
+
// `mailwoman/test/kind-intent-invariance.test.ts` covers them.
|
|
125
|
+
"tel",
|
|
126
|
+
"kuala",
|
|
127
|
+
"phnom",
|
|
128
|
+
"cape",
|
|
129
|
+
"isle",
|
|
130
|
+
"isla",
|
|
131
|
+
"ilha",
|
|
132
|
+
])
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Generic toponymic tail nouns — the other half of the same bounded morphological
|
|
136
|
+
* class, so a place name whose last token is a settlement/landform generic
|
|
137
|
+
* ("Belize City", "George Town", "Palm Springs") is one name rather than two.
|
|
138
|
+
*/
|
|
139
|
+
const TOPONYM_TAIL_NOUNS: ReadonlySet<string> = new Set([
|
|
140
|
+
"city",
|
|
141
|
+
"town",
|
|
142
|
+
"ville",
|
|
143
|
+
"village",
|
|
144
|
+
"borough",
|
|
145
|
+
"springs",
|
|
146
|
+
"falls",
|
|
147
|
+
"beach",
|
|
148
|
+
"heights",
|
|
149
|
+
"valley",
|
|
150
|
+
"island",
|
|
151
|
+
"islands",
|
|
152
|
+
"bay",
|
|
153
|
+
"harbour",
|
|
154
|
+
"harbor",
|
|
155
|
+
"park",
|
|
156
|
+
"hills",
|
|
157
|
+
"river",
|
|
158
|
+
"creek",
|
|
159
|
+
"point",
|
|
160
|
+
"stadt",
|
|
161
|
+
"burg",
|
|
162
|
+
])
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Deictic locator tails — "near me", "nearby", "around here", "in my area" —
|
|
166
|
+
* the bounded class `preposition + a reference to the asker`, where `me`, `here`,
|
|
167
|
+
* `my <noun>` are function words rather than places.
|
|
168
|
+
*
|
|
169
|
+
* Anchored to the end of the string (`$`) on purpose: the query names no anchor,
|
|
170
|
+
* so anything after the locator is an anchor and disqualifies it.
|
|
171
|
+
*
|
|
172
|
+
* Linear by construction: every alternative begins with a required literal and the only
|
|
173
|
+
* quantifiers are bounded `\s+` runs between two required literals or trailing before `$`,
|
|
174
|
+
* with no unbounded-whitespace-then-literal prefix (the `js/polynomial-redos` shape).
|
|
175
|
+
* `ANCHOR_SEPARATOR` in `poi.ts` uses the same analysis.
|
|
176
|
+
*/
|
|
177
|
+
const DEICTIC_LOCATOR_TAIL =
|
|
178
|
+
/\b(?:near|close\s+to|next\s+to|around|by|closest\s+to|nearest\s+to)\s+(?:me|us|here|my\s+(?:location|position|area|place|house|home))\s*$/
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* The adverbial half of the same class, where the deixis is baked into the word
|
|
182
|
+
* rather than introduced by a preposition.
|
|
183
|
+
*/
|
|
184
|
+
const DEICTIC_ADVERB_TAIL =
|
|
185
|
+
/\b(?:nearby|near\s?by|close\s+by|around\s+here|in\s+my\s+(?:area|neighborhood|neighbourhood))\s*$/
|
|
186
|
+
|
|
187
|
+
function hasDeicticTail(lowercased: string): boolean {
|
|
188
|
+
return DEICTIC_LOCATOR_TAIL.test(lowercased) || DEICTIC_ADVERB_TAIL.test(lowercased)
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* The conditions `bare_toponym` and `route_pair` share: no address grammar of any kind,
|
|
193
|
+
* one segment, alpha throughout.
|
|
194
|
+
*
|
|
195
|
+
* @returns the word list when the input clears them, `null` when it does not.
|
|
196
|
+
* The conditions are deliberately a superset of `scoreLocalityOnly`'s, so `bare_toponym` is
|
|
197
|
+
* a strict refinement of `locality_only` and can never fire where `locality_only` did not.
|
|
198
|
+
*/
|
|
199
|
+
function bareNameWords(input: NormalizedInputLite, shape: QueryShapeLike): string[] | null {
|
|
200
|
+
const text = input.normalized.trim()
|
|
201
|
+
|
|
202
|
+
if (!text || text.length > MAX_LOCALITY_ONLY_LENGTH) return null
|
|
203
|
+
|
|
204
|
+
// A recognized postcode/known format is address grammar, so no bare toponym survives it.
|
|
205
|
+
if (shape.knownFormats.length) return null
|
|
206
|
+
|
|
207
|
+
// `alpha` excludes every house number and postcode by construction, the cheapest statement of
|
|
208
|
+
// "no address grammar"; it is silent about whether a name is present, so the letter test stands
|
|
209
|
+
// beside it because `foldInputClass` answers `alpha` for input carrying no classified token.
|
|
210
|
+
if (shape.characterClass !== "alpha" || !carriesLetter(text)) return null
|
|
211
|
+
|
|
212
|
+
// A comma is the admin-context marker ("Paris, FR"), so one segment or the name is not bare.
|
|
213
|
+
if ((shape.segments?.length ?? 1) !== 1) return null
|
|
214
|
+
|
|
215
|
+
const lowercased = text.toLowerCase()
|
|
216
|
+
|
|
217
|
+
if (hasDeicticTail(lowercased)) return null
|
|
218
|
+
|
|
219
|
+
const words = wordsOf(text)
|
|
220
|
+
|
|
221
|
+
if (!words.length || words.length > MAX_BARE_TOPONYM_WORDS) return null
|
|
222
|
+
|
|
223
|
+
for (const word of words) {
|
|
224
|
+
if (isDisqualifyingStreetSuffix(word)) return null
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
return words
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* `bare_toponym` rule: a single coherent place-name carrying no address grammar, feeding the
|
|
232
|
+
* declared-ambiguity path without asserting which place — that is the resolver's question,
|
|
233
|
+
* decided by the answer's dominance margin in `mailwoman/query-intent.ts`.
|
|
234
|
+
*/
|
|
235
|
+
export function scoreBareToponym(input: NormalizedInputLite, shape: QueryShapeLike): number {
|
|
236
|
+
return bareNameWords(input, shape) ? BARE_TOPONYM_CONFIDENCE : 0
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* `route_pair` rule: exactly two toponym-shaped tokens with no token between them.
|
|
241
|
+
*
|
|
242
|
+
* The known confound is structural and unfixable here.
|
|
243
|
+
* "Paris London" and "Moscow Idaho" have the same string shape.
|
|
244
|
+
*
|
|
245
|
+
* A classifier needs gazetteer knowledge that Idaho is a region to distinguish them.
|
|
246
|
+
* ROAD_TO_V9 §4.3 therefore specifies classification plus a declared fork.
|
|
247
|
+
*
|
|
248
|
+
* Both readings appear in the marker, neither wins and the resolver keeps its existing answer.
|
|
249
|
+
*
|
|
250
|
+
* The structurally separable class is a two-token single name ("New York", "Fort Worth")
|
|
251
|
+
* with a toponymic head particle.
|
|
252
|
+
* That guard keeps the fork off the common case.
|
|
253
|
+
*/
|
|
254
|
+
export function scoreRoutePair(input: NormalizedInputLite, shape: QueryShapeLike): number {
|
|
255
|
+
const words = bareNameWords(input, shape)
|
|
256
|
+
|
|
257
|
+
if (!words || words.length !== 2) return 0
|
|
258
|
+
|
|
259
|
+
const [first, second] = [words[0]!.toLowerCase(), words[1]!.toLowerCase()]
|
|
260
|
+
|
|
261
|
+
// Reduplication — "Pago Pago", "Baden-Baden", "Walla Walla" — is a universal single-name
|
|
262
|
+
// signal that needs no lexicon, because a route from a place to itself is invalid.
|
|
263
|
+
if (first === second) return 0
|
|
264
|
+
|
|
265
|
+
if (TOPONYM_HEAD_PARTICLES.has(first) || TOPONYM_HEAD_PARTICLES.has(second)) return 0
|
|
266
|
+
|
|
267
|
+
if (TOPONYM_TAIL_NOUNS.has(second)) return 0
|
|
268
|
+
|
|
269
|
+
return ROUTE_PAIR_CONFIDENCE
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* `near_me` rule: a subject plus a deictic locator, with no anchor.
|
|
274
|
+
*
|
|
275
|
+
* A non-empty subject is required, so a bare "near me" stays with the `landmark` leaders rule.
|
|
276
|
+
*/
|
|
277
|
+
export function scoreNearMe(input: NormalizedInputLite, _shape: QueryShapeLike): number {
|
|
278
|
+
const lowercased = input.normalized.trim().toLowerCase()
|
|
279
|
+
|
|
280
|
+
if (!hasDeicticTail(lowercased)) return 0
|
|
281
|
+
|
|
282
|
+
// The subject is everything before the locator.
|
|
283
|
+
// `hasDeicticTail` already anchored the match to the end, so the first match
|
|
284
|
+
// index marks where the subject stops.
|
|
285
|
+
const match = DEICTIC_LOCATOR_TAIL.exec(lowercased) ?? DEICTIC_ADVERB_TAIL.exec(lowercased)
|
|
286
|
+
|
|
287
|
+
if (!match) return 0
|
|
288
|
+
|
|
289
|
+
return lowercased.slice(0, match.index).trim() ? NEAR_ME_CONFIDENCE : 0
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* The subject of a `near_me` query is the requested category or thing with the locator stripped.
|
|
294
|
+
*
|
|
295
|
+
* It is empty when the rule would not have fired.
|
|
296
|
+
* The subject builds marker evidence and never a route.
|
|
297
|
+
*/
|
|
298
|
+
export function nearMeSubject(input: NormalizedInputLite): string {
|
|
299
|
+
const trimmed = input.normalized.trim()
|
|
300
|
+
const lowercased = trimmed.toLowerCase()
|
|
301
|
+
const match = DEICTIC_LOCATOR_TAIL.exec(lowercased) ?? DEICTIC_ADVERB_TAIL.exec(lowercased)
|
|
302
|
+
|
|
303
|
+
if (!match) return ""
|
|
304
|
+
|
|
305
|
+
return trimmed.slice(0, match.index).trim()
|
|
306
|
+
}
|