@mailwoman/kind-classifier 10.0.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/lib/classify.ts +33 -34
- package/lib/index.ts +1 -8
- package/lib/intent/markers.ts +28 -25
- package/lib/intent/rules.ts +82 -83
- package/lib/poi.ts +96 -84
- package/lib/rules.ts +141 -79
- package/out/classify.d.ts +18 -20
- package/out/classify.d.ts.map +1 -1
- package/out/classify.js +28 -31
- package/out/classify.js.map +1 -1
- package/out/index.d.ts +1 -8
- package/out/index.d.ts.map +1 -1
- package/out/index.js +1 -8
- package/out/index.js.map +1 -1
- package/out/intent/markers.d.ts +17 -18
- package/out/intent/markers.d.ts.map +1 -1
- package/out/intent/markers.js +24 -23
- package/out/intent/markers.js.map +1 -1
- package/out/intent/rules.d.ts +31 -35
- package/out/intent/rules.d.ts.map +1 -1
- package/out/intent/rules.js +82 -83
- package/out/intent/rules.js.map +1 -1
- package/out/poi.d.ts +64 -56
- package/out/poi.d.ts.map +1 -1
- package/out/poi.js +60 -50
- package/out/poi.js.map +1 -1
- package/out/rules.d.ts +23 -40
- package/out/rules.d.ts.map +1 -1
- package/out/rules.js +117 -80
- package/out/rules.js.map +1 -1
- package/package.json +15 -50
package/out/poi.d.ts
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* POI subject detection for the `poi_query` kind. The lexicon is injected (`POIPhraseLookup`) so this
|
|
7
|
+
* package keeps its bitter-lesson invariant (no dictionaries in-tree); the phrase table lives in
|
|
8
|
+
* `@mailwoman/poi-taxonomy` and is wired in by `createRuntimePipeline` behind the `poiQueryKind` flag.
|
|
9
|
+
* Spec §3.1.
|
|
10
10
|
*/
|
|
11
11
|
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape";
|
|
12
12
|
/**
|
|
@@ -14,9 +14,9 @@ import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } fr
|
|
|
14
14
|
*/
|
|
15
15
|
export interface POIPhraseMatch {
|
|
16
16
|
/**
|
|
17
|
-
* The matched subject's identifier
|
|
18
|
-
* `kind: "
|
|
19
|
-
*
|
|
17
|
+
* The matched subject's identifier: a `@mailwoman/poi-taxonomy` category id for
|
|
18
|
+
* `kind: "category"`, otherwise the canonical display name; `matchPOISubject`
|
|
19
|
+
* treats it opaquely and the caller interprets it per `kind`.
|
|
20
20
|
*/
|
|
21
21
|
categoryID: string;
|
|
22
22
|
matchedPhrase: string;
|
|
@@ -24,42 +24,45 @@ export interface POIPhraseMatch {
|
|
|
24
24
|
mechanism?: "exact" | "locale_normalized" | "typo";
|
|
25
25
|
inputPhrase?: string;
|
|
26
26
|
/**
|
|
27
|
-
* Absent
|
|
28
|
-
* source-compatible.
|
|
27
|
+
* Absent means `"category"`; optional so existing `POIPhraseLookup` implementors stay source-compatible.
|
|
29
28
|
*/
|
|
30
29
|
kind?: "category" | "brand" | "name";
|
|
31
30
|
/**
|
|
32
|
-
* Wikidata QID
|
|
31
|
+
* Wikidata QID when known, `kind: "brand"` only.
|
|
32
|
+
* Absent when a brand resolved using its name only.
|
|
33
33
|
*/
|
|
34
34
|
wikidata?: string;
|
|
35
35
|
/**
|
|
36
|
-
* Whether this hit is one member of a set the caller must search
|
|
37
|
-
* list.
|
|
36
|
+
* Whether this hit is one member of a set the caller must search together
|
|
37
|
+
* rather than one candidate in a preference list.
|
|
38
38
|
*
|
|
39
|
-
* A lookup returning several hits means two different things
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
* authored.
|
|
39
|
+
* A lookup returning several hits means two different things: a phrase index
|
|
40
|
+
* returns the categories one typed phrase could name, the curated reading first
|
|
41
|
+
* (`credit union` → the `bank` rollup its synonym redirects to), while an affordance
|
|
42
|
+
* rung returns every entity kind that affords one activity in a stable enumeration.
|
|
43
|
+
* The enumeration does not express a preference.
|
|
45
44
|
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
45
|
+
* The first result would impose an ordering that the source does not provide.
|
|
46
|
+
* Set on every member of such a set, so {@link matchPOISubject} returns them all
|
|
47
|
+
* and the POI branch searches their union.
|
|
48
|
+
*
|
|
49
|
+
* Absent, the committed lexicon's shape, keeps the first-hit reading.
|
|
49
50
|
*/
|
|
50
51
|
searchAsSet?: boolean;
|
|
51
52
|
/**
|
|
52
|
-
* ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to.
|
|
53
|
-
* everywhere.
|
|
53
|
+
* ISO 3166-1 alpha-2 countries the authority behind this hit scopes its claim to.
|
|
54
|
+
* Absent means the condition is true everywhere.
|
|
54
55
|
*
|
|
55
|
-
* A scope is a statement about
|
|
56
|
-
* the caller's locale: the locale is the lens the
|
|
57
|
-
*
|
|
56
|
+
* A scope is a statement about establishments, so it is judged against the country of
|
|
57
|
+
* the place being searched rather than the caller's locale: the locale is the lens the
|
|
58
|
+
* phrase is read through and makes no statement about where the condition is true.
|
|
59
|
+
* `matchPOISubject` returns the value unchanged.
|
|
60
|
+
* The POI intent stage binds it once the anchor has resolved.
|
|
58
61
|
*/
|
|
59
62
|
countryScope?: readonly string[];
|
|
60
63
|
}
|
|
61
64
|
/**
|
|
62
|
-
* Injected phrase→category lookup
|
|
65
|
+
* Injected phrase→category lookup, exact-phrase and locale-aware, returning `[]` on miss.
|
|
63
66
|
*/
|
|
64
67
|
export type POIPhraseLookup = (phrase: string, locale?: string) => ReadonlyArray<POIPhraseMatch>;
|
|
65
68
|
export type POISpatialRelation = "comma" | "near" | "in" | "at" | "around" | "to";
|
|
@@ -72,25 +75,24 @@ export interface POIQuerySpan {
|
|
|
72
75
|
end: number;
|
|
73
76
|
}
|
|
74
77
|
/**
|
|
75
|
-
* Which lexicon this hit came from.
|
|
78
|
+
* Which lexicon this hit came from.
|
|
79
|
+
*
|
|
80
|
+
* Category lookups set `"category"` as the backward-compatible default.
|
|
76
81
|
*/
|
|
77
82
|
export interface POISubjectMatch {
|
|
78
83
|
/**
|
|
79
|
-
* The hit the subject
|
|
80
|
-
*
|
|
84
|
+
* The hit the subject scores under.
|
|
85
|
+
*
|
|
86
|
+
* Always `matches[0]`, built together in one place so they cannot disagree.
|
|
81
87
|
*/
|
|
82
88
|
match: POIPhraseMatch;
|
|
83
89
|
/**
|
|
84
|
-
* Every category the subject reaches, `match` first
|
|
85
|
-
*
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
* nothing downstream may read position as rank.
|
|
90
|
+
* Every category the subject reaches, `match` first: one entry unless the lookup
|
|
91
|
+
* returned a {@link POIPhraseMatch.searchAsSet} set, in which case it holds the
|
|
92
|
+
* whole set and the POI branch searches their union.
|
|
93
|
+
* The order is the lookup's and states no preference.
|
|
89
94
|
*/
|
|
90
95
|
matches: POIPhraseMatch[];
|
|
91
|
-
/**
|
|
92
|
-
* The matched subject text as it appeared in the query.
|
|
93
|
-
*/
|
|
94
96
|
subject: string;
|
|
95
97
|
subjectSpan: POIQuerySpan;
|
|
96
98
|
/**
|
|
@@ -105,35 +107,41 @@ export interface POISubjectMatch {
|
|
|
105
107
|
anchorSpan?: POIQuerySpan;
|
|
106
108
|
}
|
|
107
109
|
/**
|
|
108
|
-
* Match a POI subject: the whole input, or the text before the
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
*
|
|
110
|
+
* Match a POI subject: the whole input, or the text before the first anchor
|
|
111
|
+
* separator whose prefix hits the lexicon (≤ 8 tokens).
|
|
112
|
+
*
|
|
113
|
+
* Scans separators left-to-right, because a lexicon phrase may itself contain a bare separator
|
|
114
|
+
* word ("walk in clinic") and the first separator is not necessarily the right split point.
|
|
115
|
+
* Returns null when the lexicon never fires, including comma-ridden full addresses
|
|
116
|
+
* whose leading segment is not a phrase.
|
|
112
117
|
*
|
|
113
|
-
* The winning candidate's hits are
|
|
114
|
-
* declared one.
|
|
118
|
+
* The winning candidate's hits are listed per {@link reachedMatches}.
|
|
115
119
|
*/
|
|
116
120
|
export declare function matchPOISubject(text: string, locale: string | undefined, lookup: POIPhraseLookup): POISubjectMatch | null;
|
|
117
121
|
/**
|
|
118
|
-
* `poi_query` scorer over an injected lexicon
|
|
119
|
-
* 0.88 ceiling
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
+
* `poi_query` scorer over an injected lexicon, with confidence bands: a whole-input lexicon hit
|
|
123
|
+
* scores 0.92 (above venue-landmark's 0.88 ceiling, because an exact phrase beats a shape heuristic)
|
|
124
|
+
* and a subject plus anchor 0.9.
|
|
125
|
+
*
|
|
126
|
+
* The guards keep venue-led full addresses (class 2) on the structured-address path:
|
|
127
|
+
* a remainder leading with a house number, or a 4+-segment input, scores 0.
|
|
122
128
|
*/
|
|
123
129
|
export declare function createScorePOIQuery(lookup: POIPhraseLookup, locale?: string): (input: NormalizedInputLite, shape: QueryShapeLike) => number;
|
|
124
130
|
/**
|
|
125
|
-
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
126
|
-
* "drinking fountain".
|
|
131
|
+
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
132
|
+
* to search: "tacos", "grocery store", "drinking fountain".
|
|
133
|
+
*
|
|
134
|
+
* Fires only on a whole-input lexicon hit (`remainder === ""`) whose subject is a category.
|
|
135
|
+
* A brand (`kind: "brand"`) is excluded because `POIPhraseMatch.categoryID`
|
|
136
|
+
* then holds the brand's display name.
|
|
127
137
|
*
|
|
128
|
-
*
|
|
129
|
-
* is excluded: a bare "Starbucks" is a name lookup, not a category, and the taxonomy id a category marker promises to
|
|
130
|
-
* carry does not exist for it — `POIPhraseMatch.categoryID` holds the brand's display name in that case, which would
|
|
131
|
-
* make the marker's `categoryID` evidence a lie.
|
|
138
|
+
* That value would misrepresent the marker's `categoryID` evidence.
|
|
132
139
|
*/
|
|
133
140
|
export declare function createScorePOICategory(lookup: POIPhraseLookup, locale?: string): (input: NormalizedInputLite, shape: QueryShapeLike) => number;
|
|
134
141
|
/**
|
|
135
|
-
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence
|
|
136
|
-
*
|
|
142
|
+
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence;
|
|
143
|
+
* `null` when the input is not a bare category, under the same conditions as
|
|
144
|
+
* {@link createScorePOICategory} so the two cannot disagree.
|
|
137
145
|
*/
|
|
138
146
|
export declare function matchPOICategory(text: string, locale: string | undefined, lookup: POIPhraseLookup): POIPhraseMatch | null;
|
|
139
147
|
//# sourceMappingURL=poi.d.ts.map
|
package/out/poi.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"poi.d.ts","sourceRoot":"","sources":["../lib/poi.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,OAAO,KAAK,EAAE,mBAAmB,EAAE,sBAAsB,IAAI,cAAc,EAAE,MAAM,wBAAwB,CAAA;
|
|
1
|
+
{"version":3,"file":"poi.d.ts","sourceRoot":"","sources":["../lib/poi.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,OAAO,KAAK,EAAE,mBAAmB,EAAE,sBAAsB,IAAI,cAAc,EAAE,MAAM,wBAAwB,CAAA;AAS3G;;GAEG;AACH,MAAM,WAAW,cAAc;IAC9B;;;;OAIG;IACH,UAAU,EAAE,MAAM,CAAA;IAClB,aAAa,EAAE,MAAM,CAAA;IACrB,UAAU,EAAE,MAAM,CAAA;IAClB,SAAS,CAAC,EAAE,OAAO,GAAG,mBAAmB,GAAG,MAAM,CAAA;IAClD,WAAW,CAAC,EAAE,MAAM,CAAA;IACpB;;OAEG;IACH,IAAI,CAAC,EAAE,UAAU,GAAG,OAAO,GAAG,MAAM,CAAA;IACpC;;;OAGG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAA;IACjB;;;;;;;;;;;;;;;OAeG;IACH,WAAW,CAAC,EAAE,OAAO,CAAA;IACrB;;;;;;;;;OASG;IACH,YAAY,CAAC,EAAE,SAAS,MAAM,EAAE,CAAA;CAChC;AAED;;GAEG;AACH,MAAM,MAAM,eAAe,GAAG,CAAC,MAAM,EAAE,MAAM,EAAE,MAAM,CAAC,EAAE,MAAM,KAAK,aAAa,CAAC,cAAc,CAAC,CAAA;AAEhG,MAAM,MAAM,kBAAkB,GAAG,OAAO,GAAG,MAAM,GAAG,IAAI,GAAG,IAAI,GAAG,QAAQ,GAAG,IAAI,CAAA;AAEjF,MAAM,WAAW,YAAY;IAC5B,IAAI,EAAE,MAAM,CAAA;IACZ;;OAEG;IACH,KAAK,EAAE,MAAM,CAAA;IACb,GAAG,EAAE,MAAM,CAAA;CACX;AAED;;;;GAIG;AACH,MAAM,WAAW,eAAe;IAC/B;;;;OAIG;IACH,KAAK,EAAE,cAAc,CAAA;IACrB;;;;;OAKG;IACH,OAAO,EAAE,cAAc,EAAE,CAAA;IACzB,OAAO,EAAE,MAAM,CAAA;IACf,WAAW,EAAE,YAAY,CAAA;IACzB;;OAEG;IACH,QAAQ,CAAC,EAAE,kBAAkB,CAAA;IAC7B,YAAY,CAAC,EAAE,YAAY,CAAA;IAC3B;;OAEG;IACH,SAAS,EAAE,MAAM,CAAA;IACjB,UAAU,CAAC,EAAE,YAAY,CAAA;CACzB;AAuCD;;;;;;;;;;GAUG;AACH,wBAAgB,eAAe,CAC9B,IAAI,EAAE,MAAM,EACZ,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,MAAM,EAAE,eAAe,GACrB,eAAe,GAAG,IAAI,CAoExB;AAED;;;;;;;GAOG;AACH,wBAAgB,mBAAmB,CAClC,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,MAAM,GACb,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,KAAK,MAAM,CAiB/D;AAYD;;;;;;;;;GASG;AACH,wBAAgB,sBAAsB,CACrC,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,MAAM,GACb,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,KAAK,MAAM,CAU/D;AAED;;;;GAIG;AACH,wBAAgB,gBAAgB,CAC/B,IAAI,EAAE,MAAM,EACZ,MAAM,EAAE,MAAM,GAAG,SAAS,EAC1B,MAAM,EAAE,eAAe,GACrB,cAAc,GAAG,IAAI,CAQvB"}
|
package/out/poi.js
CHANGED
|
@@ -3,58 +3,62 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* POI subject detection for the `poi_query` kind. The lexicon is injected (`POIPhraseLookup`) so this
|
|
7
|
+
* package keeps its bitter-lesson invariant (no dictionaries in-tree); the phrase table lives in
|
|
8
|
+
* `@mailwoman/poi-taxonomy` and is wired in by `createRuntimePipeline` behind the `poiQueryKind` flag.
|
|
9
|
+
* Spec §3.1.
|
|
10
10
|
*/
|
|
11
11
|
/**
|
|
12
|
-
* Comma-segment ceiling for a POI-led query.
|
|
13
|
-
*
|
|
12
|
+
* Comma-segment ceiling for a POI-led query.
|
|
13
|
+
*
|
|
14
|
+
* Past it the input is a venue plus a full address (`X, 350 5th Ave, New York, NY`),
|
|
15
|
+
* which the structured-address scorer should claim instead.
|
|
14
16
|
*/
|
|
15
17
|
const MAX_POI_SEGMENTS = 3;
|
|
16
18
|
/**
|
|
17
|
-
* Anchor separator between subject and place: comma, or near/in/at/around/to —
|
|
18
|
-
* hits the lexicon.
|
|
19
|
+
* Anchor separator between subject and place: comma, or near/in/at/around/to —
|
|
20
|
+
* scanned left-to-right until a prefix hits the lexicon.
|
|
21
|
+
*
|
|
22
|
+
* Linear by construction (no polynomial ReDoS): neither alternative places an unbounded
|
|
23
|
+
* whitespace quantifier before its required literal, the classic `js/polynomial-redos` shape.
|
|
24
|
+
* The comma alternative starts at the literal `,`.
|
|
19
25
|
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
* before a fixed anchor word. Every remaining quantifier (`,\s*`, `…\s+`) is _trailing_ — it runs only after the
|
|
24
|
-
* required literal has already matched and nothing follows it, so it never backtracks. Each start offset does O(1)
|
|
25
|
-
* work, making `matchAll` O(n).
|
|
26
|
+
* The anchor alternative starts at one `\s` before a fixed anchor word.
|
|
27
|
+
* Every remaining quantifier is trailing and runs only after the required literal matches.
|
|
28
|
+
* Each start offset does O(1) work, so `matchAll` is O(n).
|
|
26
29
|
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
* lastIndex) identical, preserving the exact subsequent-match sequence. Verified: 0 divergences across 22.7k inputs
|
|
32
|
-
* (systematic + fuzzed adversarial whitespace + shared-whitespace anchor/comma chains).
|
|
30
|
+
* Behavior is byte-identical to the previous `\s*,\s*|\s+(?:…)\s+`, because `matchPOISubject`
|
|
31
|
+
* trims both the subject and the remainder, so surrounding whitespace is redundant.
|
|
32
|
+
* The leading quantifier only shifted the match start within a whitespace run, while the retained
|
|
33
|
+
* trailing greedy quantifier keeps the match end and thus `matchAll`'s lastIndex identical.
|
|
33
34
|
*/
|
|
34
35
|
const ANCHOR_SEPARATOR = /,\s*|\s(near|in|at|around|to)\s+/gi;
|
|
35
36
|
/**
|
|
36
|
-
* Longest subject
|
|
37
|
+
* Longest subject accepted, in tokens: eight covers compound taxonomy phrases
|
|
38
|
+
* while bounding lexicon probes.
|
|
37
39
|
*/
|
|
38
40
|
const MAX_SUBJECT_TOKENS = 8;
|
|
39
41
|
/**
|
|
40
|
-
* The categories one candidate subject reaches
|
|
42
|
+
* The categories one candidate subject reaches: the whole array when the first hit
|
|
43
|
+
* declares {@link POIPhraseMatch.searchAsSet}, preserved as the lookup returned it
|
|
44
|
+
* and never filtered, so a rung that flagged only some members keeps every member.
|
|
41
45
|
*
|
|
42
|
-
* The
|
|
43
|
-
*
|
|
44
|
-
* visible to whoever reads the set. Otherwise the first hit alone, which is the preference-list reading the committed
|
|
45
|
-
* phrase index has always had.
|
|
46
|
+
* The inconsistency stays visible.
|
|
47
|
+
* Otherwise, use only the first hit.
|
|
46
48
|
*/
|
|
47
49
|
function reachedMatches(hits) {
|
|
48
50
|
return hits[0].searchAsSet ? [...hits] : [hits[0]];
|
|
49
51
|
}
|
|
50
52
|
/**
|
|
51
|
-
* Match a POI subject: the whole input, or the text before the
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
53
|
+
* Match a POI subject: the whole input, or the text before the first anchor
|
|
54
|
+
* separator whose prefix hits the lexicon (≤ 8 tokens).
|
|
55
|
+
*
|
|
56
|
+
* Scans separators left-to-right, because a lexicon phrase may itself contain a bare separator
|
|
57
|
+
* word ("walk in clinic") and the first separator is not necessarily the right split point.
|
|
58
|
+
* Returns null when the lexicon never fires, including comma-ridden full addresses
|
|
59
|
+
* whose leading segment is not a phrase.
|
|
55
60
|
*
|
|
56
|
-
* The winning candidate's hits are
|
|
57
|
-
* declared one.
|
|
61
|
+
* The winning candidate's hits are listed per {@link reachedMatches}.
|
|
58
62
|
*/
|
|
59
63
|
export function matchPOISubject(text, locale, lookup) {
|
|
60
64
|
const trimmed = text.trim();
|
|
@@ -76,8 +80,8 @@ export function matchPOISubject(text, locale, lookup) {
|
|
|
76
80
|
if (separator.index === 0)
|
|
77
81
|
continue;
|
|
78
82
|
const subject = trimmed.slice(0, separator.index).trim();
|
|
79
|
-
// Subjects only grow as the scan moves right
|
|
80
|
-
// split
|
|
83
|
+
// Subjects only grow as the scan moves right, so once over budget later splits are too.
|
|
84
|
+
// Whitespace-only split rather than `wordsOf`, because a comma inside a subject is real content.
|
|
81
85
|
if (subject.split(/\s+/).length > MAX_SUBJECT_TOKENS)
|
|
82
86
|
break;
|
|
83
87
|
const hits = lookup(subject, locale);
|
|
@@ -117,10 +121,12 @@ export function matchPOISubject(text, locale, lookup) {
|
|
|
117
121
|
return null;
|
|
118
122
|
}
|
|
119
123
|
/**
|
|
120
|
-
* `poi_query` scorer over an injected lexicon
|
|
121
|
-
* 0.88 ceiling
|
|
122
|
-
*
|
|
123
|
-
*
|
|
124
|
+
* `poi_query` scorer over an injected lexicon, with confidence bands: a whole-input lexicon hit
|
|
125
|
+
* scores 0.92 (above venue-landmark's 0.88 ceiling, because an exact phrase beats a shape heuristic)
|
|
126
|
+
* and a subject plus anchor 0.9.
|
|
127
|
+
*
|
|
128
|
+
* The guards keep venue-led full addresses (class 2) on the structured-address path:
|
|
129
|
+
* a remainder leading with a house number, or a 4+-segment input, scores 0.
|
|
124
130
|
*/
|
|
125
131
|
export function createScorePOIQuery(lookup, locale) {
|
|
126
132
|
return (input, shape) => {
|
|
@@ -139,20 +145,23 @@ export function createScorePOIQuery(lookup, locale) {
|
|
|
139
145
|
};
|
|
140
146
|
}
|
|
141
147
|
/**
|
|
142
|
-
* Confidence band for a bare category
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
*
|
|
148
|
+
* Confidence band for a bare category, one notch above `poi_query`'s whole-input
|
|
149
|
+
* band (0.92) so only the anchorless subset takes the top slot and every anchored
|
|
150
|
+
* POI query keeps scoring `poi_query` as before.
|
|
151
|
+
*
|
|
152
|
+
* The coordinator's POI branch accepts both kinds, so the routing is identical either way.
|
|
153
|
+
* The split exists so the marker can say "you supplied a category and no place".
|
|
146
154
|
*/
|
|
147
155
|
const POI_CATEGORY_CONFIDENCE = 0.93;
|
|
148
156
|
/**
|
|
149
|
-
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
150
|
-
* "drinking fountain".
|
|
157
|
+
* `poi_category` scorer (ROAD_TO_V9 §4.4) — a bare taxonomy category with nowhere
|
|
158
|
+
* to search: "tacos", "grocery store", "drinking fountain".
|
|
159
|
+
*
|
|
160
|
+
* Fires only on a whole-input lexicon hit (`remainder === ""`) whose subject is a category.
|
|
161
|
+
* A brand (`kind: "brand"`) is excluded because `POIPhraseMatch.categoryID`
|
|
162
|
+
* then holds the brand's display name.
|
|
151
163
|
*
|
|
152
|
-
*
|
|
153
|
-
* is excluded: a bare "Starbucks" is a name lookup, not a category, and the taxonomy id a category marker promises to
|
|
154
|
-
* carry does not exist for it — `POIPhraseMatch.categoryID` holds the brand's display name in that case, which would
|
|
155
|
-
* make the marker's `categoryID` evidence a lie.
|
|
164
|
+
* That value would misrepresent the marker's `categoryID` evidence.
|
|
156
165
|
*/
|
|
157
166
|
export function createScorePOICategory(lookup, locale) {
|
|
158
167
|
return (input, _shape) => {
|
|
@@ -165,8 +174,9 @@ export function createScorePOICategory(lookup, locale) {
|
|
|
165
174
|
};
|
|
166
175
|
}
|
|
167
176
|
/**
|
|
168
|
-
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence
|
|
169
|
-
*
|
|
177
|
+
* The whole-input category hit behind a `poi_category` verdict, for the marker's evidence;
|
|
178
|
+
* `null` when the input is not a bare category, under the same conditions as
|
|
179
|
+
* {@link createScorePOICategory} so the two cannot disagree.
|
|
170
180
|
*/
|
|
171
181
|
export function matchPOICategory(text, locale, lookup) {
|
|
172
182
|
const matched = matchPOISubject(text, locale, lookup);
|
package/out/poi.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"poi.js","sourceRoot":"","sources":["../lib/poi.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAGH
|
|
1
|
+
{"version":3,"file":"poi.js","sourceRoot":"","sources":["../lib/poi.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAGH;;;;;GAKG;AACH,MAAM,gBAAgB,GAAG,CAAC,CAAA;AAwG1B;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,gBAAgB,GAAG,oCAAoC,CAAA;AAE7D;;;GAGG;AACH,MAAM,kBAAkB,GAAG,CAAC,CAAA;AAE5B;;;;;;;GAOG;AACH,SAAS,cAAc,CAAC,IAAmC;IAC1D,OAAO,IAAI,CAAC,CAAC,CAAE,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAE,CAAC,CAAA;AACrD,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,eAAe,CAC9B,IAAY,EACZ,MAA0B,EAC1B,MAAuB;IAEvB,MAAM,OAAO,GAAG,IAAI,CAAC,IAAI,EAAE,CAAA;IAC3B,MAAM,UAAU,GAAG,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,CAAA;IAExC,IAAI,CAAC,OAAO;QAAE,OAAO,IAAI,CAAA;IAEzB,MAAM,KAAK,GAAG,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,CAAA;IAErC,IAAI,KAAK,CAAC,MAAM,EAAE,CAAC;QAClB,MAAM,OAAO,GAAG,cAAc,CAAC,KAAK,CAAC,CAAA;QAErC,OAAO;YACN,KAAK,EAAE,OAAO,CAAC,CAAC,CAAE;YAClB,OAAO;YACP,OAAO,EAAE,OAAO;YAChB,WAAW,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,UAAU,EAAE,GAAG,EAAE,UAAU,GAAG,OAAO,CAAC,MAAM,EAAE;YACnF,SAAS,EAAE,EAAE;SACb,CAAA;IACF,CAAC;IAED,KAAK,MAAM,SAAS,IAAI,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAC,EAAE,CAAC;QAC5D,IAAI,SAAS,CAAC,KAAK,KAAK,CAAC;YAAE,SAAQ;QAEnC,MAAM,OAAO,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,SAAS,CAAC,KAAK,CAAC,CAAC,IAAI,EAAE,CAAA;QAExD,wFAAwF;QACxF,iGAAiG;QACjG,IAAI,OAAO,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,GAAG,kBAAkB;YAAE,MAAK;QAE3D,MAAM,IAAI,GAAG,MAAM,CAAC,OAAO,EAAE,MAAM,CAAC,CAAA;QAEpC,IAAI,CAAC,IAAI,CAAC,MAAM;YAAE,SAAQ;QAE1B,MAAM,SAAS,GAAG,OAAO,CAAC,KAAK,CAAC,SAAS,CAAC,KAAK,GAAG,SAAS,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,CAAA;QAE7E,IAAI,CAAC,SAAS;YAAE,SAAQ;QAExB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,SAAS,CAAC,KAAK,CAAC,CAAC,OAAO,CAAC,OAAO,CAAC,CAAA;QACxE,MAAM,YAAY,GAAG,OAAO,CAAC,OAAO,CAAC,SAAS,EAAE,SAAS,CAAC,KAAK,GAAG,SAAS,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAA;QACtF,MAAM,YAAY,GAAG,SAAS,CAAC,CAAC,CAAC,IAAI,GAAG,CAAA;QACxC,MAAM,cAAc,GAAG,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC,KAAK,GAAG,SAAS,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC,KAAK,CAAA;QAC5G,MAAM,OAAO,GAAG,cAAc,CAAC,IAAI,CAAC,CAAA;QAEpC,OAAO;YACN,KAAK,EAAE,OAAO,CAAC,CAAC,CAAE;YAClB,OAAO;YACP,OAAO;YACP,WAAW,EAAE;gBACZ,IAAI,EAAE,OAAO;gBACb,KAAK,EAAE,UAAU,GAAG,aAAa;gBACjC,GAAG,EAAE,UAAU,GAAG,aAAa,GAAG,OAAO,CAAC,MAAM;aAChD;YACD,QAAQ,EAAE,YAAY,KAAK,GAAG,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAE,YAAY,CAAC,WAAW,EAAyB;YAC7F,YAAY,EAAE;gBACb,IAAI,EAAE,YAAY;gBAClB,KAAK,EAAE,UAAU,GAAG,cAAc;gBAClC,GAAG,EAAE,UAAU,GAAG,cAAc,GAAG,YAAY,CAAC,MAAM;aACtD;YACD,SAAS;YACT,UAAU,EAAE;gBACX,IAAI,EAAE,SAAS;gBACf,KAAK,EAAE,UAAU,GAAG,YAAY;gBAChC,GAAG,EAAE,UAAU,GAAG,YAAY,GAAG,SAAS,CAAC,MAAM;aACjD;SACD,CAAA;IACF,CAAC;IAED,OAAO,IAAI,CAAA;AACZ,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,mBAAmB,CAClC,MAAuB,EACvB,MAAe;IAEf,OAAO,CAAC,KAAK,EAAE,KAAK,EAAE,EAAE;QACvB,MAAM,OAAO,GAAG,eAAe,CAAC,KAAK,CAAC,UAAU,EAAE,MAAM,IAAI,KAAK,CAAC,aAAa,EAAE,MAAM,CAAC,CAAA;QAExF,IAAI,CAAC,OAAO;YAAE,OAAO,CAAC,CAAA;QAEtB,IAAI,OAAO,CAAC,SAAS,KAAK,EAAE;YAAE,OAAO,IAAI,GAAG,OAAO,CAAC,KAAK,CAAC,UAAU,CAAA;QAEpE,gFAAgF;QAChF,IAAI,QAAQ,CAAC,IAAI,CAAC,OAAO,CAAC,SAAS,CAAC;YAAE,OAAO,CAAC,CAAA;QAE9C,MAAM,QAAQ,GAAG,KAAK,CAAC,QAAQ,EAAE,MAAM,IAAI,CAAC,CAAA;QAE5C,IAAI,QAAQ,GAAG,gBAAgB;YAAE,OAAO,CAAC,CAAA;QAEzC,OAAO,GAAG,GAAG,OAAO,CAAC,KAAK,CAAC,UAAU,CAAA;IACtC,CAAC,CAAA;AACF,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,uBAAuB,GAAG,IAAI,CAAA;AAEpC;;;;;;;;;GASG;AACH,MAAM,UAAU,sBAAsB,CACrC,MAAuB,EACvB,MAAe;IAEf,OAAO,CAAC,KAAK,EAAE,MAAM,EAAE,EAAE;QACxB,MAAM,OAAO,GAAG,eAAe,CAAC,KAAK,CAAC,UAAU,EAAE,MAAM,IAAI,KAAK,CAAC,aAAa,EAAE,MAAM,CAAC,CAAA;QAExF,IAAI,CAAC,OAAO,IAAI,OAAO,CAAC,SAAS,KAAK,EAAE;YAAE,OAAO,CAAC,CAAA;QAElD,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,IAAI,IAAI,UAAU,CAAC,KAAK,UAAU;YAAE,OAAO,CAAC,CAAA;QAE/D,OAAO,uBAAuB,GAAG,OAAO,CAAC,KAAK,CAAC,UAAU,CAAA;IAC1D,CAAC,CAAA;AACF,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAC/B,IAAY,EACZ,MAA0B,EAC1B,MAAuB;IAEvB,MAAM,OAAO,GAAG,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,MAAM,CAAC,CAAA;IAErD,IAAI,CAAC,OAAO,IAAI,OAAO,CAAC,SAAS,KAAK,EAAE;QAAE,OAAO,IAAI,CAAA;IAErD,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,IAAI,IAAI,UAAU,CAAC,KAAK,UAAU;QAAE,OAAO,IAAI,CAAA;IAElE,OAAO,OAAO,CAAC,KAAK,CAAA;AACrB,CAAC"}
|
package/out/rules.d.ts
CHANGED
|
@@ -3,80 +3,63 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
* Rule-based
|
|
7
|
-
* and returns a confidence
|
|
8
|
-
*
|
|
9
|
-
* Bitter-lesson-safe: only universal structural patterns — no place-name dictionaries. ~1 small
|
|
10
|
-
* regex set per new locale, not 50K dictionary entries.
|
|
6
|
+
* Rule-based `QueryKind` scorers. Each scorer reads structural patterns in the normalized input and
|
|
7
|
+
* its QueryShape and returns a confidence from 0 to 1.
|
|
11
8
|
*/
|
|
12
9
|
import type { NormalizedInputLite, QueryShapeSegmentsView as QueryShapeLike } from "@mailwoman/query-shape";
|
|
13
10
|
/**
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* Exported so `intent-rules.ts`'s `bare_toponym` can share the exact same ceiling. Sharing it is what makes
|
|
17
|
-
* "bare_toponym is a strict refinement of locality_only" a structural property rather than two numbers that happen to
|
|
18
|
-
* agree today.
|
|
11
|
+
* Maximum length for a bare locality.
|
|
12
|
+
* The `bare_toponym` intent rule uses the same limit.
|
|
19
13
|
*/
|
|
20
14
|
export declare const MAX_LOCALITY_ONLY_LENGTH = 30;
|
|
21
15
|
/**
|
|
22
|
-
*
|
|
23
|
-
* covers all locale variants (US "PO Box 123", FR "BP 42", etc.).
|
|
16
|
+
* Scores input in which QueryShape detected a PO Box format.
|
|
24
17
|
*/
|
|
25
18
|
export declare function scorePoBox(_input: NormalizedInputLite, shape: QueryShapeLike): number;
|
|
26
19
|
/**
|
|
27
|
-
*
|
|
20
|
+
* Scores street-intersection phrasing.
|
|
28
21
|
*/
|
|
29
22
|
export declare function scoreIntersection(input: NormalizedInputLite, _shape: QueryShapeLike): number;
|
|
30
23
|
/**
|
|
31
|
-
*
|
|
32
|
-
* location relative to another place.
|
|
24
|
+
* Scores input that begins with a relative-landmark phrase.
|
|
33
25
|
*/
|
|
34
26
|
export declare function scoreLandmark(input: NormalizedInputLite, _shape: QueryShapeLike): number;
|
|
35
27
|
/**
|
|
36
|
-
*
|
|
37
|
-
* matters: a trailing comma or doubled separator otherwise yields an empty string that inflates the word count and
|
|
38
|
-
* falsifies every-word predicates like the proper-case check below.
|
|
28
|
+
* Splits text on whitespace and commas and drops empty tokens.
|
|
39
29
|
*/
|
|
40
30
|
export declare function wordsOf(text: string): string[];
|
|
41
31
|
/**
|
|
42
|
-
*
|
|
43
|
-
* `@mailwoman/codex`, minus its curated name-prone canonicals (PARK, FIELD, HILL, LAKE, …). Those double as ordinary
|
|
44
|
-
* proper-name heads ("Wrigley Field", "Menlo Park"), and disqualifying on them would reject the very venue and place
|
|
45
|
-
* names the rules below exist to capture. Shared with `intent-rules.ts` so both rule sets read one definition.
|
|
32
|
+
* Reports whether a word is a USPS street suffix that rarely appears in place names.
|
|
46
33
|
*/
|
|
47
34
|
export declare function isDisqualifyingStreetSuffix(word: string): boolean;
|
|
48
35
|
/**
|
|
49
|
-
*
|
|
50
|
-
|
|
36
|
+
* Removes postcode spans in the final segment and collapses the remaining separators.
|
|
37
|
+
*/
|
|
38
|
+
export declare function withoutPostcodeSpans(text: string, shape: QueryShapeLike): string;
|
|
39
|
+
/**
|
|
40
|
+
* Reports whether text contains a Unicode letter.
|
|
51
41
|
*
|
|
52
|
-
*
|
|
53
|
-
|
|
42
|
+
* By itself, the `alpha` character class can match punctuation-only input.
|
|
43
|
+
*/
|
|
44
|
+
export declare function carriesLetter(text: string): boolean;
|
|
45
|
+
/**
|
|
46
|
+
* Scores short capitalized venue names that have no street suffix or recognized format.
|
|
54
47
|
*/
|
|
55
48
|
export declare function scoreVenueLandmark(input: NormalizedInputLite, shape: QueryShapeLike): number;
|
|
56
49
|
/**
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* The "covering most of it" check is what distinguishes `"10118"` (postcode-only) from `"350 5th Ave 10118"`
|
|
60
|
-
* (structured-address with a postcode in it).
|
|
50
|
+
* Scores short input in which a postcode covers most of the text.
|
|
61
51
|
*/
|
|
62
52
|
export declare function scorePostcodeOnly(input: NormalizedInputLite, shape: QueryShapeLike): number;
|
|
63
53
|
/**
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
* Examples: `"Paris"`, `"NYC NY"`, `"Tokyo"`. Distinguishes from `structured_address` (multiple segments) and `vague`
|
|
67
|
-
* (long or mixed-class).
|
|
54
|
+
* Scores a place name with at most an administrative tail and an optional postcode.
|
|
68
55
|
*/
|
|
69
56
|
export declare function scoreLocalityOnly(input: NormalizedInputLite, shape: QueryShapeLike): number;
|
|
70
57
|
/**
|
|
71
|
-
*
|
|
72
|
-
* mixed-class.
|
|
58
|
+
* Scores multi-component address shapes.
|
|
73
59
|
*/
|
|
74
60
|
export declare function scoreStructuredAddress(input: NormalizedInputLite, shape: QueryShapeLike): number;
|
|
75
61
|
/**
|
|
76
|
-
*
|
|
77
|
-
*
|
|
78
|
-
* Returns a moderate baseline so `vague` always shows up as an alternative, even when other rules dominate. The
|
|
79
|
-
* coordinator decides whether to trust vague as the primary kind.
|
|
62
|
+
* Returns a low constant fallback score for ambiguous input.
|
|
80
63
|
*/
|
|
81
64
|
export declare function scoreVague(_input: NormalizedInputLite, _shape: QueryShapeLike): number;
|
|
82
65
|
//# sourceMappingURL=rules.d.ts.map
|
package/out/rules.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"rules.d.ts","sourceRoot":"","sources":["../lib/rules.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"rules.d.ts","sourceRoot":"","sources":["../lib/rules.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAGH,OAAO,KAAK,EAAE,mBAAmB,EAAE,sBAAsB,IAAI,cAAc,EAAE,MAAM,wBAAwB,CAAA;AAkB3G;;;GAGG;AACH,eAAO,MAAM,wBAAwB,KAAK,CAAA;AAgD1C;;GAEG;AACH,wBAAgB,UAAU,CAAC,MAAM,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAOrF;AAED;;GAEG;AACH,wBAAgB,iBAAiB,CAAC,KAAK,EAAE,mBAAmB,EAAE,MAAM,EAAE,cAAc,GAAG,MAAM,CAQ5F;AAED;;GAEG;AACH,wBAAgB,aAAa,CAAC,KAAK,EAAE,mBAAmB,EAAE,MAAM,EAAE,cAAc,GAAG,MAAM,CAQxF;AAED;;GAEG;AACH,wBAAgB,OAAO,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,EAAE,CAE9C;AAED;;GAEG;AACH,wBAAgB,2BAA2B,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAIjE;AAED;;GAEG;AACH,wBAAgB,oBAAoB,CAAC,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAmBhF;AAED;;;;GAIG;AACH,wBAAgB,aAAa,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAEnD;AA0CD;;GAEG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CA2C5F;AAED;;GAEG;AACH,wBAAgB,iBAAiB,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAY3F;AAED;;GAEG;AACH,wBAAgB,iBAAiB,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAwC3F;AAED;;GAEG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,mBAAmB,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAuBhG;AAED;;GAEG;AACH,wBAAgB,UAAU,CAAC,MAAM,EAAE,mBAAmB,EAAE,MAAM,EAAE,cAAc,GAAG,MAAM,CAEtF"}
|