@mailwoman/phrase-grouper 9.4.0 → 10.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/lib/group.ts +24 -17
- package/lib/index.ts +4 -3
- package/lib/rules/tokens.ts +132 -0
- package/lib/rules.ts +68 -300
- package/lib/types.ts +10 -4
- package/out/group.d.ts +18 -12
- package/out/group.d.ts.map +1 -1
- package/out/group.js +24 -17
- package/out/group.js.map +1 -1
- package/out/index.d.ts +4 -3
- package/out/index.d.ts.map +1 -1
- package/out/index.js +4 -3
- package/out/index.js.map +1 -1
- package/out/rules/tokens.d.ts +83 -0
- package/out/rules/tokens.d.ts.map +1 -0
- package/out/rules/tokens.js +102 -0
- package/out/rules/tokens.js.map +1 -0
- package/out/rules.d.ts +39 -48
- package/out/rules.d.ts.map +1 -1
- package/out/rules.js +35 -272
- package/out/rules.js.map +1 -1
- package/out/types.d.ts +10 -4
- package/out/types.d.ts.map +1 -1
- package/package.json +15 -43
package/README.md
CHANGED
|
@@ -25,7 +25,7 @@ const groups = groupPhrases(normalizedInput, queryShape, localeHint)
|
|
|
25
25
|
| --------------------- | ----------------------------------------------------------- |
|
|
26
26
|
| `street_phrase` | Number + capitalized words, hyphenated street names |
|
|
27
27
|
| `locality_phrase` | Capitalized word sequence after comma, near region/postcode |
|
|
28
|
-
| `venue_phrase` |
|
|
28
|
+
| `venue_phrase` | Capitalized word sequence at the start of a street phrase |
|
|
29
29
|
| `postcode` | Known postcode format (ZIP5, UK outward, etc.) |
|
|
30
30
|
| `region_abbreviation` | US state / CA province / AU state abbreviations |
|
|
31
31
|
| `numeric` | Standalone number (potential house number) |
|
|
@@ -55,7 +55,7 @@ kind-classifier → phrase-grouper → classifier (neural/rule-based) → ...
|
|
|
55
55
|
|
|
56
56
|
## Design
|
|
57
57
|
|
|
58
|
-
- **Boundary discovery
|
|
58
|
+
- **Boundary discovery rather than classification.** The phrase grouper answers "where
|
|
59
59
|
are the coherent units?" — the classifier answers "what _type_ is each unit?"
|
|
60
60
|
This separation makes both problems easier.
|
|
61
61
|
- **Bitter-lesson-safe:** uses only universal structural cues (proximity,
|
|
@@ -70,8 +70,8 @@ kind-classifier → phrase-grouper → classifier (neural/rule-based) → ...
|
|
|
70
70
|
- [`@mailwoman/core`](../core) — pipeline coordinator that consumes phrase groups
|
|
71
71
|
- [`@mailwoman/kind-classifier`](../kind-classifier) — preceding stage
|
|
72
72
|
- [The Knowledge Ladder](https://mailwoman.ai/articles/concepts/the-knowledge-ladder/) — design rationale
|
|
73
|
-
- [Staged Pipeline
|
|
73
|
+
- [Staged Pipeline Interface](https://github.com/sister-software/mailwoman/blob/main/docs/records/plan/reference/STAGES.mdx)
|
|
74
74
|
|
|
75
75
|
## License
|
|
76
76
|
|
|
77
|
-
[AGPL-3.0-only](https://www.gnu.org/licenses/
|
|
77
|
+
[AGPL-3.0-only](https://www.gnu.org/licenses/AGPL-3.0.html)
|
package/lib/group.ts
CHANGED
|
@@ -5,11 +5,12 @@
|
|
|
5
5
|
*
|
|
6
6
|
* `groupPhrases` — Stage 2.7 entry point.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
8
|
+
* The function composes per-kind rules over normalized input and `QueryShape`. Each rule that fires emits
|
|
9
|
+
* one `PhraseProposal`. Proposals can overlap. The reconciler (Stage 5) selects the best non-overlapping
|
|
10
|
+
* subset.
|
|
11
11
|
*
|
|
12
|
-
* See `docs/
|
|
12
|
+
* See `docs/records/site-2026-08/understanding/our-approach/the-knowledge-ladder.mdx` § Phrase grouper
|
|
13
|
+
* for the design rationale.
|
|
13
14
|
* and `phrase-grouper/rules.ts` for per-rule documentation.
|
|
14
15
|
*/
|
|
15
16
|
|
|
@@ -30,9 +31,10 @@ import {
|
|
|
30
31
|
import type { GroupPhrasesOpts } from "#types"
|
|
31
32
|
|
|
32
33
|
/**
|
|
33
|
-
* Walk every QueryShape segment and emit one `tokens-by-segment` list.
|
|
34
|
-
*
|
|
35
|
-
* QueryShape
|
|
34
|
+
* Walk every QueryShape segment and emit one `tokens-by-segment` list.
|
|
35
|
+
*
|
|
36
|
+
* Falls back to treating the whole input as a single segment when QueryShape didn't supply
|
|
37
|
+
* segmentation (e.g. Callers wiring the grouper into a path that bypasses QueryShape).
|
|
36
38
|
*/
|
|
37
39
|
function tokensPerSegment(
|
|
38
40
|
text: string,
|
|
@@ -57,14 +59,18 @@ function tokensPerSegment(
|
|
|
57
59
|
}
|
|
58
60
|
|
|
59
61
|
/**
|
|
60
|
-
* Synchronous, pure rule-based implementation.
|
|
62
|
+
* Synchronous, pure rule-based implementation.
|
|
63
|
+
*
|
|
64
|
+
* The async wrapper matches the pipeline interface.
|
|
61
65
|
*
|
|
62
|
-
* Emits overlapping proposals freely — the consumer (Stage 5 reconcile) picks the best
|
|
63
|
-
* semantic+hierarchical constraints.
|
|
64
|
-
*
|
|
66
|
+
* Emits overlapping proposals freely — the consumer (Stage 5 reconcile) picks the best
|
|
67
|
+
* non-overlapping subset under semantic+hierarchical constraints.
|
|
68
|
+
* Confidence is a [0,1] score per proposal.
|
|
65
69
|
*
|
|
66
|
-
*
|
|
67
|
-
*
|
|
70
|
+
* Relative ordering is what matters more than absolute calibration at v0.5.0.
|
|
71
|
+
*
|
|
72
|
+
* The `_locale` parameter is reserved for future locale-aware rule packs
|
|
73
|
+
* (Japanese postcode/honorific patterns, French preposition-bound localities) — currently unused.
|
|
68
74
|
*/
|
|
69
75
|
export function groupPhrasesSync(
|
|
70
76
|
input: NormalizedInputLite,
|
|
@@ -90,8 +96,8 @@ export function groupPhrasesSync(
|
|
|
90
96
|
proposals.push(...scoreVenuePhrase(tokens, text, isFirst))
|
|
91
97
|
}
|
|
92
98
|
|
|
93
|
-
// Sort: descending confidence, ties broken by span start (left-to-right).
|
|
94
|
-
// can rely on this ordering for top-k selection without re-sorting.
|
|
99
|
+
// Sort: descending confidence, ties broken by span start (left-to-right).
|
|
100
|
+
// Downstream Stage 5 can rely on this ordering for top-k selection without re-sorting.
|
|
95
101
|
proposals.sort((a, b) => {
|
|
96
102
|
if (a.confidence !== b.confidence) return b.confidence - a.confidence
|
|
97
103
|
|
|
@@ -102,8 +108,9 @@ export function groupPhrasesSync(
|
|
|
102
108
|
}
|
|
103
109
|
|
|
104
110
|
/**
|
|
105
|
-
* Async variant matching `RuntimePipelineStages.groupPhrases`.
|
|
106
|
-
*
|
|
111
|
+
* Async variant matching `RuntimePipelineStages.groupPhrases`.
|
|
112
|
+
*
|
|
113
|
+
* Wraps the sync impl so the pipeline coordinator can use it as-is.
|
|
107
114
|
*/
|
|
108
115
|
export async function groupPhrases(
|
|
109
116
|
input: NormalizedInputLite,
|
package/lib/index.ts
CHANGED
|
@@ -13,10 +13,11 @@
|
|
|
13
13
|
*
|
|
14
14
|
* Bitter-lesson-safe: only universal structural cues (proximity, punctuation, capitalization,
|
|
15
15
|
* hyphenation, format-shape repetition) — never place-name dictionaries. v0.5.0 ships the
|
|
16
|
-
* rule-based v1
|
|
16
|
+
* rule-based v1. learned 1-2M-param span proposer reserved for v0.5.1.
|
|
17
17
|
*
|
|
18
|
-
* See `docs/
|
|
19
|
-
* and `docs/
|
|
18
|
+
* See `docs/records/site-2026-08/understanding/our-approach/the-knowledge-ladder.mdx` § Phrase grouper
|
|
19
|
+
* for the design rationale and `docs/records/plan/phases/PHASE_8_v0_5_0_fresh_slate.mdx` § E for the
|
|
20
|
+
* v0.5.0 thread.
|
|
20
21
|
*/
|
|
21
22
|
|
|
22
23
|
export { groupPhrases, groupPhrasesSync } from "#group"
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @file Phrase-grouping token spans.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { US_STATE_NAMES } from "@mailwoman/codex/us/state"
|
|
8
|
+
import { Span } from "@mailwoman/core/tokenization"
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* One token within a segment — absolute offsets into the normalized input.
|
|
12
|
+
*
|
|
13
|
+
* Built by `tokenizeSegment` from a (segment-text, segment-start) pair.
|
|
14
|
+
*/
|
|
15
|
+
export interface SegmentToken {
|
|
16
|
+
body: string
|
|
17
|
+
start: number
|
|
18
|
+
end: number
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const WHITESPACE = /\s+/
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Neutral baseline confidence for phrase proposals when no structural cue
|
|
25
|
+
* (position, length, known suffix/prefix/marker, format-hit) lifts or penalizes the score.
|
|
26
|
+
*
|
|
27
|
+
* Each rule adds bonuses on top of this base (e.g. +0.15 for 2-token locality runs, +0.1 for tail-of-last-segment)
|
|
28
|
+
* and subtracts penalties (e.g. −0.2 for a known US region name that isn't at segment-tail).
|
|
29
|
+
*/
|
|
30
|
+
export const NEUTRAL_PROPOSAL_CONFIDENCE = 0.55
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Single-token US state/territory names, derived from the codex roster.
|
|
34
|
+
*
|
|
35
|
+
* Single-token scope is deliberate: the non-tail region-name penalty below reads one token at a
|
|
36
|
+
* time. and a multi-word name ("New York", "North Carolina") can never match a single token —
|
|
37
|
+
* deriving only the single-token names keeps the set equal to what the check can ever see.
|
|
38
|
+
*/
|
|
39
|
+
export const US_REGION_NAMES: ReadonlySet<string> = new Set(
|
|
40
|
+
US_STATE_NAMES.filter((name) => !name.includes(" ")).map((name) => name.toLowerCase())
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Split a segment body into whitespace-separated tokens.
|
|
45
|
+
*
|
|
46
|
+
* Offsets are absolute into the original input (caller supplies the segment's `start` offset).
|
|
47
|
+
* Deliberately not `@mailwoman/query-shape`'s tokenizer: that one yields code-point
|
|
48
|
+
* class runs for the whole input, while this one handles segment-relative → absolute span math for the proposal spans —
|
|
49
|
+
* the two disagree on what a token boundary is.
|
|
50
|
+
*/
|
|
51
|
+
/**
|
|
52
|
+
* Digit count above which a pure-numeric token stops being unambiguously a house
|
|
53
|
+
* number. 1-4 digits are clearly numeric. 5 and up collide with postcodes,
|
|
54
|
+
* so the proposal is emitted at neutral confidence and the reconciler decides.
|
|
55
|
+
*/
|
|
56
|
+
export const MAX_UNAMBIGUOUS_HOUSE_NUMBER_DIGITS = 4
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Confidence for a pure-numeric token short enough to be unambiguous.
|
|
60
|
+
*/
|
|
61
|
+
export const UNAMBIGUOUS_NUMERIC_CONFIDENCE = 0.95
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Token count at which a run reads as a venue name in its own right rather than a stray pair.
|
|
65
|
+
*/
|
|
66
|
+
export const VENUE_RUN_MIN_TOKENS = 3
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Confidence for a venue run too short to clear {@link VENUE_RUN_MIN_TOKENS}.
|
|
70
|
+
*/
|
|
71
|
+
export const SHORT_VENUE_RUN_CONFIDENCE = 0.5
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Confidence added to a place-name run by its token count.
|
|
75
|
+
*
|
|
76
|
+
* Longer runs are less likely to be a coincidental adjacency, so they warrant more —
|
|
77
|
+
* the curve flattens past four tokens.
|
|
78
|
+
*/
|
|
79
|
+
export const PLACE_RUN_LENGTH_BONUS: ReadonlyMap<number, number> = new Map([
|
|
80
|
+
[2, 0.15],
|
|
81
|
+
[3, 0.12],
|
|
82
|
+
])
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Bonus applied to place-name runs at or beyond {@link VENUE_RUN_MIN_TOKENS} + 1 tokens.
|
|
86
|
+
*/
|
|
87
|
+
export const LONG_PLACE_RUN_BONUS = 0.08
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Penalty for a US region name appearing away from the tail, where it is more likely a locality.
|
|
91
|
+
*/
|
|
92
|
+
export const NON_TAIL_REGION_NAME_PENALTY = 0.2
|
|
93
|
+
|
|
94
|
+
export function tokenizeSegment(segmentBody: string, segmentStart: number): SegmentToken[] {
|
|
95
|
+
const tokens: SegmentToken[] = []
|
|
96
|
+
let i = 0
|
|
97
|
+
|
|
98
|
+
while (i < segmentBody.length) {
|
|
99
|
+
while (i < segmentBody.length && WHITESPACE.test(segmentBody[i]!)) {
|
|
100
|
+
i++
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (i >= segmentBody.length) break
|
|
104
|
+
const start = i
|
|
105
|
+
|
|
106
|
+
while (i < segmentBody.length && !WHITESPACE.test(segmentBody[i]!)) {
|
|
107
|
+
i++
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
tokens.push({
|
|
111
|
+
body: segmentBody.slice(start, i),
|
|
112
|
+
start: segmentStart + start,
|
|
113
|
+
end: segmentStart + i,
|
|
114
|
+
})
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
return tokens
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Build a `Section` (Span instance) from absolute offsets into the original text.
|
|
122
|
+
*/
|
|
123
|
+
export function makeSection(text: string, start: number, end: number): Span {
|
|
124
|
+
return Span.from(text.slice(start, end), { start })
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* True when token body is non-empty digits only.
|
|
129
|
+
*/
|
|
130
|
+
export function isAllDigit(s: string): boolean {
|
|
131
|
+
return s.length > 0 && /^[0-9]+$/.test(s)
|
|
132
|
+
}
|