@rebasepro/types 0.13.0 → 0.13.1-canary.g249daa1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,247 @@
1
+ /**
2
+ * Opt-in full-text search configuration.
3
+ *
4
+ * ## Why this is opt-in
5
+ *
6
+ * Without a `search` block, `.search()` behaves exactly as it always has: an
7
+ * `ILIKE '%term%'` OR-ed across the collection's top-level, non-enum `string`
8
+ * properties. That default is unchanged and will stay unchanged — declaring
9
+ * this block is the only way to get anything else.
10
+ *
11
+ * The default has three limits that no amount of tuning inside it can fix:
12
+ * it cannot reach inside `map` (JSONB) or `array` properties, it has no notion
13
+ * of relevance, and a leading `%` means it can never use an index. Collections
14
+ * that outgrow those limits declare what they want searched; collections that
15
+ * have not are left completely alone.
16
+ *
17
+ * ## What declaring it does
18
+ *
19
+ * One `tsvector` column, `GENERATED ALWAYS AS … STORED`, plus one GIN index on
20
+ * it. Postgres recomputes the column on every write of a source field, so it
21
+ * cannot drift from the row, and refuses any attempt to write it directly.
22
+ * `.search()` then compiles to `@@ websearch_to_tsquery(…)` against that
23
+ * column, which stems, drops stopwords, AND-es the terms, and ranks.
24
+ *
25
+ * These are stated consequences, not hidden ones: the column and the index
26
+ * appear in generated DDL, in `schema.generated.ts`, and in `rebase db push`
27
+ * output like any other declared object.
28
+ *
29
+ * @example
30
+ * ```ts
31
+ * const talents: PostgresCollectionConfig = {
32
+ * slug: "talents",
33
+ * table: "talents",
34
+ * properties: { … },
35
+ * search: {
36
+ * language: "spanish",
37
+ * unaccent: true,
38
+ * fields: [
39
+ * { path: "full_name", weight: "A" },
40
+ * "location",
41
+ * "questionnaire.certifications" // into the JSONB
42
+ * ]
43
+ * }
44
+ * };
45
+ * ```
46
+ *
47
+ * @group Search
48
+ */
49
+ export interface SearchConfig {
50
+ /**
51
+ * The fields to index, in the author's own words. Nothing is inferred: a
52
+ * field is searched if and only if it is named here.
53
+ *
54
+ * A bare string is shorthand for `{ path, weight: "B" }`.
55
+ *
56
+ * A path may address:
57
+ * - a top-level `string` property — `"full_name"`
58
+ * - a `string[]` property — `"tags"` (every element is indexed)
59
+ * - a path into a `map` property — `"questionnaire.certifications"`,
60
+ * which indexes every string found at or below that point, including
61
+ * nested objects and arrays of strings. JSON *keys* are never indexed,
62
+ * only values.
63
+ *
64
+ * A path that does not resolve to one of those is a boot-time error, not
65
+ * a silent omission — a search field you believe is live and is not is the
66
+ * failure this whole block exists to prevent.
67
+ */
68
+ fields: readonly (string | SearchField)[];
69
+
70
+ /**
71
+ * The Postgres text search configuration, which decides stemming and
72
+ * stopwords. `"spanish"` stems `auditores` to `auditor` and drops `de`;
73
+ * `"simple"` does neither.
74
+ *
75
+ * Defaults to `"simple"`, which is the only choice that is never wrong:
76
+ * a stemmer applied to the wrong language silently mangles lexemes. Set it
77
+ * to your content's language to get stemming.
78
+ *
79
+ * @default "simple"
80
+ */
81
+ language?: string;
82
+
83
+ /**
84
+ * Fold accents before indexing, so `auditoria` matches `auditoría`.
85
+ *
86
+ * This is not cosmetic in accented languages. Postgres stems the two
87
+ * spellings to *different* lexemes — `to_tsvector('spanish', 'auditoría')`
88
+ * yields `auditor` while `'auditoria'` yields `auditori` — so without this
89
+ * a query typed without accents misses the rows that carry them, which is
90
+ * most queries most users type.
91
+ *
92
+ * Requires the `unaccent` extension. Boot fails with an explicit message if
93
+ * it is not installed and cannot be created, rather than quietly indexing
94
+ * accented text as-is.
95
+ *
96
+ * @default false
97
+ */
98
+ unaccent?: boolean;
99
+
100
+ /**
101
+ * Name of the generated column holding the `tsvector`.
102
+ *
103
+ * Only change this if `search_vector` collides with a column you already
104
+ * have. It is part of your schema once created: renaming it later is a
105
+ * column drop and recreate, which rewrites the table.
106
+ *
107
+ * @default "search_vector"
108
+ */
109
+ column?: string;
110
+
111
+ /**
112
+ * Also match on trigram similarity, so near-misses and typos still rank —
113
+ * `iso14000` reaching `ISO 14001`, which no amount of stemming will do
114
+ * because they are simply different lexemes.
115
+ *
116
+ * Adds a second generated `text` column and a GIN trigram index alongside
117
+ * the `tsvector`, and requires the `pg_trgm` extension. Costs write time
118
+ * and disk; buys the single most common class of failed search.
119
+ *
120
+ * Also changes what `_score` means: the trigram similarity is added to
121
+ * `ts_rank`. It has to be. A typo matches nothing on the exact path, so
122
+ * every row this finds has a `ts_rank` of zero — ranking by that alone
123
+ * would order the results arbitrarily, which is the failure `fuzzy` exists
124
+ * to fix.
125
+ *
126
+ * @default false
127
+ */
128
+ fuzzy?: boolean;
129
+
130
+ /**
131
+ * Similarity floor for {@link SearchConfig.fuzzy}, between 0 and 1. A row
132
+ * whose trigram similarity to the query falls below this never matches on
133
+ * the fuzzy path (it can still match on the exact one).
134
+ *
135
+ * Lower admits more typos and more noise. Ignored unless `fuzzy` is set.
136
+ *
137
+ * @default 0.3
138
+ */
139
+ fuzzyThreshold?: number;
140
+ }
141
+
142
+ /**
143
+ * One indexed field, with the weight it carries in the ranking.
144
+ *
145
+ * @group Search
146
+ */
147
+ export interface SearchField {
148
+ /**
149
+ * Property name, or dotted path into a `map` property.
150
+ * @see SearchConfig.fields
151
+ */
152
+ path: string;
153
+
154
+ /**
155
+ * Postgres weight class. `ts_rank` scores an `A` hit far above a `D` hit,
156
+ * which is how a name outranks a passing mention in a long description.
157
+ *
158
+ * The four classes are Postgres's own and there are exactly four.
159
+ *
160
+ * @default "B"
161
+ */
162
+ weight?: SearchWeight;
163
+ }
164
+
165
+ /**
166
+ * Postgres tsvector weight classes, strongest to weakest.
167
+ *
168
+ * @group Search
169
+ */
170
+ export type SearchWeight = "A" | "B" | "C" | "D";
171
+
172
+ /** The column name used when {@link SearchConfig.column} is not given. */
173
+ export const DEFAULT_SEARCH_COLUMN = "search_vector";
174
+
175
+ /** The text search configuration used when {@link SearchConfig.language} is not given. */
176
+ export const DEFAULT_SEARCH_LANGUAGE = "simple";
177
+
178
+ /** The weight a field carries when it does not name one. */
179
+ export const DEFAULT_SEARCH_WEIGHT: SearchWeight = "B";
180
+
181
+ /** The similarity floor used when {@link SearchConfig.fuzzyThreshold} is not given. */
182
+ export const DEFAULT_FUZZY_THRESHOLD = 0.3;
183
+
184
+ /**
185
+ * Sort keys a query computes rather than reads from a column.
186
+ *
187
+ * `orderBy` is otherwise typed against the row — `keyof M` — which is exactly
188
+ * right for a column and exactly wrong for relevance: `_score` is produced by
189
+ * the query, so it appears in no generated row type and a project with a
190
+ * generated SDK could not name it. The runtime accepted it, the docs told
191
+ * people to use it, and the types rejected it.
192
+ *
193
+ * Kept as a named union rather than a loose `string` so the other half of the
194
+ * guarantee survives: a typo'd column is still a compile error, and remains a
195
+ * 400 at runtime rather than a silently unsorted list.
196
+ *
197
+ * `_distance` is deliberately not here. A vector search orders by distance on
198
+ * its own and overrides `orderBy` outright, so naming it would imply a choice
199
+ * the caller does not have.
200
+ *
201
+ * @group Search
202
+ */
203
+ export type ComputedSortField = typeof RELEVANCE_SORT_FIELD;
204
+
205
+ /**
206
+ * The relevance sort key. Valid only on a collection that declares a
207
+ * {@link SearchConfig} *and* on a query that carries a search string; anywhere
208
+ * else it is an unknown field and the request is refused.
209
+ */
210
+ export const RELEVANCE_SORT_FIELD = "_score";
211
+
212
+ /**
213
+ * One field that matched, and the text around the hit.
214
+ *
215
+ * Returned per row as `_matches` when a query asks for it — see the `explain`
216
+ * option on `.search()`. Answers the question a ranked list otherwise leaves
217
+ * open: *why is this row here?* A candidate surfacing for "iso 14001" because
218
+ * of a certification is a different result from one surfacing because the
219
+ * string appears in a paragraph about something else, and the score alone
220
+ * cannot tell them apart.
221
+ *
222
+ * @group Search
223
+ */
224
+ export interface SearchMatch {
225
+ /**
226
+ * The declared field path that matched, exactly as written in
227
+ * {@link SearchConfig.fields} — e.g. `"questionnaire.certifications"`.
228
+ * Map it to a label for display; the path is stable, a label is yours.
229
+ */
230
+ field: string;
231
+
232
+ /**
233
+ * The matching text, with each hit wrapped in `<mark>…</mark>`.
234
+ *
235
+ * Built by Postgres's `ts_headline` over the same normalized text that was
236
+ * indexed. With {@link SearchConfig.unaccent} on that means the snippet
237
+ * reads with accents folded — `Auditoria` rather than `Auditoría`. That is
238
+ * deliberate: `ts_headline` over the *original* text cannot find a hit the
239
+ * unaccented query produced, so it returns the text with nothing marked at
240
+ * all. A readable snippet that highlights beats a prettier one that
241
+ * silently does not.
242
+ *
243
+ * Contains markup by construction. Render it as HTML or strip the tags —
244
+ * do not display it raw, and do not trust it as plain text.
245
+ */
246
+ snippet: string;
247
+ }