@agent-native/core 0.200.0-nightly-20261002124406 → 0.200.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/production-agent.d.ts +5 -1
- package/dist/agent/production-agent.js +36 -1
- package/dist/app-config/runtime.d.ts +1 -0
- package/dist/app-config/runtime.js +4 -0
- package/dist/app-config/schema.d.ts +1 -0
- package/dist/cli/design-connect.d.ts +1 -0
- package/dist/cli/design-connect.js +147 -7
- package/dist/client/analytics.js +24 -1
- package/dist/client/session-replay.d.ts +6 -0
- package/dist/client/session-replay.js +47 -5
- package/dist/collab/awareness.d.ts +2 -2
- package/dist/collab/routes.d.ts +1 -1
- package/dist/file-upload/builder.js +20 -2
- package/dist/mcp-client/app-api.d.ts +11 -1
- package/dist/mcp-client/app-api.js +17 -3
- package/dist/mcp-client/index.d.ts +1 -1
- package/dist/mcp-client/manager.d.ts +3 -1
- package/dist/mcp-client/manager.js +4 -0
- package/dist/notifications/routes.d.ts +3 -3
- package/dist/observability/metrics.js +14 -7
- package/dist/provider-api/actions/custom-provider-registration.d.ts +6 -6
- package/dist/provider-api/actions/provider-api.d.ts +15 -15
- package/dist/resource-changes/store.d.ts +166 -0
- package/dist/resource-changes/store.js +499 -0
- package/dist/search/index-store.d.ts +9 -0
- package/dist/search/index-store.js +50 -0
- package/dist/search/index.d.ts +5 -1
- package/dist/search/index.js +5 -0
- package/dist/search/indexer.d.ts +32 -0
- package/dist/search/indexer.js +651 -0
- package/dist/search/query-parser.d.ts +29 -0
- package/dist/search/query-parser.js +71 -0
- package/dist/search/query.d.ts +46 -0
- package/dist/search/query.js +149 -0
- package/dist/search/registry.d.ts +55 -0
- package/dist/search/registry.js +93 -0
- package/dist/search/tokenize.d.ts +103 -0
- package/dist/search/tokenize.js +367 -0
- package/dist/server/action-change-marker-write.js +5 -1
- package/dist/server/agent-chat/run-code-tools.d.ts +7 -0
- package/dist/server/agent-chat/run-code-tools.js +7 -0
- package/dist/server/agent-chat-plugin.d.ts +5 -1
- package/dist/server/agent-chat-plugin.js +18 -12
- package/dist/server/agent-engine-api-key-route.d.ts +1 -1
- package/dist/server/agent-engine-default-model-route.d.ts +2 -2
- package/dist/server/release-schema.js +8 -0
- package/dist/triggers/actions/manage-automation.d.ts +3 -3
- package/package.json +6 -9
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
export function parseSearchQuery(input) {
|
|
2
|
+
const groups = [];
|
|
3
|
+
const negatives = [];
|
|
4
|
+
let pendingOr = false;
|
|
5
|
+
let i = 0;
|
|
6
|
+
const pushTerm = (term) => {
|
|
7
|
+
if (!/[\p{L}\p{N}]/u.test(term.text)) {
|
|
8
|
+
pendingOr = false;
|
|
9
|
+
return;
|
|
10
|
+
}
|
|
11
|
+
if (term.negated) {
|
|
12
|
+
negatives.push(term);
|
|
13
|
+
pendingOr = false;
|
|
14
|
+
return;
|
|
15
|
+
}
|
|
16
|
+
if (pendingOr && groups.length > 0) {
|
|
17
|
+
groups[groups.length - 1].terms.push(term);
|
|
18
|
+
}
|
|
19
|
+
else {
|
|
20
|
+
groups.push({ terms: [term] });
|
|
21
|
+
}
|
|
22
|
+
pendingOr = false;
|
|
23
|
+
};
|
|
24
|
+
const trimmed = input.trim();
|
|
25
|
+
const length = trimmed.length;
|
|
26
|
+
while (i < length) {
|
|
27
|
+
if (/\s/.test(trimmed[i])) {
|
|
28
|
+
i += 1;
|
|
29
|
+
continue;
|
|
30
|
+
}
|
|
31
|
+
let negated = false;
|
|
32
|
+
if (trimmed[i] === "-" && trimmed[i + 1] && !/\s/.test(trimmed[i + 1])) {
|
|
33
|
+
negated = true;
|
|
34
|
+
i += 1;
|
|
35
|
+
}
|
|
36
|
+
let titleOnly = false;
|
|
37
|
+
if (trimmed.startsWith("intitle:", i)) {
|
|
38
|
+
titleOnly = true;
|
|
39
|
+
i += "intitle:".length;
|
|
40
|
+
}
|
|
41
|
+
if (i >= length)
|
|
42
|
+
break;
|
|
43
|
+
if (trimmed[i] === '"') {
|
|
44
|
+
const close = trimmed.indexOf('"', i + 1);
|
|
45
|
+
const end = close === -1 ? length : close;
|
|
46
|
+
const text = trimmed.slice(i + 1, end);
|
|
47
|
+
i = close === -1 ? length : end + 1;
|
|
48
|
+
if (text.trim())
|
|
49
|
+
pushTerm({ text: text.trim(), phrase: true, negated, titleOnly });
|
|
50
|
+
continue;
|
|
51
|
+
}
|
|
52
|
+
let j = i;
|
|
53
|
+
while (j < length && !/\s/.test(trimmed[j]))
|
|
54
|
+
j += 1;
|
|
55
|
+
const text = trimmed.slice(i, j);
|
|
56
|
+
i = j;
|
|
57
|
+
if (!negated && !titleOnly && text === "OR" && groups.length > 0) {
|
|
58
|
+
pendingOr = true;
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
pushTerm({ text, phrase: false, negated, titleOnly });
|
|
62
|
+
}
|
|
63
|
+
return {
|
|
64
|
+
groups,
|
|
65
|
+
negatives,
|
|
66
|
+
empty: groups.length === 0 && negatives.length === 0,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
export function searchQueryNeedles(parsed) {
|
|
70
|
+
return parsed.groups.flatMap((group) => group.terms.map((term) => term.text));
|
|
71
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SQL for searching through the core index, as Drizzle fragments an app
|
|
3
|
+
* composes into its own query. The app keeps its own access rule, filters,
|
|
4
|
+
* paging, and output; the index supplies matching and ranking.
|
|
5
|
+
*
|
|
6
|
+
* Matching, for each term:
|
|
7
|
+
* - titles and summaries match anywhere, including mid-word ("prio" finds
|
|
8
|
+
* "Task Priorities"), because they're short;
|
|
9
|
+
* - bodies match whole words, every word as a prefix, through the GIN index;
|
|
10
|
+
* mid-word body matches are deliberately not supported. A phrase needs
|
|
11
|
+
* its words adjacent, except in a document too long or repetitive for
|
|
12
|
+
* Postgres to keep every word position, where every word being in one
|
|
13
|
+
* field is enough.
|
|
14
|
+
*
|
|
15
|
+
* Ranking reproduces the tiers the browser lane uses
|
|
16
|
+
* (`TITLE_MATCH_TIER` in Content): exact title 5, title prefix 4, title word
|
|
17
|
+
* prefixes 3, title substrings 2, title or summary 1, then how many query
|
|
18
|
+
* groups the title and summary cover, then whether the body contains the
|
|
19
|
+
* query as a phrase. Callers add their own tie-breaks.
|
|
20
|
+
*
|
|
21
|
+
* A query with a term longer than the index can match as a phrase throws
|
|
22
|
+
* `SearchTermTooLongError`; answer it with the app's fallback search.
|
|
23
|
+
*/
|
|
24
|
+
import { type SQL } from "drizzle-orm";
|
|
25
|
+
import type { ParsedSearchQuery } from "./query-parser.js";
|
|
26
|
+
import type { SearchableResourceRegistration } from "./registry.js";
|
|
27
|
+
export interface IndexedSearchOptions {
|
|
28
|
+
registration: SearchableResourceRegistration;
|
|
29
|
+
query: Pick<ParsedSearchQuery, "groups" | "negatives">;
|
|
30
|
+
/** "title" matches titles only. Defaults to title, summary, and body. */
|
|
31
|
+
fields?: "all" | "title";
|
|
32
|
+
}
|
|
33
|
+
export interface IndexedSearchSql {
|
|
34
|
+
/** Join target and condition: `.innerJoin(search.join, search.on)`. */
|
|
35
|
+
join: SQL;
|
|
36
|
+
on: SQL;
|
|
37
|
+
/** Every group matches and no negative does. */
|
|
38
|
+
match: SQL;
|
|
39
|
+
matchTier: SQL<number>;
|
|
40
|
+
titleCoverage: SQL<number>;
|
|
41
|
+
summaryCoverage: SQL<number>;
|
|
42
|
+
bodyPhrase: SQL<number>;
|
|
43
|
+
/** The ranking terms above, best first, leaving out any that are always 0. */
|
|
44
|
+
orderBy: SQL[];
|
|
45
|
+
}
|
|
46
|
+
export declare function indexedSearchSql(options: IndexedSearchOptions): IndexedSearchSql;
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SQL for searching through the core index, as Drizzle fragments an app
|
|
3
|
+
* composes into its own query. The app keeps its own access rule, filters,
|
|
4
|
+
* paging, and output; the index supplies matching and ranking.
|
|
5
|
+
*
|
|
6
|
+
* Matching, for each term:
|
|
7
|
+
* - titles and summaries match anywhere, including mid-word ("prio" finds
|
|
8
|
+
* "Task Priorities"), because they're short;
|
|
9
|
+
* - bodies match whole words, every word as a prefix, through the GIN index;
|
|
10
|
+
* mid-word body matches are deliberately not supported. A phrase needs
|
|
11
|
+
* its words adjacent, except in a document too long or repetitive for
|
|
12
|
+
* Postgres to keep every word position, where every word being in one
|
|
13
|
+
* field is enough.
|
|
14
|
+
*
|
|
15
|
+
* Ranking reproduces the tiers the browser lane uses
|
|
16
|
+
* (`TITLE_MATCH_TIER` in Content): exact title 5, title prefix 4, title word
|
|
17
|
+
* prefixes 3, title substrings 2, title or summary 1, then how many query
|
|
18
|
+
* groups the title and summary cover, then whether the body contains the
|
|
19
|
+
* query as a phrase. Callers add their own tie-breaks.
|
|
20
|
+
*
|
|
21
|
+
* A query with a term longer than the index can match as a phrase throws
|
|
22
|
+
* `SearchTermTooLongError`; answer it with the app's fallback search.
|
|
23
|
+
*/
|
|
24
|
+
import { and, not, or, sql } from "drizzle-orm";
|
|
25
|
+
import { SEARCH_RESOURCES_TABLE } from "./index-store.js";
|
|
26
|
+
import { anyOfTsquery, isPhraseTerm, normalizeSearchText, termTsquery, } from "./tokenize.js";
|
|
27
|
+
const ALIAS = "search_index";
|
|
28
|
+
/** The indexer's weights for title, summary, and body. */
|
|
29
|
+
const FIELD_WEIGHTS = ["A", "B", "C"];
|
|
30
|
+
function escapeLike(value) {
|
|
31
|
+
return value.replace(/([\\%_])/g, "\\$1");
|
|
32
|
+
}
|
|
33
|
+
function escapeRegex(value) {
|
|
34
|
+
return value.replace(/[\\^$.*+?()[\]{}|]/g, "\\$&");
|
|
35
|
+
}
|
|
36
|
+
function contains(column, needle) {
|
|
37
|
+
return sql `${column} LIKE ${`%${escapeLike(needle)}%`} ESCAPE '\\'`;
|
|
38
|
+
}
|
|
39
|
+
function startsWith(column, needle) {
|
|
40
|
+
return sql `${column} LIKE ${`${escapeLike(needle)}%`} ESCAPE '\\'`;
|
|
41
|
+
}
|
|
42
|
+
function wordPrefix(column, needle) {
|
|
43
|
+
const pattern = escapeRegex(needle).replace(/\s+/g, "[[:space:]]+");
|
|
44
|
+
return sql `${column} ~* ${`(^|[^[:alnum:]_])${pattern}`}`;
|
|
45
|
+
}
|
|
46
|
+
function sumOf(conditions) {
|
|
47
|
+
if (!conditions.length)
|
|
48
|
+
return null;
|
|
49
|
+
return sql `(${sql.join(conditions.map((condition) => sql `case when ${condition} then 1 else 0 end`), sql ` + `)})`;
|
|
50
|
+
}
|
|
51
|
+
export function indexedSearchSql(options) {
|
|
52
|
+
const { registration, query } = options;
|
|
53
|
+
// SQL is built per call, never at import: apps' tests stub drizzle-orm and
|
|
54
|
+
// still import core search.
|
|
55
|
+
const titleNorm = sql.raw(`${ALIAS}.title_norm`);
|
|
56
|
+
const summaryNorm = sql.raw(`${ALIAS}.summary_norm`);
|
|
57
|
+
// A bare constant in ORDER BY is read as a column position, so ranking
|
|
58
|
+
// terms that can only be zero are left out of the order rather than
|
|
59
|
+
// written as 0.
|
|
60
|
+
const zero = sql `0`;
|
|
61
|
+
const titleOnlyField = options.fields === "title";
|
|
62
|
+
const needle = (term) => normalizeSearchText(term.text);
|
|
63
|
+
const bodyTerms = (terms) => titleOnlyField ? [] : terms.filter((term) => !term.titleOnly);
|
|
64
|
+
const bodyMatch = (terms) => {
|
|
65
|
+
const tsquery = anyOfTsquery(terms.map((term) => termTsquery(term.text, { prefix: true })));
|
|
66
|
+
if (!tsquery)
|
|
67
|
+
return undefined;
|
|
68
|
+
// A document whose vector lost positions can't be phrase-matched
|
|
69
|
+
// exactly, so a phrase matches it when every word is in one field.
|
|
70
|
+
const anyOrder = anyOfTsquery(terms
|
|
71
|
+
.filter((term) => isPhraseTerm(term.text))
|
|
72
|
+
.flatMap((term) => FIELD_WEIGHTS.map((weights) => termTsquery(term.text, { prefix: true, anyOrder: true, weights }))));
|
|
73
|
+
const vectorMatch = anyOrder
|
|
74
|
+
? sql `(doc_vector @@ ${tsquery}::tsquery OR (NOT positions_complete AND doc_vector @@ ${anyOrder}::tsquery))`
|
|
75
|
+
: sql `doc_vector @@ ${tsquery}::tsquery`;
|
|
76
|
+
return sql `${sql.raw(`${ALIAS}.resource_id`)} IN (SELECT resource_id FROM ${sql.raw(SEARCH_RESOURCES_TABLE)} WHERE app = ${registration.app} AND resource_type = ${registration.type} AND ${vectorMatch})`;
|
|
77
|
+
};
|
|
78
|
+
const titleAny = (terms) => or(...terms.map((term) => contains(titleNorm, needle(term))));
|
|
79
|
+
const summaryAny = (terms) => {
|
|
80
|
+
const eligible = bodyTerms(terms);
|
|
81
|
+
return eligible.length
|
|
82
|
+
? or(...eligible.map((term) => contains(summaryNorm, needle(term))))
|
|
83
|
+
: undefined;
|
|
84
|
+
};
|
|
85
|
+
const anyField = (terms) => or(titleAny(terms), summaryAny(terms), bodyMatch(bodyTerms(terms)));
|
|
86
|
+
const groups = query.groups.filter((group) => group.terms.length > 0);
|
|
87
|
+
const match = groups.length || query.negatives.length
|
|
88
|
+
? and(...groups.map((group) => anyField(group.terms)), ...query.negatives.map((negative) => not(anyField([negative]))))
|
|
89
|
+
: sql `false`;
|
|
90
|
+
const simpleQueries = groups.length > 0 && groups.every((group) => group.terms.length === 1)
|
|
91
|
+
? [groups.map((group) => group.terms[0].text.trim()).join(" ")]
|
|
92
|
+
: groups.length === 1
|
|
93
|
+
? groups[0].terms.map((term) => term.text.trim())
|
|
94
|
+
: [];
|
|
95
|
+
const normalizedSimpleQueries = simpleQueries
|
|
96
|
+
.map(normalizeSearchText)
|
|
97
|
+
.filter(Boolean);
|
|
98
|
+
const guardTitle = groups.every((group) => group.terms.every((term) => !/\s/.test(term.text)));
|
|
99
|
+
const allTitleSubstrings = groups.length
|
|
100
|
+
? and(...groups.map((group) => titleAny(group.terms)))
|
|
101
|
+
: sql `false`;
|
|
102
|
+
const allTitleWordPrefixes = groups.length
|
|
103
|
+
? and(...groups.map((group) => or(...group.terms.map((term) => wordPrefix(titleNorm, needle(term))))))
|
|
104
|
+
: sql `false`;
|
|
105
|
+
const allTitleOrSummary = groups.length
|
|
106
|
+
? and(...groups.map((group) => or(titleAny(group.terms), summaryAny(group.terms))))
|
|
107
|
+
: sql `false`;
|
|
108
|
+
const simpleTier = (compare) => normalizedSimpleQueries.length
|
|
109
|
+
? and(guardTitle ? allTitleSubstrings : sql `true`, or(...normalizedSimpleQueries.map(compare)))
|
|
110
|
+
: sql `false`;
|
|
111
|
+
const matchTier = sql `case
|
|
112
|
+
when ${simpleTier((simple) => sql `${titleNorm} = ${simple}`)} then 5
|
|
113
|
+
when ${simpleTier((simple) => startsWith(titleNorm, simple))} then 4
|
|
114
|
+
when ${allTitleSubstrings} and ${allTitleWordPrefixes} then 3
|
|
115
|
+
when ${allTitleSubstrings} then 2
|
|
116
|
+
when ${allTitleOrSummary} then 1
|
|
117
|
+
else 0
|
|
118
|
+
end`;
|
|
119
|
+
const titleCoverage = sumOf(groups.map((group) => titleAny(group.terms)));
|
|
120
|
+
const summaryCoverage = sumOf(groups.flatMap((group) => {
|
|
121
|
+
const condition = summaryAny(group.terms);
|
|
122
|
+
return condition ? [condition] : [];
|
|
123
|
+
}));
|
|
124
|
+
// The whole query as a phrase in the body, when it is two or more plain
|
|
125
|
+
// words, one per group.
|
|
126
|
+
const phraseTerms = !titleOnlyField &&
|
|
127
|
+
groups.length >= 2 &&
|
|
128
|
+
groups.every((group) => group.terms.length === 1 && !group.terms[0].titleOnly)
|
|
129
|
+
? groups.map((group) => group.terms[0].text.trim()).join(" ")
|
|
130
|
+
: null;
|
|
131
|
+
const phraseQuery = phraseTerms
|
|
132
|
+
? termTsquery(phraseTerms, { prefix: true, weights: "C" })
|
|
133
|
+
: null;
|
|
134
|
+
const bodyPhrase = phraseQuery
|
|
135
|
+
? sql `case when ${sql.raw(`${ALIAS}.doc_vector`)} @@ ${phraseQuery}::tsquery then 1 else 0 end`
|
|
136
|
+
: null;
|
|
137
|
+
return {
|
|
138
|
+
join: sql `${sql.raw(SEARCH_RESOURCES_TABLE)} AS ${sql.raw(ALIAS)}`,
|
|
139
|
+
on: sql `${sql.raw(`${ALIAS}.app`)} = ${registration.app} AND ${sql.raw(`${ALIAS}.resource_type`)} = ${registration.type} AND ${sql.raw(`${ALIAS}.resource_id`)} = ${registration.idColumn}::text`,
|
|
140
|
+
match,
|
|
141
|
+
matchTier,
|
|
142
|
+
titleCoverage: titleCoverage ?? zero,
|
|
143
|
+
summaryCoverage: summaryCoverage ?? zero,
|
|
144
|
+
bodyPhrase: bodyPhrase ?? zero,
|
|
145
|
+
orderBy: [matchTier, titleCoverage, summaryCoverage, bodyPhrase]
|
|
146
|
+
.filter((term) => term !== null)
|
|
147
|
+
.map((term) => sql `${term} desc`),
|
|
148
|
+
};
|
|
149
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import { type AnyColumn, type Table } from "drizzle-orm";
|
|
2
|
+
import { type MigrationEntry } from "../db/migrations.js";
|
|
3
|
+
import { type ResourceChangeSource } from "../resource-changes/store.js";
|
|
4
|
+
/** The change-feed consumer name the search index subscribes as. */
|
|
5
|
+
export declare const SEARCH_CHANGE_CONSUMER = "search";
|
|
6
|
+
/** What one resource contributes to the index. */
|
|
7
|
+
export interface SearchableResourceDocument {
|
|
8
|
+
id: string;
|
|
9
|
+
title: string;
|
|
10
|
+
/** A short description, ranked between the title and the body. */
|
|
11
|
+
summary?: string | null;
|
|
12
|
+
body?: string | null;
|
|
13
|
+
modifiedAt?: string | Date | null;
|
|
14
|
+
}
|
|
15
|
+
export interface SearchableResourceRegistration {
|
|
16
|
+
/** The app that owns the table, e.g. "content". */
|
|
17
|
+
app: string;
|
|
18
|
+
/** Matches the shareable resource type, e.g. "document". */
|
|
19
|
+
type: string;
|
|
20
|
+
/** The Drizzle table whose rows are indexed. */
|
|
21
|
+
table: Table;
|
|
22
|
+
/** Its primary key column. */
|
|
23
|
+
idColumn: AnyColumn;
|
|
24
|
+
/**
|
|
25
|
+
* Bump when `load` changes what it returns. A higher version rebuilds the
|
|
26
|
+
* index; searches use the app's fallback until the rebuild completes.
|
|
27
|
+
*/
|
|
28
|
+
version: number;
|
|
29
|
+
/**
|
|
30
|
+
* Loads the searchable text for a batch of ids. Omit an id whose row no
|
|
31
|
+
* longer exists and it is removed from the index. Index every row the app
|
|
32
|
+
* might search; access and filters are checked live at query time.
|
|
33
|
+
*/
|
|
34
|
+
load(ids: string[]): Promise<SearchableResourceDocument[]>;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Makes an app table searchable through the core index. Pair it with
|
|
38
|
+
* `searchIndexMigration()` in the app's `runMigrations` list, which installs
|
|
39
|
+
* the triggers that keep the index fresh.
|
|
40
|
+
*/
|
|
41
|
+
export declare function registerSearchableResource(registration: SearchableResourceRegistration): SearchableResourceRegistration;
|
|
42
|
+
export declare function getSearchableResource(app: string, type: string): SearchableResourceRegistration | undefined;
|
|
43
|
+
export declare function listSearchableResources(): SearchableResourceRegistration[];
|
|
44
|
+
/** Removes a registration. Tests use it to isolate cases. */
|
|
45
|
+
export declare function unregisterSearchableResource(app: string, type: string): void;
|
|
46
|
+
export declare function searchableResourceSource(registration: SearchableResourceRegistration): ResourceChangeSource;
|
|
47
|
+
/**
|
|
48
|
+
* A named migration that creates the search tables and installs change
|
|
49
|
+
* capture on the registration's table. Add it to the app's `runMigrations`
|
|
50
|
+
* list with the app's next version number.
|
|
51
|
+
*/
|
|
52
|
+
export declare function searchIndexMigration(registration: SearchableResourceRegistration, entry: {
|
|
53
|
+
version: number;
|
|
54
|
+
name: string;
|
|
55
|
+
}): MigrationEntry;
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import { getTableName } from "drizzle-orm";
|
|
2
|
+
import { deferMigration } from "../db/migrations.js";
|
|
3
|
+
import { registerRecurringSweepHandler } from "../jobs/sweep-hooks.js";
|
|
4
|
+
import { assertResourceKey, installResourceChangeCapture, registerAfterWriteDrain, resourceChangeTriggerNames, } from "../resource-changes/store.js";
|
|
5
|
+
import { ensureSearchIndexTables } from "./index-store.js";
|
|
6
|
+
/** The change-feed consumer name the search index subscribes as. */
|
|
7
|
+
export const SEARCH_CHANGE_CONSUMER = "search";
|
|
8
|
+
const registrations = new Map();
|
|
9
|
+
function key(app, type) {
|
|
10
|
+
return `${app}:${type}`;
|
|
11
|
+
}
|
|
12
|
+
/** Budget for the drain a write schedules in its own request. */
|
|
13
|
+
const AFTER_WRITE_DRAIN_MS = 250;
|
|
14
|
+
/** Share of the recurring sweep's budget search may use. */
|
|
15
|
+
const SWEEP_DRAIN_MS = 20_000;
|
|
16
|
+
let hooksRegistered = false;
|
|
17
|
+
function registerDrainHooks() {
|
|
18
|
+
if (hooksRegistered)
|
|
19
|
+
return;
|
|
20
|
+
hooksRegistered = true;
|
|
21
|
+
// Both run only where the database is already awake: right after a write,
|
|
22
|
+
// and inside the per-minute sweep that already queries it. Search never
|
|
23
|
+
// polls on its own schedule.
|
|
24
|
+
registerAfterWriteDrain("search-index", async () => {
|
|
25
|
+
const { drainAllSearchIndexes } = await import("./indexer.js");
|
|
26
|
+
await drainAllSearchIndexes(Date.now() + AFTER_WRITE_DRAIN_MS);
|
|
27
|
+
});
|
|
28
|
+
registerRecurringSweepHandler("search-index", async ({ deadlineAt }) => {
|
|
29
|
+
const { drainAllSearchIndexes } = await import("./indexer.js");
|
|
30
|
+
await drainAllSearchIndexes(Math.min(deadlineAt - 1_000, Date.now() + SWEEP_DRAIN_MS));
|
|
31
|
+
});
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Makes an app table searchable through the core index. Pair it with
|
|
35
|
+
* `searchIndexMigration()` in the app's `runMigrations` list, which installs
|
|
36
|
+
* the triggers that keep the index fresh.
|
|
37
|
+
*/
|
|
38
|
+
export function registerSearchableResource(registration) {
|
|
39
|
+
assertResourceKey(registration.app, "Search app");
|
|
40
|
+
assertResourceKey(registration.type, "Search resource type");
|
|
41
|
+
if (!Number.isInteger(registration.version) || registration.version < 1) {
|
|
42
|
+
throw new Error("Search registration version must be a positive integer.");
|
|
43
|
+
}
|
|
44
|
+
const triggers = resourceChangeTriggerNames(searchableResourceSource(registration)).function;
|
|
45
|
+
for (const other of registrations.values()) {
|
|
46
|
+
if (key(other.app, other.type) === key(registration.app, registration.type))
|
|
47
|
+
continue;
|
|
48
|
+
if (resourceChangeTriggerNames(searchableResourceSource(other)).function ===
|
|
49
|
+
triggers) {
|
|
50
|
+
throw new Error(`Search registrations ${other.app}/${other.type} and ${registration.app}/${registration.type} would share change-capture trigger names. Rename one resource type.`);
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
registrations.set(key(registration.app, registration.type), registration);
|
|
54
|
+
registerDrainHooks();
|
|
55
|
+
return registration;
|
|
56
|
+
}
|
|
57
|
+
export function getSearchableResource(app, type) {
|
|
58
|
+
return registrations.get(key(app, type));
|
|
59
|
+
}
|
|
60
|
+
export function listSearchableResources() {
|
|
61
|
+
return [...registrations.values()];
|
|
62
|
+
}
|
|
63
|
+
/** Removes a registration. Tests use it to isolate cases. */
|
|
64
|
+
export function unregisterSearchableResource(app, type) {
|
|
65
|
+
registrations.delete(key(app, type));
|
|
66
|
+
}
|
|
67
|
+
export function searchableResourceSource(registration) {
|
|
68
|
+
return {
|
|
69
|
+
app: registration.app,
|
|
70
|
+
resourceType: registration.type,
|
|
71
|
+
table: getTableName(registration.table),
|
|
72
|
+
idColumn: registration.idColumn.name,
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* A named migration that creates the search tables and installs change
|
|
77
|
+
* capture on the registration's table. Add it to the app's `runMigrations`
|
|
78
|
+
* list with the app's next version number.
|
|
79
|
+
*/
|
|
80
|
+
export function searchIndexMigration(registration, entry) {
|
|
81
|
+
return {
|
|
82
|
+
version: entry.version,
|
|
83
|
+
name: entry.name,
|
|
84
|
+
sql: {},
|
|
85
|
+
run: async (exec) => {
|
|
86
|
+
await ensureSearchIndexTables(exec);
|
|
87
|
+
const installed = await installResourceChangeCapture(exec, searchableResourceSource(registration), SEARCH_CHANGE_CONSUMER);
|
|
88
|
+
// A busy table: try again on the next boot rather than block writes.
|
|
89
|
+
if (!installed)
|
|
90
|
+
return deferMigration();
|
|
91
|
+
},
|
|
92
|
+
};
|
|
93
|
+
}
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Search tokens, computed in JavaScript for both the index and the query.
|
|
3
|
+
*
|
|
4
|
+
* Postgres's text parser depends on the database's locale: on some locales it
|
|
5
|
+
* drops Japanese text entirely, and PGlite and Neon don't agree. So core never
|
|
6
|
+
* asks Postgres to tokenize. It builds `tsvector` and `tsquery` literals
|
|
7
|
+
* itself, and the database only stores, indexes, and matches them. The same
|
|
8
|
+
* code tokenizes documents and queries, so they always agree.
|
|
9
|
+
*
|
|
10
|
+
* Rules:
|
|
11
|
+
* - Text is NFKC-normalized and lowercased. There is no stemming and there are
|
|
12
|
+
* no stopwords.
|
|
13
|
+
* - A word is a run of letters, numbers, and combining marks. Everything else
|
|
14
|
+
* separates words, so `snake_case`, `kebab-case`, URLs, and paths become
|
|
15
|
+
* their parts at consecutive positions, and a query for the same text
|
|
16
|
+
* matches them as a phrase.
|
|
17
|
+
* - A camelCase or PascalCase word is indexed whole and as its parts, so
|
|
18
|
+
* "camelCase", "camel", and "case camel" all find it.
|
|
19
|
+
* - A word longer than Postgres's 2,046-byte limit keeps its longest prefix
|
|
20
|
+
* that fits.
|
|
21
|
+
* - Chinese, Japanese, and Korean runs become overlapping character pairs,
|
|
22
|
+
* because they have no spaces to split on. A query becomes the same pairs
|
|
23
|
+
* as a phrase, which matches exactly that substring. A run's last character
|
|
24
|
+
* starts no pair, so it is also indexed alone, and a one-character query
|
|
25
|
+
* finds every character as a prefix.
|
|
26
|
+
*/
|
|
27
|
+
export type SearchWeight = "A" | "B" | "C" | "D";
|
|
28
|
+
/** Normalized form used for title and summary comparisons. */
|
|
29
|
+
export declare function normalizeSearchText(value: string): string;
|
|
30
|
+
/** UTF-8 length without encoding: 1-3 bytes per UTF-16 unit, 4 per pair. */
|
|
31
|
+
export declare function utf8ByteLength(value: string): number;
|
|
32
|
+
export interface SearchToken {
|
|
33
|
+
lexeme: string;
|
|
34
|
+
position: number;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Document tokens with positions starting at `start`. Returns the next free
|
|
38
|
+
* position so several fields can share one position space.
|
|
39
|
+
*/
|
|
40
|
+
export declare function documentTokens(text: string, start?: number): {
|
|
41
|
+
tokens: SearchToken[];
|
|
42
|
+
next: number;
|
|
43
|
+
};
|
|
44
|
+
/**
|
|
45
|
+
* Query lexemes for one term, in order. Unlike documents, a query word is
|
|
46
|
+
* never split on case: "camelCase" looks for the whole word, which documents
|
|
47
|
+
* index alongside its parts.
|
|
48
|
+
*/
|
|
49
|
+
export declare function queryLexemes(text: string): string[];
|
|
50
|
+
export interface WeightedField {
|
|
51
|
+
text: string | null | undefined;
|
|
52
|
+
weight: SearchWeight;
|
|
53
|
+
}
|
|
54
|
+
export interface SearchVector {
|
|
55
|
+
/** A `tsvector` literal. */
|
|
56
|
+
literal: string;
|
|
57
|
+
/**
|
|
58
|
+
* False when Postgres's limits made the vector drop or merge word
|
|
59
|
+
* positions, so phrase matching can't be exact for this document.
|
|
60
|
+
*/
|
|
61
|
+
positionsComplete: boolean;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* A `tsvector` for the fields, in order, sharing one position space with a
|
|
65
|
+
* gap between fields, so a phrase never spans two of them. Postgres keeps at
|
|
66
|
+
* most 255 positions per word and none past 16,383. Past the last position,
|
|
67
|
+
* each field's words collapse onto its own position, because Postgres merges
|
|
68
|
+
* equal positions and keeps only the higher weight. A word over 255
|
|
69
|
+
* positions keeps its first position in each field. A very large document
|
|
70
|
+
* keeps one position per word in each field, and one with more distinct
|
|
71
|
+
* words than fit keeps the words that come first, so the vector stays under
|
|
72
|
+
* Postgres's size limit. Any of these makes `positionsComplete` false; each
|
|
73
|
+
* keeps which fields hold each word.
|
|
74
|
+
*/
|
|
75
|
+
export declare function buildSearchVector(fields: readonly WeightedField[]): SearchVector;
|
|
76
|
+
/**
|
|
77
|
+
* A term with more words than the index can match. The query can't be
|
|
78
|
+
* answered from the index; the app's fallback search can answer it.
|
|
79
|
+
*/
|
|
80
|
+
export declare class SearchTermTooLongError extends RangeError {
|
|
81
|
+
constructor();
|
|
82
|
+
}
|
|
83
|
+
export interface QueryPhraseOptions {
|
|
84
|
+
/** Treat the last lexeme as a prefix. */
|
|
85
|
+
prefix?: boolean;
|
|
86
|
+
/** Restrict every lexeme to these weights. */
|
|
87
|
+
weights?: string;
|
|
88
|
+
/**
|
|
89
|
+
* Match the lexemes anywhere, in any order, instead of as a phrase. For
|
|
90
|
+
* documents whose positions aren't complete.
|
|
91
|
+
*/
|
|
92
|
+
anyOrder?: boolean;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* One term as a `tsquery` literal: a single lexeme, or a phrase of adjacent
|
|
96
|
+
* lexemes. Returns null when the term has nothing to match. Throws
|
|
97
|
+
* `SearchTermTooLongError` past `MAX_QUERY_LEXEMES`, which Postgres can't
|
|
98
|
+
* evaluate.
|
|
99
|
+
*/
|
|
100
|
+
export declare function termTsquery(text: string, options?: QueryPhraseOptions): string | null;
|
|
101
|
+
/** Whether a term is more than one lexeme, so it matches as a phrase. */
|
|
102
|
+
export declare function isPhraseTerm(text: string): boolean;
|
|
103
|
+
export declare function anyOfTsquery(parts: readonly (string | null)[]): string | null;
|