@agent-native/core 0.200.0-nightly-20261002124406 → 0.200.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/production-agent.d.ts +5 -1
- package/dist/agent/production-agent.js +36 -1
- package/dist/app-config/runtime.d.ts +1 -0
- package/dist/app-config/runtime.js +4 -0
- package/dist/app-config/schema.d.ts +1 -0
- package/dist/cli/design-connect.d.ts +1 -0
- package/dist/cli/design-connect.js +147 -7
- package/dist/client/analytics.js +24 -1
- package/dist/client/session-replay.d.ts +6 -0
- package/dist/client/session-replay.js +47 -5
- package/dist/collab/awareness.d.ts +2 -2
- package/dist/collab/routes.d.ts +1 -1
- package/dist/file-upload/builder.js +20 -2
- package/dist/mcp-client/app-api.d.ts +11 -1
- package/dist/mcp-client/app-api.js +17 -3
- package/dist/mcp-client/index.d.ts +1 -1
- package/dist/mcp-client/manager.d.ts +3 -1
- package/dist/mcp-client/manager.js +4 -0
- package/dist/notifications/routes.d.ts +3 -3
- package/dist/observability/metrics.js +14 -7
- package/dist/provider-api/actions/custom-provider-registration.d.ts +6 -6
- package/dist/provider-api/actions/provider-api.d.ts +15 -15
- package/dist/resource-changes/store.d.ts +166 -0
- package/dist/resource-changes/store.js +499 -0
- package/dist/search/index-store.d.ts +9 -0
- package/dist/search/index-store.js +50 -0
- package/dist/search/index.d.ts +5 -1
- package/dist/search/index.js +5 -0
- package/dist/search/indexer.d.ts +32 -0
- package/dist/search/indexer.js +651 -0
- package/dist/search/query-parser.d.ts +29 -0
- package/dist/search/query-parser.js +71 -0
- package/dist/search/query.d.ts +46 -0
- package/dist/search/query.js +149 -0
- package/dist/search/registry.d.ts +55 -0
- package/dist/search/registry.js +93 -0
- package/dist/search/tokenize.d.ts +103 -0
- package/dist/search/tokenize.js +367 -0
- package/dist/server/action-change-marker-write.js +5 -1
- package/dist/server/agent-chat/run-code-tools.d.ts +7 -0
- package/dist/server/agent-chat/run-code-tools.js +7 -0
- package/dist/server/agent-chat-plugin.d.ts +5 -1
- package/dist/server/agent-chat-plugin.js +18 -12
- package/dist/server/agent-engine-api-key-route.d.ts +1 -1
- package/dist/server/agent-engine-default-model-route.d.ts +2 -2
- package/dist/server/release-schema.js +8 -0
- package/dist/triggers/actions/manage-automation.d.ts +3 -3
- package/package.json +6 -9
|
@@ -0,0 +1,499 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The app resource change feed: a durable note that one of an app's
|
|
3
|
+
* resources changed, kept once for every consumer that asked to hear about
|
|
4
|
+
* that resource type.
|
|
5
|
+
*
|
|
6
|
+
* Producers call one SQL function, `agent_native_app_resource_changed`, in the
|
|
7
|
+
* writer's own transaction, so nothing is recorded unless the write commits.
|
|
8
|
+
* Today the producers are row triggers that core generates for a registered
|
|
9
|
+
* table. They catch every writer: actions, sync jobs, raw SQL, and deletes.
|
|
10
|
+
* If core later gains a data layer that every write goes through, it calls the
|
|
11
|
+
* same function and the triggers retire; consumers don't change.
|
|
12
|
+
*
|
|
13
|
+
* Each (consumer, resource) pair is one row. Repeated changes update that row
|
|
14
|
+
* rather than adding rows, so autosave doesn't grow the feed. Every change
|
|
15
|
+
* takes a new value from one sequence, taken after the row lock, so for any
|
|
16
|
+
* one resource a larger `seq` always means a later committed change. Consumers
|
|
17
|
+
* use it to reject stale work and to delete only what they processed.
|
|
18
|
+
*
|
|
19
|
+
* A change that fails is retried with backoff, and after its last allowed
|
|
20
|
+
* attempt it is marked failed but still retried every few minutes. It stays
|
|
21
|
+
* pending the whole time, so a consumer that reports freshness never treats
|
|
22
|
+
* it as done.
|
|
23
|
+
*
|
|
24
|
+
* Consumers never poll. They process changes only where the database is
|
|
25
|
+
* already awake: before a read that needs fresh data, right after a write, or
|
|
26
|
+
* inside the framework's recurring sweep. See docs/search-architecture.md.
|
|
27
|
+
*/
|
|
28
|
+
import { getDbExec } from "../db/client.js";
|
|
29
|
+
import { ensureIndexExists, ensureSchemaObject, ensureTableExists, runGuardedDdl, } from "../db/ddl-guard.js";
|
|
30
|
+
export const RESOURCE_CHANGES_TABLE = "app_resource_changes";
|
|
31
|
+
export const RESOURCE_CHANGE_CONSUMERS_TABLE = "app_resource_change_consumers";
|
|
32
|
+
export const RESOURCE_CHANGE_SEQUENCE = "app_resource_change_seq";
|
|
33
|
+
export const RESOURCE_CHANGED_FUNCTION = "agent_native_app_resource_changed";
|
|
34
|
+
/** After this many attempts a change is marked failed. */
|
|
35
|
+
export const RESOURCE_CHANGE_MAX_ATTEMPTS = 5;
|
|
36
|
+
const CLAIM_LEASE_SECONDS = 60;
|
|
37
|
+
/** The longest wait between retries of a failing change. */
|
|
38
|
+
const MAX_RETRY_SECONDS = 300;
|
|
39
|
+
/**
|
|
40
|
+
* Caps the exponent: a failing change keeps being retried, and
|
|
41
|
+
* `power(2, attempts)` overflows after about a thousand attempts.
|
|
42
|
+
*/
|
|
43
|
+
const MAX_BACKOFF_DOUBLINGS = 10;
|
|
44
|
+
const RESOURCE_CHANGE_CONSUMERS_CREATE_SQL = `
|
|
45
|
+
CREATE TABLE IF NOT EXISTS ${RESOURCE_CHANGE_CONSUMERS_TABLE} (
|
|
46
|
+
app TEXT NOT NULL,
|
|
47
|
+
resource_type TEXT NOT NULL,
|
|
48
|
+
consumer TEXT NOT NULL,
|
|
49
|
+
registered_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
50
|
+
PRIMARY KEY (app, resource_type, consumer)
|
|
51
|
+
)
|
|
52
|
+
`;
|
|
53
|
+
const RESOURCE_CHANGES_CREATE_SQL = `
|
|
54
|
+
CREATE TABLE IF NOT EXISTS ${RESOURCE_CHANGES_TABLE} (
|
|
55
|
+
consumer TEXT NOT NULL,
|
|
56
|
+
app TEXT NOT NULL,
|
|
57
|
+
resource_type TEXT NOT NULL,
|
|
58
|
+
resource_id TEXT NOT NULL,
|
|
59
|
+
reason TEXT NOT NULL,
|
|
60
|
+
seq BIGINT NOT NULL,
|
|
61
|
+
changed_at TIMESTAMPTZ NOT NULL DEFAULT clock_timestamp(),
|
|
62
|
+
available_at TIMESTAMPTZ NOT NULL DEFAULT clock_timestamp(),
|
|
63
|
+
attempts INTEGER NOT NULL DEFAULT 0,
|
|
64
|
+
failed_at TIMESTAMPTZ,
|
|
65
|
+
PRIMARY KEY (consumer, app, resource_type, resource_id)
|
|
66
|
+
)
|
|
67
|
+
`;
|
|
68
|
+
const RESOURCE_CHANGES_ORDER_INDEX_SQL = `CREATE INDEX IF NOT EXISTS app_resource_changes_seq_idx ON ${RESOURCE_CHANGES_TABLE} (consumer, app, resource_type, seq)`;
|
|
69
|
+
// Failed changes are rare, so a partial index keeps "is anything failing?"
|
|
70
|
+
// instant even while a large backlog is pending.
|
|
71
|
+
const RESOURCE_CHANGES_FAILED_INDEX_SQL = `CREATE INDEX IF NOT EXISTS app_resource_changes_failed_idx ON ${RESOURCE_CHANGES_TABLE} (consumer, app, resource_type) WHERE failed_at IS NOT NULL`;
|
|
72
|
+
const RESOURCE_CHANGE_SEQUENCE_SQL = `CREATE SEQUENCE IF NOT EXISTS ${RESOURCE_CHANGE_SEQUENCE}`;
|
|
73
|
+
// `nextval` in DO UPDATE runs after the conflicting row is locked, so a later
|
|
74
|
+
// writer to the same resource always gets a larger seq than the one it waited
|
|
75
|
+
// on. Resetting `available_at` makes a leased row claimable again, and the
|
|
76
|
+
// seq check on delete keeps the earlier claimant from dropping it.
|
|
77
|
+
const RESOURCE_CHANGED_FUNCTION_SQL = `
|
|
78
|
+
CREATE OR REPLACE FUNCTION ${RESOURCE_CHANGED_FUNCTION}(
|
|
79
|
+
p_app TEXT, p_resource_type TEXT, p_resource_id TEXT, p_reason TEXT
|
|
80
|
+
) RETURNS void LANGUAGE sql AS $an_resource_changed$
|
|
81
|
+
INSERT INTO ${RESOURCE_CHANGES_TABLE} AS existing
|
|
82
|
+
(consumer, app, resource_type, resource_id, reason, seq, changed_at, available_at, attempts, failed_at)
|
|
83
|
+
SELECT consumer, p_app, p_resource_type, p_resource_id, p_reason,
|
|
84
|
+
nextval('${RESOURCE_CHANGE_SEQUENCE}'), clock_timestamp(), clock_timestamp(), 0, NULL
|
|
85
|
+
FROM ${RESOURCE_CHANGE_CONSUMERS_TABLE}
|
|
86
|
+
WHERE app = p_app AND resource_type = p_resource_type
|
|
87
|
+
ON CONFLICT (consumer, app, resource_type, resource_id) DO UPDATE SET
|
|
88
|
+
reason = EXCLUDED.reason,
|
|
89
|
+
seq = nextval('${RESOURCE_CHANGE_SEQUENCE}'),
|
|
90
|
+
changed_at = EXCLUDED.changed_at,
|
|
91
|
+
available_at = EXCLUDED.available_at,
|
|
92
|
+
attempts = 0,
|
|
93
|
+
failed_at = NULL
|
|
94
|
+
$an_resource_changed$
|
|
95
|
+
`;
|
|
96
|
+
let ensured;
|
|
97
|
+
/**
|
|
98
|
+
* Creates the feed's tables, sequence, and producer function. Listed in
|
|
99
|
+
* `server/release-schema.ts`; trigger installation calls it too, so the
|
|
100
|
+
* function a trigger calls always exists before the trigger does.
|
|
101
|
+
*/
|
|
102
|
+
export function ensureResourceChangeTables(injectedClient) {
|
|
103
|
+
if (injectedClient)
|
|
104
|
+
return ensureAll(injectedClient);
|
|
105
|
+
ensured ??= ensureAll().catch((error) => {
|
|
106
|
+
ensured = undefined;
|
|
107
|
+
throw error;
|
|
108
|
+
});
|
|
109
|
+
return ensured;
|
|
110
|
+
}
|
|
111
|
+
async function ensureAll(injectedClient) {
|
|
112
|
+
const options = { injectedClient };
|
|
113
|
+
await ensureTableExists(RESOURCE_CHANGE_CONSUMERS_TABLE, RESOURCE_CHANGE_CONSUMERS_CREATE_SQL, options);
|
|
114
|
+
await ensureTableExists(RESOURCE_CHANGES_TABLE, RESOURCE_CHANGES_CREATE_SQL, options);
|
|
115
|
+
await ensureIndexExists("app_resource_changes_seq_idx", RESOURCE_CHANGES_ORDER_INDEX_SQL, options);
|
|
116
|
+
await ensureIndexExists("app_resource_changes_failed_idx", RESOURCE_CHANGES_FAILED_INDEX_SQL, options);
|
|
117
|
+
await ensureSchemaObject({
|
|
118
|
+
probe: () => catalogHas(`SELECT 1 FROM pg_class WHERE relkind = 'S' AND relname = ?`, RESOURCE_CHANGE_SEQUENCE, injectedClient),
|
|
119
|
+
ddl: RESOURCE_CHANGE_SEQUENCE_SQL,
|
|
120
|
+
label: `sequence ${RESOURCE_CHANGE_SEQUENCE}`,
|
|
121
|
+
injectedClient,
|
|
122
|
+
});
|
|
123
|
+
await ensureSchemaObject({
|
|
124
|
+
probe: () => catalogHas(`SELECT 1 FROM pg_proc WHERE proname = ?`, RESOURCE_CHANGED_FUNCTION, injectedClient),
|
|
125
|
+
ddl: RESOURCE_CHANGED_FUNCTION_SQL,
|
|
126
|
+
label: `function ${RESOURCE_CHANGED_FUNCTION}`,
|
|
127
|
+
injectedClient,
|
|
128
|
+
});
|
|
129
|
+
}
|
|
130
|
+
async function catalogHas(sql, name, injectedClient) {
|
|
131
|
+
try {
|
|
132
|
+
const { rows } = await (injectedClient ?? getDbExec()).execute({
|
|
133
|
+
sql,
|
|
134
|
+
args: [name],
|
|
135
|
+
});
|
|
136
|
+
return rows.length > 0;
|
|
137
|
+
}
|
|
138
|
+
catch {
|
|
139
|
+
// coercion-ok: an unreadable catalog is not evidence of absence;
|
|
140
|
+
// ensureSchemaObject fails closed on undefined.
|
|
141
|
+
return undefined;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
const IDENTIFIER = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
145
|
+
const KEY = /^[a-z0-9][a-z0-9_-]{0,62}$/;
|
|
146
|
+
function assertIdentifier(value, label) {
|
|
147
|
+
if (!IDENTIFIER.test(value)) {
|
|
148
|
+
throw new Error(`${label} must be a plain SQL identifier, got "${value}".`);
|
|
149
|
+
}
|
|
150
|
+
return value;
|
|
151
|
+
}
|
|
152
|
+
export function assertResourceKey(value, label) {
|
|
153
|
+
if (!KEY.test(value)) {
|
|
154
|
+
throw new Error(`${label} must be lowercase letters, numbers, "-" or "_", got "${value}".`);
|
|
155
|
+
}
|
|
156
|
+
return value;
|
|
157
|
+
}
|
|
158
|
+
const FNV64_OFFSET = 0xcbf29ce484222325n;
|
|
159
|
+
const FNV64_PRIME = 0x100000001b3n;
|
|
160
|
+
const UINT64 = 0xffffffffffffffffn;
|
|
161
|
+
/** 64-bit FNV-1a, as 16 hex digits. */
|
|
162
|
+
function shortHash(value) {
|
|
163
|
+
let hash = FNV64_OFFSET;
|
|
164
|
+
for (let index = 0; index < value.length; index += 1) {
|
|
165
|
+
hash ^= BigInt(value.charCodeAt(index));
|
|
166
|
+
hash = (hash * FNV64_PRIME) & UINT64;
|
|
167
|
+
}
|
|
168
|
+
return hash.toString(16).padStart(16, "0");
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* Names of the function and triggers generated for one source. Functions are
|
|
172
|
+
* schema-wide and the readable part is shortened to fit Postgres's 63
|
|
173
|
+
* characters, so a hash of the app, table, and type keeps every source's
|
|
174
|
+
* names distinct. Installing a source replaces whatever has its names, so
|
|
175
|
+
* the hash is wide enough that two sources never share them.
|
|
176
|
+
*/
|
|
177
|
+
export function resourceChangeTriggerNames(source) {
|
|
178
|
+
const table = assertIdentifier(source.table, "Resource table");
|
|
179
|
+
const app = assertResourceKey(source.app, "App");
|
|
180
|
+
const type = assertResourceKey(source.resourceType, "Resource type");
|
|
181
|
+
const readable = `an_rc_${table}__${type.replace(/-/g, "_")}`.slice(0, 42);
|
|
182
|
+
const base = `${readable}_${shortHash(`${app}\u0000${table}\u0000${type}`)}`;
|
|
183
|
+
return {
|
|
184
|
+
function: base,
|
|
185
|
+
insertDeleteTrigger: `${base}_iud`,
|
|
186
|
+
updateTrigger: `${base}_upd`,
|
|
187
|
+
truncateTrigger: `${base}_trn`,
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* SQL that makes every committed insert, update, delete, and truncate on the
|
|
192
|
+
* source table call the producer function. An update that changes nothing is
|
|
193
|
+
* skipped. An update that changes the id reports both ids. A truncate
|
|
194
|
+
* reports every row it removes as deleted.
|
|
195
|
+
*
|
|
196
|
+
* "Changes nothing" compares the rows' stored bytes (`*<>`), not their
|
|
197
|
+
* values: a `json` or `point` column has no equality operator, and comparing
|
|
198
|
+
* such rows by value would make every update on the table fail.
|
|
199
|
+
*
|
|
200
|
+
* Triggers are replaced in place, never dropped and recreated: a write
|
|
201
|
+
* committed between a drop and a create would never be recorded.
|
|
202
|
+
*/
|
|
203
|
+
export function resourceChangeTriggerSql(source) {
|
|
204
|
+
const table = assertIdentifier(source.table, "Resource table");
|
|
205
|
+
const id = assertIdentifier(source.idColumn, "Resource id column");
|
|
206
|
+
const app = assertResourceKey(source.app, "App");
|
|
207
|
+
const type = assertResourceKey(source.resourceType, "Resource type");
|
|
208
|
+
const names = resourceChangeTriggerNames(source);
|
|
209
|
+
const changed = (row, reason) => `${RESOURCE_CHANGED_FUNCTION}('${app}', '${type}', ${row}."${id}"::text, '${reason}')`;
|
|
210
|
+
return [
|
|
211
|
+
`CREATE OR REPLACE FUNCTION "${names.function}"() RETURNS trigger LANGUAGE plpgsql AS $an_rc$
|
|
212
|
+
BEGIN
|
|
213
|
+
IF TG_OP = 'TRUNCATE' THEN
|
|
214
|
+
PERFORM ${changed("truncated", "delete")} FROM "${table}" AS truncated;
|
|
215
|
+
ELSIF TG_OP = 'INSERT' THEN
|
|
216
|
+
PERFORM ${changed("NEW", "insert")};
|
|
217
|
+
ELSIF TG_OP = 'DELETE' THEN
|
|
218
|
+
PERFORM ${changed("OLD", "delete")};
|
|
219
|
+
ELSE
|
|
220
|
+
IF OLD."${id}" IS DISTINCT FROM NEW."${id}" THEN
|
|
221
|
+
PERFORM ${changed("OLD", "delete")};
|
|
222
|
+
END IF;
|
|
223
|
+
PERFORM ${changed("NEW", "update")};
|
|
224
|
+
END IF;
|
|
225
|
+
RETURN NULL;
|
|
226
|
+
END
|
|
227
|
+
$an_rc$`,
|
|
228
|
+
`CREATE OR REPLACE TRIGGER "${names.insertDeleteTrigger}" AFTER INSERT OR DELETE ON "${table}" FOR EACH ROW EXECUTE FUNCTION "${names.function}"()`,
|
|
229
|
+
`CREATE OR REPLACE TRIGGER "${names.updateTrigger}" AFTER UPDATE ON "${table}" FOR EACH ROW WHEN (OLD *<> NEW) EXECUTE FUNCTION "${names.function}"()`,
|
|
230
|
+
`CREATE OR REPLACE TRIGGER "${names.truncateTrigger}" BEFORE TRUNCATE ON "${table}" FOR EACH STATEMENT EXECUTE FUNCTION "${names.function}"()`,
|
|
231
|
+
];
|
|
232
|
+
}
|
|
233
|
+
/**
|
|
234
|
+
* Installs change capture for a source and subscribes a consumer to it.
|
|
235
|
+
* Apps call this from a named migration, so the triggers ship with the app's
|
|
236
|
+
* schema rather than being created on a request path.
|
|
237
|
+
*
|
|
238
|
+
* Creating a trigger waits for every open transaction on the table, and
|
|
239
|
+
* queues the table's writes behind it while it waits, so each waits at most
|
|
240
|
+
* a few seconds. Returns false when one gave up; the caller retries later.
|
|
241
|
+
*/
|
|
242
|
+
export async function installResourceChangeCapture(exec, source, consumer) {
|
|
243
|
+
await ensureResourceChangeTables(exec);
|
|
244
|
+
await subscribeResourceChangeConsumer(exec, source, consumer);
|
|
245
|
+
for (const statement of resourceChangeTriggerSql(source)) {
|
|
246
|
+
const applied = await runGuardedDdl(statement, {
|
|
247
|
+
lockTimeout: "3s",
|
|
248
|
+
injectedClient: exec,
|
|
249
|
+
});
|
|
250
|
+
if (!applied)
|
|
251
|
+
return false;
|
|
252
|
+
}
|
|
253
|
+
return true;
|
|
254
|
+
}
|
|
255
|
+
export async function subscribeResourceChangeConsumer(exec, source, consumer) {
|
|
256
|
+
await exec.execute({
|
|
257
|
+
sql: `INSERT INTO ${RESOURCE_CHANGE_CONSUMERS_TABLE} (app, resource_type, consumer) VALUES (?, ?, ?) ON CONFLICT DO NOTHING`,
|
|
258
|
+
args: [source.app, source.resourceType, consumer],
|
|
259
|
+
});
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* True when every generated trigger exists on the source table and fires.
|
|
263
|
+
* A disabled trigger (`ALTER TABLE ... DISABLE TRIGGER`) records nothing, so
|
|
264
|
+
* it counts as missing.
|
|
265
|
+
*/
|
|
266
|
+
export async function resourceChangeCaptureInstalled(exec, source) {
|
|
267
|
+
const names = resourceChangeTriggerNames(source);
|
|
268
|
+
const { rows } = await exec.execute({
|
|
269
|
+
sql: `SELECT t.tgname
|
|
270
|
+
FROM pg_trigger t
|
|
271
|
+
JOIN pg_class c ON c.oid = t.tgrelid
|
|
272
|
+
JOIN pg_namespace n ON n.oid = c.relnamespace
|
|
273
|
+
WHERE n.nspname = 'public' AND c.relname = ? AND t.tgname IN (?, ?, ?)
|
|
274
|
+
AND NOT t.tgisinternal AND t.tgenabled IN ('O', 'A')`,
|
|
275
|
+
args: [
|
|
276
|
+
source.table,
|
|
277
|
+
names.insertDeleteTrigger,
|
|
278
|
+
names.updateTrigger,
|
|
279
|
+
names.truncateTrigger,
|
|
280
|
+
],
|
|
281
|
+
});
|
|
282
|
+
return rows.length === 3;
|
|
283
|
+
}
|
|
284
|
+
function fenced(fence) {
|
|
285
|
+
return fence
|
|
286
|
+
? { sql: ` AND ${fence.sql}`, args: fence.args }
|
|
287
|
+
: { sql: "", args: [] };
|
|
288
|
+
}
|
|
289
|
+
/**
|
|
290
|
+
* Leases up to `limit` ready changes, oldest first. Rows another claimant is
|
|
291
|
+
* locking right now are skipped, and the lease expires on its own if this
|
|
292
|
+
* process dies, so there is nothing to clean up.
|
|
293
|
+
*/
|
|
294
|
+
export async function claimResourceChanges(exec, feed, limit, fence) {
|
|
295
|
+
const guard = fenced(fence);
|
|
296
|
+
// The pick is an uncorrelated ARRAY(...) sub-select, which Postgres runs
|
|
297
|
+
// once, and the update then finds each row by its full primary key. Two
|
|
298
|
+
// other shapes go wrong: a subquery in FROM can be rescanned inside a
|
|
299
|
+
// nested loop, and each rescan skips the rows it already locked and takes
|
|
300
|
+
// the next ones, claiming far past the limit; and any join back to this
|
|
301
|
+
// table depends on statistics, which PGlite never gathers, so it can
|
|
302
|
+
// compare every pending row against every picked one.
|
|
303
|
+
const { rows } = await exec.execute({
|
|
304
|
+
sql: `UPDATE ${RESOURCE_CHANGES_TABLE} AS c
|
|
305
|
+
SET available_at = clock_timestamp() + interval '${CLAIM_LEASE_SECONDS} seconds',
|
|
306
|
+
attempts = c.attempts + 1
|
|
307
|
+
WHERE c.consumer = ? AND c.app = ? AND c.resource_type = ?
|
|
308
|
+
AND c.resource_id = ANY (ARRAY(
|
|
309
|
+
SELECT resource_id FROM ${RESOURCE_CHANGES_TABLE}
|
|
310
|
+
WHERE consumer = ? AND app = ? AND resource_type = ?
|
|
311
|
+
AND available_at <= clock_timestamp()
|
|
312
|
+
ORDER BY seq
|
|
313
|
+
LIMIT ?
|
|
314
|
+
FOR UPDATE SKIP LOCKED
|
|
315
|
+
))${guard.sql}
|
|
316
|
+
RETURNING c.resource_id, c.seq::text AS seq, c.attempts`,
|
|
317
|
+
args: [
|
|
318
|
+
feed.consumer,
|
|
319
|
+
feed.app,
|
|
320
|
+
feed.resourceType,
|
|
321
|
+
feed.consumer,
|
|
322
|
+
feed.app,
|
|
323
|
+
feed.resourceType,
|
|
324
|
+
limit,
|
|
325
|
+
...guard.args,
|
|
326
|
+
],
|
|
327
|
+
});
|
|
328
|
+
return rows
|
|
329
|
+
.map((row) => ({
|
|
330
|
+
resourceId: String(row.resource_id),
|
|
331
|
+
seq: String(row.seq),
|
|
332
|
+
attempts: Number(row.attempts),
|
|
333
|
+
}))
|
|
334
|
+
.sort((a, b) => compareSeq(a.seq, b.seq));
|
|
335
|
+
}
|
|
336
|
+
export function compareSeq(a, b) {
|
|
337
|
+
const left = BigInt(a);
|
|
338
|
+
const right = BigInt(b);
|
|
339
|
+
return left < right ? -1 : left > right ? 1 : 0;
|
|
340
|
+
}
|
|
341
|
+
/**
|
|
342
|
+
* Claimed changes as a VALUES list for seq-guarded statements, plus an id
|
|
343
|
+
* list. The id list lets Postgres find the rows by primary key; without it,
|
|
344
|
+
* a join against VALUES can scan every pending row when the table has no
|
|
345
|
+
* statistics, as in PGlite.
|
|
346
|
+
*/
|
|
347
|
+
function valuesList(changes) {
|
|
348
|
+
return {
|
|
349
|
+
sql: changes.map(() => "(?, ?::bigint)").join(", "),
|
|
350
|
+
args: changes.flatMap((change) => [change.resourceId, change.seq]),
|
|
351
|
+
ids: changes.map(() => "?").join(", "),
|
|
352
|
+
idArgs: changes.map((change) => change.resourceId),
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
/**
|
|
356
|
+
* Deletes processed changes, but only rows whose seq is still the one that
|
|
357
|
+
* was claimed. A change recorded meanwhile stays for the next pass.
|
|
358
|
+
*/
|
|
359
|
+
export async function completeResourceChanges(exec, feed, changes, fence) {
|
|
360
|
+
if (!changes.length)
|
|
361
|
+
return;
|
|
362
|
+
const values = valuesList(changes);
|
|
363
|
+
const guard = fenced(fence);
|
|
364
|
+
await exec.execute({
|
|
365
|
+
sql: `DELETE FROM ${RESOURCE_CHANGES_TABLE} AS c
|
|
366
|
+
USING (VALUES ${values.sql}) AS done (resource_id, seq)
|
|
367
|
+
WHERE c.consumer = ? AND c.app = ? AND c.resource_type = ?
|
|
368
|
+
AND c.resource_id IN (${values.ids})
|
|
369
|
+
AND c.resource_id = done.resource_id AND c.seq = done.seq${guard.sql}`,
|
|
370
|
+
args: [
|
|
371
|
+
...values.args,
|
|
372
|
+
feed.consumer,
|
|
373
|
+
feed.app,
|
|
374
|
+
feed.resourceType,
|
|
375
|
+
...values.idArgs,
|
|
376
|
+
...guard.args,
|
|
377
|
+
],
|
|
378
|
+
});
|
|
379
|
+
}
|
|
380
|
+
/**
|
|
381
|
+
* Backs off failed changes, doubling the wait up to five minutes. After the
|
|
382
|
+
* last allowed attempt a change is also marked failed, which a consumer can
|
|
383
|
+
* report, but it keeps being retried: a failure that was only temporary
|
|
384
|
+
* clears itself, and a later write to the resource starts it fresh.
|
|
385
|
+
*/
|
|
386
|
+
export async function failResourceChanges(exec, feed, changes, fence) {
|
|
387
|
+
if (!changes.length)
|
|
388
|
+
return;
|
|
389
|
+
const values = valuesList(changes);
|
|
390
|
+
const guard = fenced(fence);
|
|
391
|
+
await exec.execute({
|
|
392
|
+
sql: `UPDATE ${RESOURCE_CHANGES_TABLE} AS c
|
|
393
|
+
SET available_at = clock_timestamp() + make_interval(secs => least(${MAX_RETRY_SECONDS}, 5 * power(2, least(c.attempts, ${MAX_BACKOFF_DOUBLINGS})))),
|
|
394
|
+
failed_at = CASE WHEN c.attempts >= ${RESOURCE_CHANGE_MAX_ATTEMPTS} THEN coalesce(c.failed_at, clock_timestamp()) ELSE c.failed_at END
|
|
395
|
+
FROM (VALUES ${values.sql}) AS failed (resource_id, seq)
|
|
396
|
+
WHERE c.consumer = ? AND c.app = ? AND c.resource_type = ?
|
|
397
|
+
AND c.resource_id IN (${values.ids})
|
|
398
|
+
AND c.resource_id = failed.resource_id AND c.seq = failed.seq${guard.sql}`,
|
|
399
|
+
args: [
|
|
400
|
+
...values.args,
|
|
401
|
+
feed.consumer,
|
|
402
|
+
feed.app,
|
|
403
|
+
feed.resourceType,
|
|
404
|
+
...values.idArgs,
|
|
405
|
+
...guard.args,
|
|
406
|
+
],
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
/**
|
|
410
|
+
* Two boolean columns for a caller's own SELECT, so checking the feed costs
|
|
411
|
+
* no extra round trip: `pending` while any change is waiting, leased, or
|
|
412
|
+
* backing off, and `failing` while any has used up its attempts.
|
|
413
|
+
*/
|
|
414
|
+
export function resourceChangeBacklogColumns(feed) {
|
|
415
|
+
const where = `consumer = ? AND app = ? AND resource_type = ?`;
|
|
416
|
+
const args = [feed.consumer, feed.app, feed.resourceType];
|
|
417
|
+
return {
|
|
418
|
+
sql: `EXISTS (SELECT 1 FROM ${RESOURCE_CHANGES_TABLE} WHERE ${where}) AS pending,
|
|
419
|
+
EXISTS (SELECT 1 FROM ${RESOURCE_CHANGES_TABLE} WHERE ${where} AND failed_at IS NOT NULL) AS failing`,
|
|
420
|
+
args: [...args, ...args],
|
|
421
|
+
};
|
|
422
|
+
}
|
|
423
|
+
/** True while any change at or below `seq` is still pending for this feed. */
|
|
424
|
+
export async function hasPendingResourceChanges(exec, feed, options = {}) {
|
|
425
|
+
const bound = options.atOrBelowSeq === undefined ? "" : " AND seq <= ?::bigint";
|
|
426
|
+
const { rows } = await exec.execute({
|
|
427
|
+
sql: `SELECT 1 FROM ${RESOURCE_CHANGES_TABLE}
|
|
428
|
+
WHERE consumer = ? AND app = ? AND resource_type = ?${bound}
|
|
429
|
+
LIMIT 1`,
|
|
430
|
+
args: [
|
|
431
|
+
feed.consumer,
|
|
432
|
+
feed.app,
|
|
433
|
+
feed.resourceType,
|
|
434
|
+
...(options.atOrBelowSeq === undefined ? [] : [options.atOrBelowSeq]),
|
|
435
|
+
],
|
|
436
|
+
});
|
|
437
|
+
return rows.length > 0;
|
|
438
|
+
}
|
|
439
|
+
/**
|
|
440
|
+
* Records a change for every row of the source table, for this consumer
|
|
441
|
+
* only, and returns a seq at or above every one it assigned. Consumers use
|
|
442
|
+
* it to rebuild from scratch.
|
|
443
|
+
*
|
|
444
|
+
* A change already queued is replaced by a fresh one, even if another
|
|
445
|
+
* process has it leased or it has failed: that work was done for the old
|
|
446
|
+
* consumer state, so it must not complete what the rebuild queued.
|
|
447
|
+
*/
|
|
448
|
+
export async function enqueueAllResourceChanges(exec, source, consumer, reason) {
|
|
449
|
+
const table = assertIdentifier(source.table, "Resource table");
|
|
450
|
+
const id = assertIdentifier(source.idColumn, "Resource id column");
|
|
451
|
+
// As in the producer function, the replacement seq is taken after the
|
|
452
|
+
// row lock, so it is newer than any change a writer committed meanwhile.
|
|
453
|
+
await exec.execute({
|
|
454
|
+
sql: `INSERT INTO ${RESOURCE_CHANGES_TABLE} (consumer, app, resource_type, resource_id, reason, seq)
|
|
455
|
+
SELECT ?, ?, ?, "${id}"::text, ?, nextval('${RESOURCE_CHANGE_SEQUENCE}') FROM "${table}"
|
|
456
|
+
ON CONFLICT (consumer, app, resource_type, resource_id) DO UPDATE SET
|
|
457
|
+
reason = EXCLUDED.reason,
|
|
458
|
+
seq = nextval('${RESOURCE_CHANGE_SEQUENCE}'),
|
|
459
|
+
changed_at = clock_timestamp(),
|
|
460
|
+
available_at = clock_timestamp(),
|
|
461
|
+
attempts = 0,
|
|
462
|
+
failed_at = NULL`,
|
|
463
|
+
args: [consumer, source.app, source.resourceType, reason],
|
|
464
|
+
});
|
|
465
|
+
const { rows } = await exec.execute(`SELECT nextval('${RESOURCE_CHANGE_SEQUENCE}')::text AS seq`);
|
|
466
|
+
return String(rows[0]?.seq);
|
|
467
|
+
}
|
|
468
|
+
const afterWriteDrains = new Map();
|
|
469
|
+
/**
|
|
470
|
+
* Registers work to run right after a request that changed data, while the
|
|
471
|
+
* database is known to be awake. Returns an unregister function.
|
|
472
|
+
*/
|
|
473
|
+
export function registerAfterWriteDrain(id, drain) {
|
|
474
|
+
afterWriteDrains.set(id, drain);
|
|
475
|
+
return () => {
|
|
476
|
+
if (afterWriteDrains.get(id) === drain)
|
|
477
|
+
afterWriteDrains.delete(id);
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
/**
|
|
481
|
+
* Runs registered after-write drains without delaying the caller. On
|
|
482
|
+
* serverless the request's `waitUntil` keeps the function alive for them.
|
|
483
|
+
*/
|
|
484
|
+
export function runAfterWriteDrains(waitUntil) {
|
|
485
|
+
if (afterWriteDrains.size === 0)
|
|
486
|
+
return;
|
|
487
|
+
const work = Promise.allSettled([...afterWriteDrains.entries()].map(async ([id, drain]) => {
|
|
488
|
+
try {
|
|
489
|
+
await drain();
|
|
490
|
+
}
|
|
491
|
+
catch (error) {
|
|
492
|
+
console.warn(`[resource-changes] after-write drain ${id} failed:`, error instanceof Error ? error.message : String(error));
|
|
493
|
+
}
|
|
494
|
+
}));
|
|
495
|
+
if (waitUntil)
|
|
496
|
+
waitUntil(work);
|
|
497
|
+
else
|
|
498
|
+
void work;
|
|
499
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tables for the core search index. Fixed names, scoped by `app` and
|
|
3
|
+
* `resource_type`, so several apps can share one database. Listed in
|
|
4
|
+
* `server/release-schema.ts`.
|
|
5
|
+
*/
|
|
6
|
+
import type { DbExec } from "../db/client.js";
|
|
7
|
+
export declare const SEARCH_RESOURCES_TABLE = "search_resources";
|
|
8
|
+
export declare const SEARCH_INDEX_STATE_TABLE = "search_index_state";
|
|
9
|
+
export declare function ensureSearchIndexTables(injectedClient?: DbExec): Promise<void>;
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import { ensureIndexExists, ensureTableExists } from "../db/ddl-guard.js";
|
|
2
|
+
export const SEARCH_RESOURCES_TABLE = "search_resources";
|
|
3
|
+
export const SEARCH_INDEX_STATE_TABLE = "search_index_state";
|
|
4
|
+
const SEARCH_RESOURCES_CREATE_SQL = `
|
|
5
|
+
CREATE TABLE IF NOT EXISTS ${SEARCH_RESOURCES_TABLE} (
|
|
6
|
+
app TEXT NOT NULL,
|
|
7
|
+
resource_type TEXT NOT NULL,
|
|
8
|
+
resource_id TEXT NOT NULL,
|
|
9
|
+
title TEXT NOT NULL DEFAULT '',
|
|
10
|
+
title_norm TEXT NOT NULL DEFAULT '',
|
|
11
|
+
summary_norm TEXT NOT NULL DEFAULT '',
|
|
12
|
+
doc_vector TSVECTOR NOT NULL DEFAULT ''::tsvector,
|
|
13
|
+
positions_complete BOOLEAN NOT NULL DEFAULT true,
|
|
14
|
+
modified_at TIMESTAMPTZ,
|
|
15
|
+
content_hash TEXT NOT NULL,
|
|
16
|
+
index_version INTEGER NOT NULL,
|
|
17
|
+
indexed_seq BIGINT NOT NULL,
|
|
18
|
+
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
19
|
+
PRIMARY KEY (app, resource_type, resource_id)
|
|
20
|
+
)
|
|
21
|
+
`;
|
|
22
|
+
const SEARCH_RESOURCES_VECTOR_INDEX_SQL = `CREATE INDEX IF NOT EXISTS search_resources_doc_vector_gin ON ${SEARCH_RESOURCES_TABLE} USING GIN (doc_vector)`;
|
|
23
|
+
const SEARCH_INDEX_STATE_CREATE_SQL = `
|
|
24
|
+
CREATE TABLE IF NOT EXISTS ${SEARCH_INDEX_STATE_TABLE} (
|
|
25
|
+
app TEXT NOT NULL,
|
|
26
|
+
resource_type TEXT NOT NULL,
|
|
27
|
+
target_version INTEGER NOT NULL,
|
|
28
|
+
index_version INTEGER,
|
|
29
|
+
rebuild_high_seq BIGINT,
|
|
30
|
+
rebuild_started_at TIMESTAMPTZ,
|
|
31
|
+
rebuild_completed_at TIMESTAMPTZ,
|
|
32
|
+
PRIMARY KEY (app, resource_type)
|
|
33
|
+
)
|
|
34
|
+
`;
|
|
35
|
+
let ensured;
|
|
36
|
+
export function ensureSearchIndexTables(injectedClient) {
|
|
37
|
+
if (injectedClient)
|
|
38
|
+
return ensureAll(injectedClient);
|
|
39
|
+
ensured ??= ensureAll().catch((error) => {
|
|
40
|
+
ensured = undefined;
|
|
41
|
+
throw error;
|
|
42
|
+
});
|
|
43
|
+
return ensured;
|
|
44
|
+
}
|
|
45
|
+
async function ensureAll(injectedClient) {
|
|
46
|
+
const options = { injectedClient };
|
|
47
|
+
await ensureTableExists(SEARCH_RESOURCES_TABLE, SEARCH_RESOURCES_CREATE_SQL, options);
|
|
48
|
+
await ensureIndexExists("search_resources_doc_vector_gin", SEARCH_RESOURCES_VECTOR_INDEX_SQL, options);
|
|
49
|
+
await ensureTableExists(SEARCH_INDEX_STATE_TABLE, SEARCH_INDEX_STATE_CREATE_SQL, options);
|
|
50
|
+
}
|
package/dist/search/index.d.ts
CHANGED
|
@@ -1,4 +1,9 @@
|
|
|
1
1
|
import type { DbExec } from "../db/index.js";
|
|
2
|
+
export { getSearchableResource, listSearchableResources, registerSearchableResource, searchIndexMigration, unregisterSearchableResource, type SearchableResourceDocument, type SearchableResourceRegistration, } from "./registry.js";
|
|
3
|
+
export { drainAllSearchIndexes, prepareSearchIndex, resetSearchIndexRuntime, type SearchIndexNotReadyReason, type SearchIndexStatus, } from "./indexer.js";
|
|
4
|
+
export { indexedSearchSql, type IndexedSearchOptions, type IndexedSearchSql, } from "./query.js";
|
|
5
|
+
export { parseSearchQuery, searchQueryNeedles, type ParsedSearchQuery, type SearchQueryGroup, type SearchQueryTerm, } from "./query-parser.js";
|
|
6
|
+
export { buildSearchVector, documentTokens, isPhraseTerm, normalizeSearchText, queryLexemes, SearchTermTooLongError, termTsquery, type SearchVector, } from "./tokenize.js";
|
|
2
7
|
export declare const DEFAULT_SEARCH_NAMESPACE = "creative_context";
|
|
3
8
|
export declare const PGVECTOR_REQUIRED_MESSAGE = "Vector search requires Postgres with the pgvector extension in the configured DATABASE_URL database.";
|
|
4
9
|
export interface SearchNamespaceIdentifiers {
|
|
@@ -83,4 +88,3 @@ export declare function reciprocalRankFusion<T>(lanes: Readonly<Record<string, r
|
|
|
83
88
|
rankConstant?: number;
|
|
84
89
|
limit?: number;
|
|
85
90
|
}): FusedCandidate<T>[];
|
|
86
|
-
export {};
|
package/dist/search/index.js
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
export { getSearchableResource, listSearchableResources, registerSearchableResource, searchIndexMigration, unregisterSearchableResource, } from "./registry.js";
|
|
2
|
+
export { drainAllSearchIndexes, prepareSearchIndex, resetSearchIndexRuntime, } from "./indexer.js";
|
|
3
|
+
export { indexedSearchSql, } from "./query.js";
|
|
4
|
+
export { parseSearchQuery, searchQueryNeedles, } from "./query-parser.js";
|
|
5
|
+
export { buildSearchVector, documentTokens, isPhraseTerm, normalizeSearchText, queryLexemes, SearchTermTooLongError, termTsquery, } from "./tokenize.js";
|
|
1
6
|
export const DEFAULT_SEARCH_NAMESPACE = "creative_context";
|
|
2
7
|
export const PGVECTOR_REQUIRED_MESSAGE = "Vector search requires Postgres with the pgvector extension in the configured DATABASE_URL database.";
|
|
3
8
|
function pgVectorOptions(options) {
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { type SearchableResourceRegistration } from "./registry.js";
|
|
2
|
+
export type SearchIndexNotReadyReason = "capture-missing" | "rebuilding" | "backlog" | "failed-changes" | "outdated-registration" | "unavailable";
|
|
3
|
+
export interface SearchIndexStatus {
|
|
4
|
+
ready: boolean;
|
|
5
|
+
reason?: SearchIndexNotReadyReason;
|
|
6
|
+
}
|
|
7
|
+
/** Forgets per-process memos. Tests use it between databases. */
|
|
8
|
+
export declare function resetSearchIndexRuntime(): void;
|
|
9
|
+
/**
|
|
10
|
+
* Drains pending changes for one registration until `deadline` and reports
|
|
11
|
+
* whether the index is complete and current, meaning a search can trust it.
|
|
12
|
+
* Concurrent calls in one process share a single drain, which finishes the
|
|
13
|
+
* batch it is on before stopping.
|
|
14
|
+
*/
|
|
15
|
+
export declare function drainSearchIndex(registration: SearchableResourceRegistration, deadline: number): Promise<SearchIndexStatus>;
|
|
16
|
+
/**
|
|
17
|
+
* Call before searching. Reports whether the index can answer: it reflects
|
|
18
|
+
* every change committed before this call and was built by this
|
|
19
|
+
* registration's version. When changes are pending, it spends up to the
|
|
20
|
+
* budget indexing them, and answers by then either way. When the index can't
|
|
21
|
+
* answer (first build, a backlog, a change that keeps failing, missing
|
|
22
|
+
* capture, a newer deploy), use the app's fallback search for this request.
|
|
23
|
+
*
|
|
24
|
+
* `budgetMs` defaults to `runtime.searchDrainBudgetMs`. Zero indexes nothing
|
|
25
|
+
* here and leaves pending changes to the drains that follow writes and the
|
|
26
|
+
* recurring sweep.
|
|
27
|
+
*/
|
|
28
|
+
export declare function prepareSearchIndex(registration: SearchableResourceRegistration, options?: {
|
|
29
|
+
budgetMs?: number;
|
|
30
|
+
}): Promise<SearchIndexStatus>;
|
|
31
|
+
/** Drains every registration, in turn, until `deadline`. */
|
|
32
|
+
export declare function drainAllSearchIndexes(deadline: number): Promise<void>;
|