@agilesyndrome/cf-genai-llm 4.1.2 → 5.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRACT.md +2 -1
- package/README.md +20 -0
- package/changelog.md +8 -0
- package/package.json +1 -1
- package/src/index.js +34 -3
- package/tests/feature.test.mjs +44 -1
package/CONTRACT.md
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
# LLM feature contract
|
|
2
2
|
|
|
3
3
|
- createFeature(options) returns name and middleware.
|
|
4
|
-
- createLLM(options) returns `generate`, `generateMulti`, `review`, and `reviewMulti`.
|
|
4
|
+
- createLLM(options) returns `generate`, `generateJob`, `generateMulti`, `review`, and `reviewMulti`.
|
|
5
|
+
- `generateJob(jobId, ...)` executes an already-dispatched base job, reports progress without model text, and stores `{}` unless `job.toJobResult` supplies a compact domain result.
|
|
5
6
|
- `generateMulti` and `reviewMulti` start all requests concurrently and preserve input order.
|
|
6
7
|
- The client uses the OpenAI Responses-compatible contract; OpenAI is the default endpoint and arbitrary compatible endpoints are supported.
|
|
7
8
|
- `LLM_API_URL`, `LLM_API_TOKEN`, and `LLM_MODEL` are the universal configuration variables.
|
package/README.md
CHANGED
|
@@ -11,6 +11,26 @@ const result = await llm.generate("Write a summary", schema, { schemaName: "summ
|
|
|
11
11
|
const reviews = await llm.reviewMulti(result, reviewerPrompts, reviewSchema);
|
|
12
12
|
```
|
|
13
13
|
|
|
14
|
+
Long-running generation is explicit. Dispatch a base job to an application-owned
|
|
15
|
+
Cloudflare Workflow, then execute the generation from that Workflow with the
|
|
16
|
+
same job ID:
|
|
17
|
+
|
|
18
|
+
```js
|
|
19
|
+
const execution = await llm.generateJob(event.payload.jobId, prompt, schema, {
|
|
20
|
+
env: this.env,
|
|
21
|
+
job: {
|
|
22
|
+
toJobResult: async (recipe) => {
|
|
23
|
+
await recipes.save(recipe);
|
|
24
|
+
return { resourceType: "recipe", resourceId: recipe.id };
|
|
25
|
+
},
|
|
26
|
+
},
|
|
27
|
+
});
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`generateJob` reports phases through base's durable job events. It deliberately
|
|
31
|
+
stores no generated content by default; `job.toJobResult` should persist the
|
|
32
|
+
domain value and return only the small result descriptor needed by the UI.
|
|
33
|
+
|
|
14
34
|
## Providers and AI Gateway
|
|
15
35
|
|
|
16
36
|
The client speaks the OpenAI Responses-compatible API. OpenAI is the default
|
package/changelog.md
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 5.0.0
|
|
4
|
+
|
|
5
|
+
- Require cf-genai-base 5.
|
|
6
|
+
- Add explicit durable generation through `generateJob`.
|
|
7
|
+
- Report generation progress through base job events.
|
|
8
|
+
- Keep model output out of durable job records unless the application supplies a compact result mapper.
|
package/package.json
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"name":"@agilesyndrome/cf-genai-llm","version":"
|
|
1
|
+
{"name":"@agilesyndrome/cf-genai-llm","version":"5.0.0","description":"Composable Cloudflare Worker LLM generation and review client.","type":"module","exports":{".":"./src/index.js"},"files":["src","tests","README.md","CONTRACT.md","changelog.md","LICENSE"],"scripts":{"check":"node --check src/index.js","test":"node --test tests/*.test.mjs","build":"npm run check && npm test && npm pack --dry-run"},"license":"MIT","publishConfig":{"access":"public","provenance":true},"repository":{"type":"git","url":"git+https://github.com/agilesyndrome/cf-genai-llm.git"},"homepage":"https://github.com/agilesyndrome/cf-genai-llm#readme","dependencies":{"@agilesyndrome/cf-genai-base":"^5.0.0"}}
|
package/src/index.js
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
const DEFAULT_ENDPOINT = "https://api.openai.com/v1/responses";
|
|
2
2
|
const DEFAULT_MODEL = "gpt-5.4";
|
|
3
3
|
export const PACKAGE_NAME = "@agilesyndrome/cf-genai-llm";
|
|
4
|
-
export const VERSION = "
|
|
4
|
+
export const VERSION = "5.0.0";
|
|
5
5
|
export class LLMCircuitBreakerError extends Error { constructor(message = "LLM generation is temporarily unavailable") { super(message); this.name = "LLMCircuitBreakerError"; this.code = "circuit_breaker_open"; this.circuitBreakerOpen = true; } }
|
|
6
6
|
|
|
7
|
-
import { getCircuitBreaker, registerCircuitBreaker, registerHealthcheck, setCircuitBreaker } from "@agilesyndrome/cf-genai-base";
|
|
7
|
+
import { executeJob, getCircuitBreaker, registerCircuitBreaker, registerHealthcheck, setCircuitBreaker } from "@agilesyndrome/cf-genai-base";
|
|
8
8
|
export class LLMResponseError extends Error {
|
|
9
9
|
constructor(message, llmResponse, cause) {
|
|
10
10
|
super(message, { cause });
|
|
@@ -87,6 +87,36 @@ export function createLLM(options = {}) {
|
|
|
87
87
|
return Promise.all(items.map((item) => generate(typeof item === "string" ? { prompt: item, schema: shared.schema } : { ...item, schema: item.schema || shared.schema }, { ...shared.options, ...(item.options || {}) })));
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
+
async function generateJob(jobId, promptOrRequest, schemaOrOptions, maybeOptions) {
|
|
91
|
+
const input = normalizeGenerateArgs(promptOrRequest, schemaOrOptions, maybeOptions);
|
|
92
|
+
const jobOptions = input.options.job || {};
|
|
93
|
+
const generationOptions = { ...input.options };
|
|
94
|
+
delete generationOptions.job;
|
|
95
|
+
const env = generationOptions.env || options.env;
|
|
96
|
+
if (!env?.DB) throw new TypeError("generateJob requires an environment with DB");
|
|
97
|
+
const onText = generationOptions.onText;
|
|
98
|
+
const onUsage = generationOptions.onUsage;
|
|
99
|
+
return executeJob(env, jobId, async ({ report }) => {
|
|
100
|
+
await report({ phase: "generating" });
|
|
101
|
+
const value = await generate(input.prompt, input.schema, {
|
|
102
|
+
...generationOptions,
|
|
103
|
+
onText: async (text) => {
|
|
104
|
+
if (onText) await onText(text);
|
|
105
|
+
await report({ phase: "generated" });
|
|
106
|
+
},
|
|
107
|
+
onUsage: async (usage) => {
|
|
108
|
+
if (onUsage) await onUsage(usage);
|
|
109
|
+
await report({ phase: "generating", usage });
|
|
110
|
+
},
|
|
111
|
+
});
|
|
112
|
+
return value;
|
|
113
|
+
}, {
|
|
114
|
+
who: jobOptions.who || generationOptions.who || "system:update",
|
|
115
|
+
ctx: jobOptions.ctx || generationOptions.ctx,
|
|
116
|
+
toJobResult: jobOptions.toJobResult || emptyJobResult,
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
|
|
90
120
|
async function review(originalTextOrRequest, reviewPromptOrPrompts, schemaOrOptions, maybeOptions) {
|
|
91
121
|
if (Array.isArray(reviewPromptOrPrompts)) return reviewMulti(originalTextOrRequest, reviewPromptOrPrompts, schemaOrOptions, maybeOptions);
|
|
92
122
|
const args = normalizeReviewArgs(originalTextOrRequest, reviewPromptOrPrompts, schemaOrOptions, maybeOptions);
|
|
@@ -102,7 +132,7 @@ export function createLLM(options = {}) {
|
|
|
102
132
|
}));
|
|
103
133
|
}
|
|
104
134
|
|
|
105
|
-
return { listModels, generate, generateMulti, generateWithSchema: generate, generateMultiWithSchema: generateMulti, review, reviewMulti, reviewWithSchema: review, reviewMultiWithSchema: reviewMulti };
|
|
135
|
+
return { listModels, generate, generateJob, generateMulti, generateWithSchema: generate, generateMultiWithSchema: generateMulti, review, reviewMulti, reviewWithSchema: review, reviewMultiWithSchema: reviewMulti };
|
|
106
136
|
}
|
|
107
137
|
|
|
108
138
|
function normalizeGenerateArgs(promptOrRequest, schemaOrOptions, maybeOptions) {
|
|
@@ -155,5 +185,6 @@ function validate(value, schema, path) {
|
|
|
155
185
|
if (schema.maximum !== undefined && value > schema.maximum) throw new Error(`${path} is above maximum`);
|
|
156
186
|
}
|
|
157
187
|
function normalizeUsage(usage = {}) { return { inputTokens: usage.input_tokens ?? usage.prompt_tokens ?? 0, outputTokens: usage.output_tokens ?? usage.completion_tokens ?? 0, totalTokens: usage.total_tokens ?? 0 }; }
|
|
188
|
+
function emptyJobResult() { return {}; }
|
|
158
189
|
|
|
159
190
|
export function createFeature(options = {}) { const name = options.name || "cf-genai-llm"; const client = createLLM({ ...options, feature: name }); return { name, displayName: options.displayName || name, packageName: PACKAGE_NAME, version: VERSION, dataResources: options.dataResources || [], routes: options.routes || [], healthcheck: async (env) => { try { resolveConfig(options, {}, env); } catch { return [{ feature: name, component: "configuration", displayName: "LLM configuration", state: "red" }]; } try { await client.listModels({ env, who: "system:update" }); return [{ feature: name, component: "configuration", displayName: "LLM configuration", state: "green" }]; } catch { return [{ feature: name, component: "configuration", displayName: "LLM configuration", state: "yellow" }]; } }, healthchecks: options.healthchecks || [{ feature: name, component: "llm-models", displayName: "LLM model availability", state: "yellow" }], circuitBreakers: options.circuitBreakers || [{ id: name + ":llm-models", feature: name, name: "llm-models", displayName: "LLM model access", state: "on", allowSelfHealing: true, healthchecks: [name + ":llm-models"] }], middleware: async (request, env, ctx, next, state) => { if (options.boot) await options.boot(env, { request, ctx, state }); return options.handle ? options.handle(request, env, ctx, next, state) : next(); } }; }
|
package/tests/feature.test.mjs
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import assert from "node:assert/strict";
|
|
3
|
+
import { createJob, getJob } from "@agilesyndrome/cf-genai-base";
|
|
3
4
|
import { createFeature, createLLM, LLMResponseError } from "../src/index.js";
|
|
4
5
|
|
|
5
6
|
test("feature delegates to the next handler", async () => {
|
|
6
7
|
const feature = createFeature({ name: "example" });
|
|
7
|
-
assert.equal(feature.version, "
|
|
8
|
+
assert.equal(feature.version, "5.0.0");
|
|
8
9
|
const response = await feature.middleware(new Request("https://example.test/"), {}, {}, () => Response.json({ ok: true }), {});
|
|
9
10
|
assert.equal(response.status, 200);
|
|
10
11
|
assert.deepEqual(await response.json(), { ok: true });
|
|
@@ -13,6 +14,34 @@ test("feature delegates to the next handler", async () => {
|
|
|
13
14
|
const schema = { type: "object", additionalProperties: false, properties: { answer: { type: "string" } }, required: ["answer"] };
|
|
14
15
|
function response(text, usage = { input_tokens: 3, output_tokens: 2, total_tokens: 5 }) { return new Response(JSON.stringify({ output_text: text, usage }), { status: 200, headers: { "content-type": "application/json" } }); }
|
|
15
16
|
|
|
17
|
+
function jobDatabase() {
|
|
18
|
+
const jobs = new Map();
|
|
19
|
+
const events = [];
|
|
20
|
+
return {
|
|
21
|
+
jobs,
|
|
22
|
+
events,
|
|
23
|
+
prepare(sql) {
|
|
24
|
+
const statement = { args: [], bind(...args) { this.args = args; return this; } };
|
|
25
|
+
statement.run = async () => {
|
|
26
|
+
if (sql.includes("INSERT INTO core_jobs")) {
|
|
27
|
+
const [id, type, status, ownerId, tenantId, resourceType, resourceId, input, progress, createdAt, updatedAt, expiresAt] = statement.args;
|
|
28
|
+
jobs.set(id, { id, type, status, owner_id: ownerId, tenant_id: tenantId, resource_type: resourceType, resource_id: resourceId, input_json: input, result_json: "{}", error_json: null, progress_json: progress, created_at: createdAt, started_at: null, finished_at: null, updated_at: updatedAt, expires_at: expiresAt });
|
|
29
|
+
} else if (sql.includes("INSERT INTO core_job_events")) {
|
|
30
|
+
const [id, jobId, type, payload] = statement.args;
|
|
31
|
+
events.push({ id, job_id: jobId, type, payload_json: payload, created_at: new Date().toISOString() });
|
|
32
|
+
} else if (sql.startsWith("UPDATE core_jobs SET")) {
|
|
33
|
+
const row = jobs.get(statement.args.at(-1));
|
|
34
|
+
for (const [index, assignment] of [...sql.matchAll(/([a-z_]+) = \?/g)].entries()) row[assignment[1]] = statement.args[index];
|
|
35
|
+
}
|
|
36
|
+
return {};
|
|
37
|
+
};
|
|
38
|
+
statement.first = async () => sql.includes("SELECT * FROM core_jobs WHERE id") ? jobs.get(statement.args[0]) || null : null;
|
|
39
|
+
statement.all = async () => ({ results: [] });
|
|
40
|
+
return statement;
|
|
41
|
+
},
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
|
|
16
45
|
test("generate sends metadata, validates typed output, and logs token counts", async () => {
|
|
17
46
|
const calls = []; const logs = [];
|
|
18
47
|
const llm = createLLM({ env: { LLM_API_TOKEN: "test" }, fetch: async (url, init) => { calls.push({ url, headers: init.headers, body: JSON.parse(init.body) }); return response("{\"answer\":\"ok\"}"); }, logger: { debug: (line) => logs.push(JSON.parse(line)), info: (line) => logs.push(JSON.parse(line)) }, metadata: { app: "test" } });
|
|
@@ -26,6 +55,20 @@ test("generate sends metadata, validates typed output, and logs token counts", a
|
|
|
26
55
|
assert.equal(responseLog.gateway, false);
|
|
27
56
|
});
|
|
28
57
|
|
|
58
|
+
test("generateJob updates a dispatched base job without persisting generated content", async () => {
|
|
59
|
+
const DB = jobDatabase();
|
|
60
|
+
const env = { DB, LLM_API_TOKEN: "test", eventHandler: async () => {} };
|
|
61
|
+
const job = await createJob(env, { type: "llm.recipe", ownerId: "user-1" });
|
|
62
|
+
const llm = createLLM({ env, fetch: async () => response("{\"answer\":\"secret output\"}") });
|
|
63
|
+
const execution = await llm.generateJob(job.id, "hello", schema, {
|
|
64
|
+
job: { toJobResult: (value) => ({ answerLength: value.answer.length }) },
|
|
65
|
+
});
|
|
66
|
+
assert.deepEqual(execution.value, { answer: "secret output" });
|
|
67
|
+
assert.deepEqual(execution.job.result, { answerLength: 13 });
|
|
68
|
+
assert.equal((await getJob(env, job.id)).status, "succeeded");
|
|
69
|
+
assert.deepEqual(DB.events.map((event) => event.type), ["job.created", "job.running", "job.progress", "job.progress", "job.progress", "job.succeeded"]);
|
|
70
|
+
});
|
|
71
|
+
|
|
29
72
|
test("Cloudflare AI Gateway routes Responses and model health through the gateway", async () => {
|
|
30
73
|
const calls = [];
|
|
31
74
|
const llm = createLLM({
|