@intentface/latch-core 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +45 -0
- package/dist/agent.d.ts +200 -0
- package/dist/agent.d.ts.map +1 -0
- package/dist/agent.js +9 -0
- package/dist/agent.js.map +1 -0
- package/dist/compaction.d.ts +33 -0
- package/dist/compaction.d.ts.map +1 -0
- package/dist/compaction.js +104 -0
- package/dist/compaction.js.map +1 -0
- package/dist/connections.d.ts +16 -0
- package/dist/connections.d.ts.map +1 -0
- package/dist/connections.js +41 -0
- package/dist/connections.js.map +1 -0
- package/dist/context.d.ts +41 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +25 -0
- package/dist/context.js.map +1 -0
- package/dist/current-date.d.ts +12 -0
- package/dist/current-date.d.ts.map +1 -0
- package/dist/current-date.js +25 -0
- package/dist/current-date.js.map +1 -0
- package/dist/extensions.d.ts +333 -0
- package/dist/extensions.d.ts.map +1 -0
- package/dist/extensions.js +569 -0
- package/dist/extensions.js.map +1 -0
- package/dist/harness/index.d.ts +17 -0
- package/dist/harness/index.d.ts.map +1 -0
- package/dist/harness/index.js +15 -0
- package/dist/harness/index.js.map +1 -0
- package/dist/harness/tools.d.ts +88 -0
- package/dist/harness/tools.d.ts.map +1 -0
- package/dist/harness/tools.js +296 -0
- package/dist/harness/tools.js.map +1 -0
- package/dist/harness/web-fetch.d.ts +47 -0
- package/dist/harness/web-fetch.d.ts.map +1 -0
- package/dist/harness/web-fetch.js +247 -0
- package/dist/harness/web-fetch.js.map +1 -0
- package/dist/index.d.ts +25 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/limits.d.ts +152 -0
- package/dist/limits.d.ts.map +1 -0
- package/dist/limits.js +97 -0
- package/dist/limits.js.map +1 -0
- package/dist/memory.d.ts +93 -0
- package/dist/memory.d.ts.map +1 -0
- package/dist/memory.js +13 -0
- package/dist/memory.js.map +1 -0
- package/dist/message.d.ts +46 -0
- package/dist/message.d.ts.map +1 -0
- package/dist/message.js +2 -0
- package/dist/message.js.map +1 -0
- package/dist/models/catalog.d.ts +56 -0
- package/dist/models/catalog.d.ts.map +1 -0
- package/dist/models/catalog.js +211 -0
- package/dist/models/catalog.js.map +1 -0
- package/dist/models/defaults.d.ts +23 -0
- package/dist/models/defaults.d.ts.map +1 -0
- package/dist/models/defaults.js +19 -0
- package/dist/models/defaults.js.map +1 -0
- package/dist/models/index.d.ts +23 -0
- package/dist/models/index.d.ts.map +1 -0
- package/dist/models/index.js +19 -0
- package/dist/models/index.js.map +1 -0
- package/dist/models/prompt-caching.d.ts +19 -0
- package/dist/models/prompt-caching.d.ts.map +1 -0
- package/dist/models/prompt-caching.js +18 -0
- package/dist/models/prompt-caching.js.map +1 -0
- package/dist/models/provider.d.ts +19 -0
- package/dist/models/provider.d.ts.map +1 -0
- package/dist/models/provider.js +22 -0
- package/dist/models/provider.js.map +1 -0
- package/dist/models/reasoning.d.ts +21 -0
- package/dist/models/reasoning.d.ts.map +1 -0
- package/dist/models/reasoning.js +59 -0
- package/dist/models/reasoning.js.map +1 -0
- package/dist/pricing.d.ts +52 -0
- package/dist/pricing.d.ts.map +1 -0
- package/dist/pricing.js +37 -0
- package/dist/pricing.js.map +1 -0
- package/dist/principal.d.ts +37 -0
- package/dist/principal.d.ts.map +1 -0
- package/dist/principal.js +30 -0
- package/dist/principal.js.map +1 -0
- package/dist/projections.d.ts +37 -0
- package/dist/projections.d.ts.map +1 -0
- package/dist/projections.js +128 -0
- package/dist/projections.js.map +1 -0
- package/dist/prompt-caching.d.ts +106 -0
- package/dist/prompt-caching.d.ts.map +1 -0
- package/dist/prompt-caching.js +165 -0
- package/dist/prompt-caching.js.map +1 -0
- package/dist/runtime.d.ts +670 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +2425 -0
- package/dist/runtime.js.map +1 -0
- package/dist/scheduler.d.ts +31 -0
- package/dist/scheduler.d.ts.map +1 -0
- package/dist/scheduler.js +43 -0
- package/dist/scheduler.js.map +1 -0
- package/dist/storage.d.ts +389 -0
- package/dist/storage.d.ts.map +1 -0
- package/dist/storage.js +38 -0
- package/dist/storage.js.map +1 -0
- package/dist/telemetry.d.ts +155 -0
- package/dist/telemetry.d.ts.map +1 -0
- package/dist/telemetry.js +2 -0
- package/dist/telemetry.js.map +1 -0
- package/dist/vault-node.d.ts +20 -0
- package/dist/vault-node.d.ts.map +1 -0
- package/dist/vault-node.js +30 -0
- package/dist/vault-node.js.map +1 -0
- package/dist/vault.d.ts +62 -0
- package/dist/vault.d.ts.map +1 -0
- package/dist/vault.js +88 -0
- package/dist/vault.js.map +1 -0
- package/package.json +95 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"provider.d.ts","sourceRoot":"","sources":["../../src/models/provider.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,MAAM,UAAU,GAAG,WAAW,GAAG,QAAQ,GAAG,QAAQ,CAAC;AAE3D,eAAO,MAAM,YAAY,EAAE,SAAS,UAAU,EAAsC,CAAC;AAErF,qGAAqG;AACrG,wBAAgB,cAAc,CAAC,OAAO,EAAE,MAAM,GAAG,UAAU,GAAG,SAAS,CAKtE;AAED;;;GAGG;AACH,wBAAgB,UAAU,CACxB,OAAO,EAAE,MAAM,EACf,OAAO,CAAC,EAAE,aAAa,CAAC;IAAE,EAAE,EAAE,MAAM,CAAC;IAAC,QAAQ,EAAE,MAAM,CAAA;CAAE,CAAC,GACxD,UAAU,GAAG,SAAS,CAIxB"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export const PROVIDER_IDS = ["anthropic", "openai", "google"];
|
|
2
|
+
/** Provider by id prefix alone: claude → anthropic; gpt/o-series → openai; gemini/gemma → google. */
|
|
3
|
+
export function providerFromId(modelId) {
|
|
4
|
+
if (/^claude/i.test(modelId))
|
|
5
|
+
return "anthropic";
|
|
6
|
+
if (/^(gpt|o\d|chatgpt)/i.test(modelId))
|
|
7
|
+
return "openai";
|
|
8
|
+
if (/^(gemini|gemma)/i.test(modelId))
|
|
9
|
+
return "google";
|
|
10
|
+
return undefined;
|
|
11
|
+
}
|
|
12
|
+
/**
|
|
13
|
+
* Provider for a model id: the catalog's answer when the id is listed, else the
|
|
14
|
+
* id-prefix heuristic. `catalog` is any list of `{ id, provider }`.
|
|
15
|
+
*/
|
|
16
|
+
export function providerOf(modelId, catalog) {
|
|
17
|
+
const listed = catalog?.find((m) => m.id === modelId)?.provider;
|
|
18
|
+
if (listed && PROVIDER_IDS.includes(listed))
|
|
19
|
+
return listed;
|
|
20
|
+
return providerFromId(modelId);
|
|
21
|
+
}
|
|
22
|
+
//# sourceMappingURL=provider.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"provider.js","sourceRoot":"","sources":["../../src/models/provider.ts"],"names":[],"mappings":"AAQA,MAAM,CAAC,MAAM,YAAY,GAA0B,CAAC,WAAW,EAAE,QAAQ,EAAE,QAAQ,CAAC,CAAC;AAErF,qGAAqG;AACrG,MAAM,UAAU,cAAc,CAAC,OAAe;IAC5C,IAAI,UAAU,CAAC,IAAI,CAAC,OAAO,CAAC;QAAE,OAAO,WAAW,CAAC;IACjD,IAAI,qBAAqB,CAAC,IAAI,CAAC,OAAO,CAAC;QAAE,OAAO,QAAQ,CAAC;IACzD,IAAI,kBAAkB,CAAC,IAAI,CAAC,OAAO,CAAC;QAAE,OAAO,QAAQ,CAAC;IACtD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,UAAU,CACxB,OAAe,EACf,OAAyD;IAEzD,MAAM,MAAM,GAAG,OAAO,EAAE,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,EAAE,KAAK,OAAO,CAAC,EAAE,QAAQ,CAAC;IAChE,IAAI,MAAM,IAAK,YAAkC,CAAC,QAAQ,CAAC,MAAM,CAAC;QAAE,OAAO,MAAoB,CAAC;IAChG,OAAO,cAAc,CAAC,OAAO,CAAC,CAAC;AACjC,CAAC"}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import type { ModelInfo } from "./catalog.js";
|
|
2
|
+
/**
|
|
3
|
+
* Per-agent reasoning effort. Providers don't share one knob — OpenAI takes an
|
|
4
|
+
* effort enum, Anthropic/Google take a thinking-token budget — so a single
|
|
5
|
+
* `effort` level is mapped to each provider's shape here. Gated to
|
|
6
|
+
* reasoning-capable models so a non-thinking model is a no-op rather than an
|
|
7
|
+
* API error.
|
|
8
|
+
*/
|
|
9
|
+
export type ReasoningEffort = "minimal" | "low" | "medium" | "high";
|
|
10
|
+
export interface ReasoningResolution {
|
|
11
|
+
providerOptions?: Record<string, unknown>;
|
|
12
|
+
maxOutputTokens?: number;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Build the `RuntimeConfig.resolveReasoningOptions` hook over the model catalog.
|
|
16
|
+
* Returns `providerOptions` (keyed by provider id) plus a safe `maxOutputTokens`
|
|
17
|
+
* for the token-budget providers (the budget must stay under the model's max
|
|
18
|
+
* output, and the answer needs room after thinking).
|
|
19
|
+
*/
|
|
20
|
+
export declare function makeResolveReasoningOptions(models: readonly ModelInfo[]): (modelId: string, effort: string) => ReasoningResolution | undefined;
|
|
21
|
+
//# sourceMappingURL=reasoning.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reasoning.d.ts","sourceRoot":"","sources":["../../src/models/reasoning.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,cAAc,CAAC;AAE9C;;;;;;GAMG;AACH,MAAM,MAAM,eAAe,GAAG,SAAS,GAAG,KAAK,GAAG,QAAQ,GAAG,MAAM,CAAC;AAYpE,MAAM,WAAW,mBAAmB;IAClC,eAAe,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAC1C,eAAe,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED;;;;;GAKG;AACH,wBAAgB,2BAA2B,CAAC,MAAM,EAAE,SAAS,SAAS,EAAE,IAE9D,SAAS,MAAM,EAAE,QAAQ,MAAM,KAAG,mBAAmB,GAAG,SAAS,CA6C1E"}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
const EFFORTS = ["minimal", "low", "medium", "high"];
|
|
2
|
+
/** Thinking-token budgets per effort (Anthropic/Google). OpenAI uses the enum directly. */
|
|
3
|
+
const BUDGET = {
|
|
4
|
+
minimal: 0,
|
|
5
|
+
low: 2048,
|
|
6
|
+
medium: 8192,
|
|
7
|
+
high: 16384,
|
|
8
|
+
};
|
|
9
|
+
/**
|
|
10
|
+
* Build the `RuntimeConfig.resolveReasoningOptions` hook over the model catalog.
|
|
11
|
+
* Returns `providerOptions` (keyed by provider id) plus a safe `maxOutputTokens`
|
|
12
|
+
* for the token-budget providers (the budget must stay under the model's max
|
|
13
|
+
* output, and the answer needs room after thinking).
|
|
14
|
+
*/
|
|
15
|
+
export function makeResolveReasoningOptions(models) {
|
|
16
|
+
const byId = new Map(models.map((m) => [m.id, m]));
|
|
17
|
+
return (modelId, effort) => {
|
|
18
|
+
const info = byId.get(modelId);
|
|
19
|
+
const eff = EFFORTS.includes(effort) ? effort : undefined;
|
|
20
|
+
// No (valid) effort, or a model not known to support reasoning: don't
|
|
21
|
+
// touch thinking config — but still lift the SDK's tiny Anthropic/Google
|
|
22
|
+
// default max_tokens (4096). Models that think by DEFAULT (e.g.
|
|
23
|
+
// claude-sonnet-5) can burn all 4096 inside the thinking block on a heavy
|
|
24
|
+
// turn and end the run with an empty message.
|
|
25
|
+
if (!eff || !info?.reasoning) {
|
|
26
|
+
// Same fallback as the effort path: a catalog entry without maxOutput
|
|
27
|
+
// must still lift the 4096 default.
|
|
28
|
+
return info?.provider === "anthropic" || info?.provider === "google"
|
|
29
|
+
? { maxOutputTokens: Math.min(info.maxOutput ?? 32_000, 32_000) }
|
|
30
|
+
: undefined;
|
|
31
|
+
}
|
|
32
|
+
const provider = info.provider;
|
|
33
|
+
if (provider === "openai") {
|
|
34
|
+
return { providerOptions: { openai: { reasoningEffort: eff } } };
|
|
35
|
+
}
|
|
36
|
+
// Token-budget providers (Anthropic, Google): cap the budget under the
|
|
37
|
+
// model's max output, and set maxOutputTokens so the answer has room.
|
|
38
|
+
const max = info.maxOutput ?? 32_000;
|
|
39
|
+
const budget = Math.min(BUDGET[eff], Math.max(1024, Math.floor(max / 2)));
|
|
40
|
+
const maxOutputTokens = Math.min(max, budget + 8192);
|
|
41
|
+
if (provider === "google") {
|
|
42
|
+
return {
|
|
43
|
+
providerOptions: {
|
|
44
|
+
google: { thinkingConfig: { thinkingBudget: eff === "minimal" ? 0 : budget, includeThoughts: true } },
|
|
45
|
+
},
|
|
46
|
+
maxOutputTokens,
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
// Anthropic: minimal turns thinking off; otherwise an explicit budget.
|
|
50
|
+
if (eff === "minimal") {
|
|
51
|
+
return { providerOptions: { anthropic: { thinking: { type: "disabled" } } } };
|
|
52
|
+
}
|
|
53
|
+
return {
|
|
54
|
+
providerOptions: { anthropic: { thinking: { type: "enabled", budgetTokens: budget }, sendReasoning: true } },
|
|
55
|
+
maxOutputTokens,
|
|
56
|
+
};
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
//# sourceMappingURL=reasoning.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reasoning.js","sourceRoot":"","sources":["../../src/models/reasoning.ts"],"names":[],"mappings":"AAWA,MAAM,OAAO,GAA+B,CAAC,SAAS,EAAE,KAAK,EAAE,QAAQ,EAAE,MAAM,CAAC,CAAC;AAEjF,2FAA2F;AAC3F,MAAM,MAAM,GAAoC;IAC9C,OAAO,EAAE,CAAC;IACV,GAAG,EAAE,IAAI;IACT,MAAM,EAAE,IAAI;IACZ,IAAI,EAAE,KAAK;CACZ,CAAC;AAOF;;;;;GAKG;AACH,MAAM,UAAU,2BAA2B,CAAC,MAA4B;IACtE,MAAM,IAAI,GAAG,IAAI,GAAG,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;IACnD,OAAO,CAAC,OAAe,EAAE,MAAc,EAAmC,EAAE;QAC1E,MAAM,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC;QAC/B,MAAM,GAAG,GAAG,OAAO,CAAC,QAAQ,CAAC,MAAyB,CAAC,CAAC,CAAC,CAAE,MAA0B,CAAC,CAAC,CAAC,SAAS,CAAC;QAClG,sEAAsE;QACtE,yEAAyE;QACzE,gEAAgE;QAChE,0EAA0E;QAC1E,8CAA8C;QAC9C,IAAI,CAAC,GAAG,IAAI,CAAC,IAAI,EAAE,SAAS,EAAE,CAAC;YAC7B,sEAAsE;YACtE,oCAAoC;YACpC,OAAO,IAAI,EAAE,QAAQ,KAAK,WAAW,IAAI,IAAI,EAAE,QAAQ,KAAK,QAAQ;gBAClE,CAAC,CAAC,EAAE,eAAe,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,SAAS,IAAI,MAAM,EAAE,MAAM,CAAC,EAAE;gBACjE,CAAC,CAAC,SAAS,CAAC;QAChB,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;QAE/B,IAAI,QAAQ,KAAK,QAAQ,EAAE,CAAC;YAC1B,OAAO,EAAE,eAAe,EAAE,EAAE,MAAM,EAAE,EAAE,eAAe,EAAE,GAAG,EAAE,EAAE,EAAE,CAAC;QACnE,CAAC;QAED,uEAAuE;QACvE,sEAAsE;QACtE,MAAM,GAAG,GAAG,IAAI,CAAC,SAAS,IAAI,MAAM,CAAC;QACrC,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,GAAG,CAAC,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;QAC1E,MAAM,eAAe,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,MAAM,GAAG,IAAI,CAAC,CAAC;QAErD,IAAI,QAAQ,KAAK,QAAQ,EAAE,CAAC;YAC1B,OAAO;gBACL,eAAe,EAAE;oBACf,MAAM,EAAE,EAAE,cAAc,EAAE,EAAE,cAAc,EAAE,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,MAAM,EAAE,eAAe,EAAE,IAAI,EAAE,EAAE;iBACtG;gBACD,eAAe;aAChB,CAAC;QACJ,CAAC;QAED,uEAAuE;QACvE,IAAI,GAAG,KAAK,SAAS,EAAE,CAAC;YACtB,OAAO,EAAE,eAAe,EAAE,EAAE,SAAS,EAAE,EAAE,QAAQ,EAAE,EAAE,IAAI,EAAE,UAAU,EAAE,EAAE,EAAE,EAAE,CAAC;QAChF,CAAC;QACD,OAAO;YACL,eAAe,EAAE,EAAE,SAAS,EAAE,EAAE,QAAQ,EAAE,EAAE,IAAI,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,EAAE,aAAa,EAAE,IAAI,EAAE,EAAE;YAC5G,eAAe;SAChB,CAAC;IACJ,CAAC,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cost calculation — a small seam over token usage.
|
|
3
|
+
*
|
|
4
|
+
* `Pricing` maps a model id to its per-million-token rates; the runtime applies
|
|
5
|
+
* it to each turn's usage and records the cost on the message, the run, and the
|
|
6
|
+
* usage ledger. Inject a static table now (`staticPricing`); a models.dev-backed
|
|
7
|
+
* catalog (live rates, cached in the DB) can replace it behind the same seam.
|
|
8
|
+
*/
|
|
9
|
+
/** Rate card for one model, in the account's currency unit per 1M tokens. */
|
|
10
|
+
export interface ModelPrice {
|
|
11
|
+
inputPer1M: number;
|
|
12
|
+
outputPer1M: number;
|
|
13
|
+
/** Cached-prompt READ rate (much cheaper). Falls back to the input rate. */
|
|
14
|
+
cacheReadPer1M?: number;
|
|
15
|
+
/** Cache WRITE/creation rate (Anthropic charges a premium). Falls back to input. */
|
|
16
|
+
cacheWritePer1M?: number;
|
|
17
|
+
/**
|
|
18
|
+
* Higher rates that apply to the WHOLE turn once its input exceeds
|
|
19
|
+
* `thresholdTokens` (long-context pricing, e.g. OpenAI's >200k tier). Cache
|
|
20
|
+
* rates fall back to the tier's input rate when absent.
|
|
21
|
+
*/
|
|
22
|
+
tier?: {
|
|
23
|
+
thresholdTokens: number;
|
|
24
|
+
inputPer1M: number;
|
|
25
|
+
outputPer1M: number;
|
|
26
|
+
cacheReadPer1M?: number;
|
|
27
|
+
cacheWritePer1M?: number;
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
/** A turn's token usage, split so cached input can be priced separately. */
|
|
31
|
+
export interface UsageBreakdown {
|
|
32
|
+
/** Total prompt tokens (cached + non-cached). */
|
|
33
|
+
inputTokens: number;
|
|
34
|
+
outputTokens: number;
|
|
35
|
+
/** Cached prompt tokens READ (billed at `cacheReadPer1M`). */
|
|
36
|
+
cacheReadTokens?: number;
|
|
37
|
+
/** Cache-creation tokens WRITTEN (billed at `cacheWritePer1M`). */
|
|
38
|
+
cacheWriteTokens?: number;
|
|
39
|
+
}
|
|
40
|
+
/** Resolve a model id to its price, or undefined if unknown (→ cost 0). */
|
|
41
|
+
export type Pricing = (model: string) => ModelPrice | undefined;
|
|
42
|
+
/**
|
|
43
|
+
* Cost for a turn's token usage under a price (0 if no price). Cached prompt
|
|
44
|
+
* tokens are billed at the cache read/write rates (providers auto-cache repeated
|
|
45
|
+
* prefixes — e.g. a tool loop re-sending context — so cache reads are far
|
|
46
|
+
* cheaper than fresh input). `inputTokens` is the TOTAL; the non-cached
|
|
47
|
+
* remainder bills at the full input rate.
|
|
48
|
+
*/
|
|
49
|
+
export declare function computeCost(price: ModelPrice | undefined, usage: UsageBreakdown): number;
|
|
50
|
+
/** A `Pricing` from a static `{ modelId: ModelPrice }` table. */
|
|
51
|
+
export declare function staticPricing(table: Record<string, ModelPrice>): Pricing;
|
|
52
|
+
//# sourceMappingURL=pricing.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pricing.d.ts","sourceRoot":"","sources":["../src/pricing.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,6EAA6E;AAC7E,MAAM,WAAW,UAAU;IACzB,UAAU,EAAE,MAAM,CAAC;IACnB,WAAW,EAAE,MAAM,CAAC;IACpB,4EAA4E;IAC5E,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,oFAAoF;IACpF,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;;OAIG;IACH,IAAI,CAAC,EAAE;QACL,eAAe,EAAE,MAAM,CAAC;QACxB,UAAU,EAAE,MAAM,CAAC;QACnB,WAAW,EAAE,MAAM,CAAC;QACpB,cAAc,CAAC,EAAE,MAAM,CAAC;QACxB,eAAe,CAAC,EAAE,MAAM,CAAC;KAC1B,CAAC;CACH;AAED,4EAA4E;AAC5E,MAAM,WAAW,cAAc;IAC7B,iDAAiD;IACjD,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC;IACrB,8DAA8D;IAC9D,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,mEAAmE;IACnE,gBAAgB,CAAC,EAAE,MAAM,CAAC;CAC3B;AAED,2EAA2E;AAC3E,MAAM,MAAM,OAAO,GAAG,CAAC,KAAK,EAAE,MAAM,KAAK,UAAU,GAAG,SAAS,CAAC;AAEhE;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,UAAU,GAAG,SAAS,EAAE,KAAK,EAAE,cAAc,GAAG,MAAM,CAiBxF;AAED,iEAAiE;AACjE,wBAAgB,aAAa,CAAC,KAAK,EAAE,MAAM,CAAC,MAAM,EAAE,UAAU,CAAC,GAAG,OAAO,CAExE"}
|
package/dist/pricing.js
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cost calculation — a small seam over token usage.
|
|
3
|
+
*
|
|
4
|
+
* `Pricing` maps a model id to its per-million-token rates; the runtime applies
|
|
5
|
+
* it to each turn's usage and records the cost on the message, the run, and the
|
|
6
|
+
* usage ledger. Inject a static table now (`staticPricing`); a models.dev-backed
|
|
7
|
+
* catalog (live rates, cached in the DB) can replace it behind the same seam.
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Cost for a turn's token usage under a price (0 if no price). Cached prompt
|
|
11
|
+
* tokens are billed at the cache read/write rates (providers auto-cache repeated
|
|
12
|
+
* prefixes — e.g. a tool loop re-sending context — so cache reads are far
|
|
13
|
+
* cheaper than fresh input). `inputTokens` is the TOTAL; the non-cached
|
|
14
|
+
* remainder bills at the full input rate.
|
|
15
|
+
*/
|
|
16
|
+
export function computeCost(price, usage) {
|
|
17
|
+
if (!price)
|
|
18
|
+
return 0;
|
|
19
|
+
// Long-context tier: once input crosses the threshold, the whole turn bills
|
|
20
|
+
// at the higher rates (cache rates fall back to that tier's input rate).
|
|
21
|
+
const useTier = !!price.tier && usage.inputTokens > price.tier.thresholdTokens;
|
|
22
|
+
const r = useTier && price.tier ? price.tier : price;
|
|
23
|
+
const inputRate = r.inputPer1M;
|
|
24
|
+
const cacheRead = usage.cacheReadTokens ?? 0;
|
|
25
|
+
const cacheWrite = usage.cacheWriteTokens ?? 0;
|
|
26
|
+
const nonCached = Math.max(0, usage.inputTokens - cacheRead - cacheWrite);
|
|
27
|
+
const per = (n, rate) => (n / 1_000_000) * rate;
|
|
28
|
+
return (per(nonCached, inputRate) +
|
|
29
|
+
per(cacheRead, r.cacheReadPer1M ?? inputRate) +
|
|
30
|
+
per(cacheWrite, r.cacheWritePer1M ?? inputRate) +
|
|
31
|
+
per(usage.outputTokens, r.outputPer1M));
|
|
32
|
+
}
|
|
33
|
+
/** A `Pricing` from a static `{ modelId: ModelPrice }` table. */
|
|
34
|
+
export function staticPricing(table) {
|
|
35
|
+
return (model) => table[model];
|
|
36
|
+
}
|
|
37
|
+
//# sourceMappingURL=pricing.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pricing.js","sourceRoot":"","sources":["../src/pricing.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAsCH;;;;;;GAMG;AACH,MAAM,UAAU,WAAW,CAAC,KAA6B,EAAE,KAAqB;IAC9E,IAAI,CAAC,KAAK;QAAE,OAAO,CAAC,CAAC;IACrB,4EAA4E;IAC5E,yEAAyE;IACzE,MAAM,OAAO,GAAG,CAAC,CAAC,KAAK,CAAC,IAAI,IAAI,KAAK,CAAC,WAAW,GAAG,KAAK,CAAC,IAAI,CAAC,eAAe,CAAC;IAC/E,MAAM,CAAC,GAAG,OAAO,IAAI,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC;IACrD,MAAM,SAAS,GAAG,CAAC,CAAC,UAAU,CAAC;IAC/B,MAAM,SAAS,GAAG,KAAK,CAAC,eAAe,IAAI,CAAC,CAAC;IAC7C,MAAM,UAAU,GAAG,KAAK,CAAC,gBAAgB,IAAI,CAAC,CAAC;IAC/C,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,KAAK,CAAC,WAAW,GAAG,SAAS,GAAG,UAAU,CAAC,CAAC;IAC1E,MAAM,GAAG,GAAG,CAAC,CAAS,EAAE,IAAY,EAAE,EAAE,CAAC,CAAC,CAAC,GAAG,SAAS,CAAC,GAAG,IAAI,CAAC;IAChE,OAAO,CACL,GAAG,CAAC,SAAS,EAAE,SAAS,CAAC;QACzB,GAAG,CAAC,SAAS,EAAE,CAAC,CAAC,cAAc,IAAI,SAAS,CAAC;QAC7C,GAAG,CAAC,UAAU,EAAE,CAAC,CAAC,eAAe,IAAI,SAAS,CAAC;QAC/C,GAAG,CAAC,KAAK,CAAC,YAAY,EAAE,CAAC,CAAC,WAAW,CAAC,CACvC,CAAC;AACJ,CAAC;AAED,iEAAiE;AACjE,MAAM,UAAU,aAAa,CAAC,KAAiC;IAC7D,OAAO,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;AACjC,CAAC"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Identity & tenancy primitives.
|
|
3
|
+
*
|
|
4
|
+
* Latch carries NO tenancy vocabulary of its own. The customer defines their
|
|
5
|
+
* own `Principal` type (`P`) — whatever their auth produces (tenant/team/user,
|
|
6
|
+
* a workspace id, a single user id, anything). It must be JSON-serializable, so
|
|
7
|
+
* the runtime can persist it per run and rebuild it for resume/cron.
|
|
8
|
+
*
|
|
9
|
+
* The framework treats `P` as opaque: it threads it to the storage adapter,
|
|
10
|
+
* secret store, context builder, tools, and agents — but never reads a field of
|
|
11
|
+
* it. Isolation is enforced in the adapter/store (via an `owner` key derived
|
|
12
|
+
* from the principal), and authorization lives in the customer's app. There is
|
|
13
|
+
* no `Scope`, no `orgId`/`userId` — those, if they exist, live in the
|
|
14
|
+
* customer's `P` and the `owner` function they give their adapter.
|
|
15
|
+
*
|
|
16
|
+
* CONTRACT — principals persist in CLEARTEXT. The runtime serializes `P`
|
|
17
|
+
* verbatim onto durable rows (`runs.identity`, `schedules.identity`, memory
|
|
18
|
+
* episodes) and into OAuth flow state, and replays it on resume/cron — days
|
|
19
|
+
* or months later. Therefore a principal must carry DURABLE IDENTIFIERS ONLY:
|
|
20
|
+
* - never secrets or tokens (they'd be stored unencrypted and replayed
|
|
21
|
+
* stale — credentials belong in the Vault);
|
|
22
|
+
* - no PII beyond ids (emails, names — they'd spread across tables and
|
|
23
|
+
* complicate erasure);
|
|
24
|
+
* - nothing volatile (roles, flags, session state — a replayed copy would
|
|
25
|
+
* silently disagree with reality; derive those per turn in
|
|
26
|
+
* `context.build`, and re-validate stale identity in
|
|
27
|
+
* `reconstructPrincipal`).
|
|
28
|
+
*/
|
|
29
|
+
/** Maps an incoming web request to a `Principal`. Customer-provided. Null = unauthenticated. */
|
|
30
|
+
export type ContextResolver<P> = (request: Request) => P | null | Promise<P | null>;
|
|
31
|
+
/**
|
|
32
|
+
* Rebuilds a `Principal` from the opaque identity blob persisted on a run, for
|
|
33
|
+
* non-interactive triggers (resume, cron) where there is no request. The mirror
|
|
34
|
+
* of `ContextResolver`. Default in the runtime is a passthrough.
|
|
35
|
+
*/
|
|
36
|
+
export type ReconstructPrincipal<P> = (identity: unknown) => P | Promise<P>;
|
|
37
|
+
//# sourceMappingURL=principal.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"principal.d.ts","sourceRoot":"","sources":["../src/principal.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AAEH,gGAAgG;AAChG,MAAM,MAAM,eAAe,CAAC,CAAC,IAAI,CAC/B,OAAO,EAAE,OAAO,KACb,CAAC,GAAG,IAAI,GAAG,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC,CAAC;AAElC;;;;GAIG;AACH,MAAM,MAAM,oBAAoB,CAAC,CAAC,IAAI,CAAC,QAAQ,EAAE,OAAO,KAAK,CAAC,GAAG,OAAO,CAAC,CAAC,CAAC,CAAC"}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Identity & tenancy primitives.
|
|
3
|
+
*
|
|
4
|
+
* Latch carries NO tenancy vocabulary of its own. The customer defines their
|
|
5
|
+
* own `Principal` type (`P`) — whatever their auth produces (tenant/team/user,
|
|
6
|
+
* a workspace id, a single user id, anything). It must be JSON-serializable, so
|
|
7
|
+
* the runtime can persist it per run and rebuild it for resume/cron.
|
|
8
|
+
*
|
|
9
|
+
* The framework treats `P` as opaque: it threads it to the storage adapter,
|
|
10
|
+
* secret store, context builder, tools, and agents — but never reads a field of
|
|
11
|
+
* it. Isolation is enforced in the adapter/store (via an `owner` key derived
|
|
12
|
+
* from the principal), and authorization lives in the customer's app. There is
|
|
13
|
+
* no `Scope`, no `orgId`/`userId` — those, if they exist, live in the
|
|
14
|
+
* customer's `P` and the `owner` function they give their adapter.
|
|
15
|
+
*
|
|
16
|
+
* CONTRACT — principals persist in CLEARTEXT. The runtime serializes `P`
|
|
17
|
+
* verbatim onto durable rows (`runs.identity`, `schedules.identity`, memory
|
|
18
|
+
* episodes) and into OAuth flow state, and replays it on resume/cron — days
|
|
19
|
+
* or months later. Therefore a principal must carry DURABLE IDENTIFIERS ONLY:
|
|
20
|
+
* - never secrets or tokens (they'd be stored unencrypted and replayed
|
|
21
|
+
* stale — credentials belong in the Vault);
|
|
22
|
+
* - no PII beyond ids (emails, names — they'd spread across tables and
|
|
23
|
+
* complicate erasure);
|
|
24
|
+
* - nothing volatile (roles, flags, session state — a replayed copy would
|
|
25
|
+
* silently disagree with reality; derive those per turn in
|
|
26
|
+
* `context.build`, and re-validate stale identity in
|
|
27
|
+
* `reconstructPrincipal`).
|
|
28
|
+
*/
|
|
29
|
+
export {};
|
|
30
|
+
//# sourceMappingURL=principal.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"principal.js","sourceRoot":"","sources":["../src/principal.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { type ModelMessage } from "ai";
|
|
2
|
+
import type { AppMessage } from "./message.js";
|
|
3
|
+
/**
|
|
4
|
+
* The two message projections (see `message.ts`). `AppMessage` is the stored
|
|
5
|
+
* source of truth; these derive what each *audience* sees, reading the two
|
|
6
|
+
* metadata axes. Both default permissively: absent `visibility` → shown to the
|
|
7
|
+
* user; absent `sendToModel` → fed to the model. Exceptions are explicit.
|
|
8
|
+
*/
|
|
9
|
+
/** Client projection: keep user-visible messages, drop `internal`/`redacted`. */
|
|
10
|
+
export declare function toClientMessages(messages: AppMessage[]): AppMessage[];
|
|
11
|
+
/** The part type that marks a compaction boundary (see `sliceAtCompaction`). */
|
|
12
|
+
export declare const COMPACTION_PART_TYPE = "data-compaction";
|
|
13
|
+
/**
|
|
14
|
+
* The live context window: everything from the LAST compaction marker onward.
|
|
15
|
+
*
|
|
16
|
+
* `compactChat` summarizes a conversation and appends the summary as one message
|
|
17
|
+
* carrying a `data-compaction` part; from then on the model sees that summary
|
|
18
|
+
* instead of the turns it replaces, while stored history — and the client
|
|
19
|
+
* projection — keep everything. No marker → the messages unchanged, so chats
|
|
20
|
+
* that were never compacted are untouched.
|
|
21
|
+
*
|
|
22
|
+
* Slicing at a marker can orphan a `tool_result` from its `tool_use` only if the
|
|
23
|
+
* two straddled the boundary, which can't happen: the marker is appended at the
|
|
24
|
+
* tail of a settled conversation, so a call and its result are always on the
|
|
25
|
+
* same side of it.
|
|
26
|
+
*/
|
|
27
|
+
export declare function sliceAtCompaction(messages: AppMessage[]): AppMessage[];
|
|
28
|
+
/**
|
|
29
|
+
* Model projection: cut to the live context window, drop `sendToModel === false`,
|
|
30
|
+
* neutralize audio file parts, close any dangling tool calls, then convert to
|
|
31
|
+
* model messages. (`data-*` parts — e.g. the compaction marker — are dropped by
|
|
32
|
+
* the SDK during conversion, so they never reach the model.)
|
|
33
|
+
*/
|
|
34
|
+
export declare function toModelMessages(messages: AppMessage[], opts?: {
|
|
35
|
+
tools?: Record<string, unknown>;
|
|
36
|
+
}): Promise<ModelMessage[]>;
|
|
37
|
+
//# sourceMappingURL=projections.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"projections.d.ts","sourceRoot":"","sources":["../src/projections.ts"],"names":[],"mappings":"AAAA,OAAO,EAA0B,KAAK,YAAY,EAAE,MAAM,IAAI,CAAC;AAC/D,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,cAAc,CAAC;AAE/C;;;;;GAKG;AAEH,iFAAiF;AACjF,wBAAgB,gBAAgB,CAAC,QAAQ,EAAE,UAAU,EAAE,GAAG,UAAU,EAAE,CAErE;AAmFD,gFAAgF;AAChF,eAAO,MAAM,oBAAoB,oBAAoB,CAAC;AAEtD;;;;;;;;;;;;;GAaG;AACH,wBAAgB,iBAAiB,CAAC,QAAQ,EAAE,UAAU,EAAE,GAAG,UAAU,EAAE,CAQtE;AAED;;;;;GAKG;AACH,wBAAsB,eAAe,CACnC,QAAQ,EAAE,UAAU,EAAE,EACtB,IAAI,CAAC,EAAE;IAAE,KAAK,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAA;CAAE,GACzC,OAAO,CAAC,YAAY,EAAE,CAAC,CAOzB"}
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
import { convertToModelMessages } from "ai";
|
|
2
|
+
/**
|
|
3
|
+
* The two message projections (see `message.ts`). `AppMessage` is the stored
|
|
4
|
+
* source of truth; these derive what each *audience* sees, reading the two
|
|
5
|
+
* metadata axes. Both default permissively: absent `visibility` → shown to the
|
|
6
|
+
* user; absent `sendToModel` → fed to the model. Exceptions are explicit.
|
|
7
|
+
*/
|
|
8
|
+
/** Client projection: keep user-visible messages, drop `internal`/`redacted`. */
|
|
9
|
+
export function toClientMessages(messages) {
|
|
10
|
+
return messages.filter((m) => (m.metadata?.visibility ?? "user") === "user");
|
|
11
|
+
}
|
|
12
|
+
/**
|
|
13
|
+
* Close any tool call left without a result — an undecided `approval-requested`
|
|
14
|
+
* part, or a tool that was still running when the turn was interrupted (e.g. the
|
|
15
|
+
* user sent a new message before it finished). `convertToModelMessages` throws
|
|
16
|
+
* (and Anthropic rejects) on a `tool_use` with no following `tool_result`, so we
|
|
17
|
+
* give such parts a synthetic terminal output. Projection-only — the stored
|
|
18
|
+
* history is untouched; the synthetic result just keeps the model view valid.
|
|
19
|
+
*/
|
|
20
|
+
function closeDanglingToolCalls(messages) {
|
|
21
|
+
const lastIndex = messages.length - 1;
|
|
22
|
+
return messages.map((m, i) => {
|
|
23
|
+
const parts = m.parts;
|
|
24
|
+
if (m.role !== "assistant" || !Array.isArray(parts))
|
|
25
|
+
return m;
|
|
26
|
+
// Tail = the final message of the history: the continuation the tool loop
|
|
27
|
+
// re-runs. Only there will an approved-but-unexecuted call actually execute.
|
|
28
|
+
const isTail = i === lastIndex;
|
|
29
|
+
let changed = false;
|
|
30
|
+
const next = parts.map((p) => {
|
|
31
|
+
const type = typeof p.type === "string" ? p.type : "";
|
|
32
|
+
const isToolCall = type === "dynamic-tool" || type.startsWith("tool-");
|
|
33
|
+
if (!isToolCall)
|
|
34
|
+
return p;
|
|
35
|
+
const state = p.state;
|
|
36
|
+
const approval = p.approval;
|
|
37
|
+
// Close ONLY calls that nothing will resolve:
|
|
38
|
+
// - an interrupted execution (a call with no result, no approval flow),
|
|
39
|
+
// - an UN-decided approval (abandoned by a new message instead of a decision), or
|
|
40
|
+
// - an APPROVED call with no result that is no longer the tail — its
|
|
41
|
+
// continuation was interrupted (crash/restart mid-execution) and a later
|
|
42
|
+
// message buried it, so no re-run will ever execute it.
|
|
43
|
+
// Leave tail-position approved calls and denied approvals alone — the
|
|
44
|
+
// re-run executes the former; the SDK synthesizes a denial result for the latter.
|
|
45
|
+
const interrupted = state === "input-available" || state === "input-streaming";
|
|
46
|
+
const undecidedApproval = state === "approval-requested" && (!approval || approval.approved === undefined);
|
|
47
|
+
const buriedApproved = !isTail &&
|
|
48
|
+
state === "approval-responded" &&
|
|
49
|
+
approval?.approved === true &&
|
|
50
|
+
p.output == null;
|
|
51
|
+
if (!interrupted && !undecidedApproval && !buriedApproved)
|
|
52
|
+
return p;
|
|
53
|
+
changed = true;
|
|
54
|
+
return {
|
|
55
|
+
...p,
|
|
56
|
+
state: "output-available",
|
|
57
|
+
output: p.output ?? {
|
|
58
|
+
interrupted: true,
|
|
59
|
+
reason: "Not completed — superseded by a new message before this tool finished.",
|
|
60
|
+
},
|
|
61
|
+
};
|
|
62
|
+
});
|
|
63
|
+
return changed ? { ...m, parts: next } : m;
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Replace audio file parts with a text placeholder. Voice messages ingested
|
|
68
|
+
* before channel-side transcription existed were stored as `audio/*` file
|
|
69
|
+
* parts, and a chat model without audio input (Anthropic) rejects the whole
|
|
70
|
+
* prompt — bricking every later turn of that chat. Projection-only — the
|
|
71
|
+
* stored history keeps the original part.
|
|
72
|
+
*/
|
|
73
|
+
function replaceAudioFileParts(messages) {
|
|
74
|
+
return messages.map((m) => {
|
|
75
|
+
const parts = m.parts;
|
|
76
|
+
if (!Array.isArray(parts))
|
|
77
|
+
return m;
|
|
78
|
+
let changed = false;
|
|
79
|
+
const next = parts.map((p) => {
|
|
80
|
+
const isAudio = p.type === "file" && typeof p.mediaType === "string" && p.mediaType.startsWith("audio/");
|
|
81
|
+
if (!isAudio)
|
|
82
|
+
return p;
|
|
83
|
+
changed = true;
|
|
84
|
+
const name = typeof p.filename === "string" ? p.filename : "audio";
|
|
85
|
+
return {
|
|
86
|
+
type: "text",
|
|
87
|
+
text: `[A voice message ("${name}") was attached here but could not be included — audio input is not supported.]`,
|
|
88
|
+
};
|
|
89
|
+
});
|
|
90
|
+
return changed ? { ...m, parts: next } : m;
|
|
91
|
+
});
|
|
92
|
+
}
|
|
93
|
+
/** The part type that marks a compaction boundary (see `sliceAtCompaction`). */
|
|
94
|
+
export const COMPACTION_PART_TYPE = "data-compaction";
|
|
95
|
+
/**
|
|
96
|
+
* The live context window: everything from the LAST compaction marker onward.
|
|
97
|
+
*
|
|
98
|
+
* `compactChat` summarizes a conversation and appends the summary as one message
|
|
99
|
+
* carrying a `data-compaction` part; from then on the model sees that summary
|
|
100
|
+
* instead of the turns it replaces, while stored history — and the client
|
|
101
|
+
* projection — keep everything. No marker → the messages unchanged, so chats
|
|
102
|
+
* that were never compacted are untouched.
|
|
103
|
+
*
|
|
104
|
+
* Slicing at a marker can orphan a `tool_result` from its `tool_use` only if the
|
|
105
|
+
* two straddled the boundary, which can't happen: the marker is appended at the
|
|
106
|
+
* tail of a settled conversation, so a call and its result are always on the
|
|
107
|
+
* same side of it.
|
|
108
|
+
*/
|
|
109
|
+
export function sliceAtCompaction(messages) {
|
|
110
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
111
|
+
const parts = messages[i].parts;
|
|
112
|
+
if (Array.isArray(parts) && parts.some((p) => p.type === COMPACTION_PART_TYPE)) {
|
|
113
|
+
return messages.slice(i);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
return messages;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Model projection: cut to the live context window, drop `sendToModel === false`,
|
|
120
|
+
* neutralize audio file parts, close any dangling tool calls, then convert to
|
|
121
|
+
* model messages. (`data-*` parts — e.g. the compaction marker — are dropped by
|
|
122
|
+
* the SDK during conversion, so they never reach the model.)
|
|
123
|
+
*/
|
|
124
|
+
export async function toModelMessages(messages, opts) {
|
|
125
|
+
const forModel = closeDanglingToolCalls(replaceAudioFileParts(sliceAtCompaction(messages).filter((m) => m.metadata?.sendToModel !== false)));
|
|
126
|
+
return convertToModelMessages(forModel, (opts ?? {}));
|
|
127
|
+
}
|
|
128
|
+
//# sourceMappingURL=projections.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"projections.js","sourceRoot":"","sources":["../src/projections.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,sBAAsB,EAAqB,MAAM,IAAI,CAAC;AAG/D;;;;;GAKG;AAEH,iFAAiF;AACjF,MAAM,UAAU,gBAAgB,CAAC,QAAsB;IACrD,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,QAAQ,EAAE,UAAU,IAAI,MAAM,CAAC,KAAK,MAAM,CAAC,CAAC;AAC/E,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,sBAAsB,CAAC,QAAsB;IACpD,MAAM,SAAS,GAAG,QAAQ,CAAC,MAAM,GAAG,CAAC,CAAC;IACtC,OAAO,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QAC3B,MAAM,KAAK,GAAI,CAA2C,CAAC,KAAK,CAAC;QACjE,IAAI,CAAC,CAAC,IAAI,KAAK,WAAW,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YAAE,OAAO,CAAC,CAAC;QAC9D,0EAA0E;QAC1E,6EAA6E;QAC7E,MAAM,MAAM,GAAG,CAAC,KAAK,SAAS,CAAC;QAC/B,IAAI,OAAO,GAAG,KAAK,CAAC;QACpB,MAAM,IAAI,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE;YAC3B,MAAM,IAAI,GAAG,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC;YACtD,MAAM,UAAU,GAAG,IAAI,KAAK,cAAc,IAAI,IAAI,CAAC,UAAU,CAAC,OAAO,CAAC,CAAC;YACvE,IAAI,CAAC,UAAU;gBAAE,OAAO,CAAC,CAAC;YAC1B,MAAM,KAAK,GAAG,CAAC,CAAC,KAA2B,CAAC;YAC5C,MAAM,QAAQ,GAAG,CAAC,CAAC,QAA8C,CAAC;YAClE,8CAA8C;YAC9C,yEAAyE;YACzE,mFAAmF;YACnF,sEAAsE;YACtE,4EAA4E;YAC5E,2DAA2D;YAC3D,sEAAsE;YACtE,kFAAkF;YAClF,MAAM,WAAW,GAAG,KAAK,KAAK,iBAAiB,IAAI,KAAK,KAAK,iBAAiB,CAAC;YAC/E,MAAM,iBAAiB,GACrB,KAAK,KAAK,oBAAoB,IAAI,CAAC,CAAC,QAAQ,IAAI,QAAQ,CAAC,QAAQ,KAAK,SAAS,CAAC,CAAC;YACnF,MAAM,cAAc,GAClB,CAAC,MAAM;gBACP,KAAK,KAAK,oBAAoB;gBAC9B,QAAQ,EAAE,QAAQ,KAAK,IAAI;gBAC3B,CAAC,CAAC,MAAM,IAAI,IAAI,CAAC;YACnB,IAAI,CAAC,WAAW,IAAI,CAAC,iBAAiB,IAAI,CAAC,cAAc;gBAAE,OAAO,CAAC,CAAC;YACpE,OAAO,GAAG,IAAI,CAAC;YACf,OAAO;gBACL,GAAG,CAAC;gBACJ,KAAK,EAAE,kBAAkB;gBACzB,MAAM,EAAE,CAAC,CAAC,MAAM,IAAI;oBAClB,WAAW,EAAE,IAAI;oBACjB,MAAM,EAAE,wEAAwE;iBACjF;aACF,CAAC;QACJ,CAAC,CAAC,CAAC;QACH,OAAO,OAAO,CAAC,CAAC,CAAE,EAAE,GAAG,CAAC,EAAE,KAAK,EAAE,IAAI,EAAiB,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7D,CAAC,CAAC,CAAC;AACL,CAAC;AAED;;;;;;GAMG;AACH,SAAS,qBAAqB,CAAC,QAAsB;IACnD,OAAO,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE;QACxB,MAAM,KAAK,GAAI,CAA2C,CAAC,KAAK,CAAC;QACjE,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;YAAE,OAAO,CAAC,CAAC;QACpC,IAAI,OAAO,GAAG,KAAK,CAAC;QACpB,MAAM,IAAI,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE;YAC3B,MAAM,OAAO,GACX,CAAC,CAAC,IAAI,KAAK,MAAM,IAAI,OAAO,CAAC,CAAC,SAAS,KAAK,QAAQ,IAAI,CAAC,CAAC,SAAS,CAAC,UAAU,CAAC,QAAQ,CAAC,CAAC;YAC3F,IAAI,CAAC,OAAO;gBAAE,OAAO,CAAC,CAAC;YACvB,OAAO,GAAG,IAAI,CAAC;YACf,MAAM,IAAI,GAAG,OAAO,CAAC,CAAC,QAAQ,KAAK,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,OAAO,CAAC;YACnE,OAAO;gBACL,IAAI,EAAE,MAAM;gBACZ,IAAI,EAAE,sBAAsB,IAAI,iFAAiF;aAClH,CAAC;QACJ,CAAC,CAAC,CAAC;QACH,OAAO,OAAO,CAAC,CAAC,CAAE,EAAE,GAAG,CAAC,EAAE,KAAK,EAAE,IAAI,EAAiB,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7D,CAAC,CAAC,CAAC;AACL,CAAC;AAED,gFAAgF;AAChF,MAAM,CAAC,MAAM,oBAAoB,GAAG,iBAAiB,CAAC;AAEtD;;;;;;;;;;;;;GAaG;AACH,MAAM,UAAU,iBAAiB,CAAC,QAAsB;IACtD,KAAK,IAAI,CAAC,GAAG,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;QAC9C,MAAM,KAAK,GAAI,QAAQ,CAAC,CAAC,CAAsC,CAAC,KAAK,CAAC;QACtE,IAAI,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC,IAAI,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,KAAK,oBAAoB,CAAC,EAAE,CAAC;YAC/E,OAAO,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC;QAC3B,CAAC;IACH,CAAC;IACD,OAAO,QAAQ,CAAC;AAClB,CAAC;AAED;;;;;GAKG;AACH,MAAM,CAAC,KAAK,UAAU,eAAe,CACnC,QAAsB,EACtB,IAA0C;IAE1C,MAAM,QAAQ,GAAG,sBAAsB,CACrC,qBAAqB,CACnB,iBAAiB,CAAC,QAAQ,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,QAAQ,EAAE,WAAW,KAAK,KAAK,CAAC,CAC7E,CACF,CAAC;IACF,OAAO,sBAAsB,CAAC,QAAiB,EAAE,CAAC,IAAI,IAAI,EAAE,CAAU,CAAC,CAAC;AAC1E,CAAC"}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Prompt-caching plans: which provider options a turn should set so the model
|
|
3
|
+
* provider caches the stable prompt prefix (tool definitions, system prompt,
|
|
4
|
+
* conversation history) instead of re-billing it every step.
|
|
5
|
+
*
|
|
6
|
+
* Kept provider-generic in core: the PLATFORM decides which provider a model id
|
|
7
|
+
* belongs to (it owns the catalog) and calls `promptCachingPlanFor` from its
|
|
8
|
+
* `RuntimeConfig.resolvePromptCaching` hook; core only applies the plan in
|
|
9
|
+
* `buildTurn`. Providers not named here (e.g. Google: implicit caching only)
|
|
10
|
+
* get no plan and are untouched.
|
|
11
|
+
*
|
|
12
|
+
* Anthropic layout — three of the four allowed `cache_control` breakpoints:
|
|
13
|
+
* 1. The LAST function tool definition. Tools render before the system prompt
|
|
14
|
+
* in the request, so this keeps the (often large, MCP-heavy) tool prefix
|
|
15
|
+
* cached even when the system block changes mid-conversation (e.g. a
|
|
16
|
+
* memory-index write).
|
|
17
|
+
* 2. The system message that holds the agent instructions + memory block. The
|
|
18
|
+
* day-granular date line is a SEPARATE, unmarked system message after it,
|
|
19
|
+
* so the daily date flip never invalidates this breakpoint.
|
|
20
|
+
* 3. A request-level (top-level) `cache_control` — Anthropic's automatic
|
|
21
|
+
* caching: the API places a breakpoint on the last cacheable block of the
|
|
22
|
+
* request. The AI SDK re-sends call-level providerOptions on every step of
|
|
23
|
+
* the tool loop, so this breakpoint ROLLS to the newest tool result with no
|
|
24
|
+
* per-step code. Note: top-level cache_control is a first-party Claude API
|
|
25
|
+
* feature (not Bedrock/Vertex); block-level markers work everywhere. Also
|
|
26
|
+
* note the SDK's client-side breakpoint validator counts only block-level
|
|
27
|
+
* markers, so the request-level one is invisible to it — harmless.
|
|
28
|
+
*
|
|
29
|
+
* TTL is left at the provider default (5m). Agents whose turns arrive slower
|
|
30
|
+
* than that won't benefit — that's surfaced by the platform's caching lint,
|
|
31
|
+
* not worked around here.
|
|
32
|
+
*/
|
|
33
|
+
export interface PromptCachingPlan {
|
|
34
|
+
/**
|
|
35
|
+
* Call-level providerOptions (merged with e.g. reasoning options). Anthropic:
|
|
36
|
+
* the rolling request-level breakpoint. OpenAI: `promptCacheKey` for cache
|
|
37
|
+
* routing (OpenAI caches automatically; the key only improves shard routing).
|
|
38
|
+
*/
|
|
39
|
+
requestProviderOptions?: Record<string, Record<string, unknown>>;
|
|
40
|
+
/**
|
|
41
|
+
* providerOptions for the system message carrying the agent instructions.
|
|
42
|
+
* Presence also signals `buildTurn` to split the system prompt into two
|
|
43
|
+
* messages so volatile trailing content (the date line) lands AFTER the
|
|
44
|
+
* breakpoint.
|
|
45
|
+
*/
|
|
46
|
+
systemProviderOptions?: Record<string, Record<string, unknown>>;
|
|
47
|
+
/**
|
|
48
|
+
* providerOptions to place on the LAST active function tool definition
|
|
49
|
+
* (a breakpoint covering the whole tools prefix). Applied via
|
|
50
|
+
* `markLastFunctionTool`.
|
|
51
|
+
*/
|
|
52
|
+
toolProviderOptions?: Record<string, Record<string, unknown>>;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* The caching plan for a provider family (`"anthropic"`, `"openai"`, …).
|
|
56
|
+
* `cacheKeySuffix` scopes OpenAI's cache routing key beyond the agent name —
|
|
57
|
+
* pass a per-principal key for agents whose system prompt is per-principal
|
|
58
|
+
* (e.g. memory-enabled agents), so structurally different prefixes don't
|
|
59
|
+
* co-route to one shard.
|
|
60
|
+
*/
|
|
61
|
+
export declare function promptCachingPlanFor(provider: string | undefined, opts: {
|
|
62
|
+
agent: string;
|
|
63
|
+
cacheKeySuffix?: string;
|
|
64
|
+
}): PromptCachingPlan | undefined;
|
|
65
|
+
/**
|
|
66
|
+
* The DEFAULT caching plan, used when the runtime has no `resolvePromptCaching`
|
|
67
|
+
* hook: derive the provider family from the model object itself — every AI SDK
|
|
68
|
+
* `LanguageModel` carries a `provider` string (`"anthropic.messages"`,
|
|
69
|
+
* `"openai.responses"`, …) — and apply {@link promptCachingPlanFor}. Measured
|
|
70
|
+
* on a real MCP-heavy agent, this is a ~6x difference in unit cost (a ~32k-token
|
|
71
|
+
* tool catalogue re-billed every step vs read from cache), which is not a
|
|
72
|
+
* default a customer should have to discover by reading source.
|
|
73
|
+
*
|
|
74
|
+
* Deliberately conservative: only FIRST-PARTY Anthropic/OpenAI provider ids
|
|
75
|
+
* match (Bedrock/Vertex ids don't contain these substrings), because the
|
|
76
|
+
* Anthropic plan's request-level `cache_control` is a first-party feature.
|
|
77
|
+
* Unknown providers get no plan and are untouched. Opt out entirely with
|
|
78
|
+
* `resolvePromptCaching: () => undefined`, or per turn with
|
|
79
|
+
* `promptCaching: false`.
|
|
80
|
+
*/
|
|
81
|
+
export declare function defaultPromptCachingPlan(model: unknown, agent: string): PromptCachingPlan | undefined;
|
|
82
|
+
/**
|
|
83
|
+
* One-level-deep merge of providerOptions objects (per provider key), so e.g.
|
|
84
|
+
* `{ anthropic: { thinking } }` from reasoning and `{ anthropic: { cacheControl } }`
|
|
85
|
+
* from caching coexist — a plain spread would clobber one with the other. Later
|
|
86
|
+
* sources win on conflicting inner keys. Returns undefined when nothing to merge.
|
|
87
|
+
*/
|
|
88
|
+
export declare function mergeProviderOptions(...sources: Array<Record<string, unknown> | undefined>): Record<string, unknown> | undefined;
|
|
89
|
+
/**
|
|
90
|
+
* Stable fingerprint of a toolset (FNV-1a 64-bit over the sorted names) —
|
|
91
|
+
* equality is all that matters (same hash ⇔ same tool names), so a
|
|
92
|
+
* dependency-free non-cryptographic hash beats pulling in node:crypto (core
|
|
93
|
+
* stays runtime-agnostic). Stored on the run row to detect toolset churn
|
|
94
|
+
* across a chat's turns, one of the main prompt-cache invalidators.
|
|
95
|
+
*/
|
|
96
|
+
export declare function toolsetHash(names: Iterable<string>): string;
|
|
97
|
+
/**
|
|
98
|
+
* Mark the LAST active function tool with the given providerOptions (in place).
|
|
99
|
+
* Skips provider-executed tools (`type: "provider"`, e.g. Anthropic's native
|
|
100
|
+
* web_search/web_fetch) — the provider silently drops `cache_control` on those —
|
|
101
|
+
* and, when `activeNames` is given, tools the turn filtered out via
|
|
102
|
+
* `activeTools` (a marker on an unsent tool would vanish). Returns the marked
|
|
103
|
+
* tool name, or undefined when no eligible tool exists.
|
|
104
|
+
*/
|
|
105
|
+
export declare function markLastFunctionTool(tools: Record<string, unknown>, providerOptions: Record<string, Record<string, unknown>>, activeNames?: readonly string[]): string | undefined;
|
|
106
|
+
//# sourceMappingURL=prompt-caching.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"prompt-caching.d.ts","sourceRoot":"","sources":["../src/prompt-caching.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA+BG;AAKH,MAAM,WAAW,iBAAiB;IAChC;;;;OAIG;IACH,sBAAsB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC;IACjE;;;;;OAKG;IACH,qBAAqB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC;IAChE;;;;OAIG;IACH,mBAAmB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC;CAC/D;AAED;;;;;;GAMG;AACH,wBAAgB,oBAAoB,CAClC,QAAQ,EAAE,MAAM,GAAG,SAAS,EAC5B,IAAI,EAAE;IAAE,KAAK,EAAE,MAAM,CAAC;IAAC,cAAc,CAAC,EAAE,MAAM,CAAA;CAAE,GAC/C,iBAAiB,GAAG,SAAS,CAoB/B;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,wBAAwB,CACtC,KAAK,EAAE,OAAO,EACd,KAAK,EAAE,MAAM,GACZ,iBAAiB,GAAG,SAAS,CAW/B;AAaD;;;;;GAKG;AACH,wBAAgB,oBAAoB,CAClC,GAAG,OAAO,EAAE,KAAK,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,GAAG,SAAS,CAAC,GACrD,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,GAAG,SAAS,CAerC;AAED;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,QAAQ,CAAC,MAAM,CAAC,GAAG,MAAM,CAE3D;AAED;;;;;;;GAOG;AACH,wBAAgB,oBAAoB,CAClC,KAAK,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC9B,eAAe,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,EACxD,WAAW,CAAC,EAAE,SAAS,MAAM,EAAE,GAC9B,MAAM,GAAG,SAAS,CAmBpB"}
|