@sema-agent/server 7.80.1 → 7.80.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@
9
9
  * (`runnerTierFrozen` 判据就取自构造那一刻的 `config.tiers`)。
10
10
  * `getRunStore()` 是唯一的晚绑取值(runStore 在本段之后才构造),原文靠同作用域前向引用。
11
11
  */
12
- import { type RunnerDeps, type WorkflowRunStore, type AskOutcome } from "@sema-agent/core";
12
+ import { type CacheBreakFinding, type RunnerDeps, type WorkflowRunStore, type AskOutcome } from "@sema-agent/core";
13
13
  import type { createBrain } from "../brain.js";
14
14
  import type { buildPricing, createTracer } from "../budget.js";
15
15
  import type { ServiceConfig } from "../config.js";
@@ -60,6 +60,22 @@ export declare function createRunnerDepsOnAsk(toolApproval: ToolApprovalCoordina
60
60
  * `classificationKnown:false`(指标标签折 `unknown`),不吞不编;与安装包的锁步由 test/boot-runner-deps-shared-base.test.ts
61
61
  * 的 dist 扫描格钉住(core 加词/收进 d.ts 那天先红)。 */
62
62
  export declare const CONFIG_ADVISORY_CLASSIFICATIONS: ReadonlySet<string>;
63
+ /** S-386:core 的 [ref] 前缀缓存**断裂**探测器交给宿主的根因**闭集**,以及每个词的日志级别。
64
+ * 闭集**不抄词**:词源直接取 core 公开导出的 `CacheBreakFinding["cause"]`(`@sema-agent/core` 根导出,
65
+ * `dist/index.d.ts:128`;实现在 `dist/core/cache-break-detector.d.ts`),于是 core 加词/删词 ⇒
66
+ * `Record<PromptCacheBreakCause, …>` 少键/多键 = **编译错**(CLAUDE.md [ref]「词表是闭集 + 穷举」)。
67
+ * ⚠️ 首版在这里**手抄**了五个字面量,并在本注与 CHANGELOG 里声称「core 加词 = 编译错」—— 复审实测
68
+ * 证伪(给 core d.ts 加词后 tsc 照绿,只有一格正则扫描会红):抄来的联合与 core 无编译期联系。
69
+ * 运行期收到词表外的值(`ctx.classification` 在 core 的 onError 契约上只是 `string`)走最响的那一档
70
+ * (error)并标 `classificationKnown:false`,退化方向永远是**多一条告警**。
71
+ *
72
+ * 🔴 定级判据是**事实**,不是 provider 名单:`dist/core/cache-break-detector.js` 的 `observe()` 里
73
+ * `const before = prev.cacheRead; if (before <= 0) return undefined` —— 探测器只有在**上一轮真的从
74
+ * 前缀缓存读到过 token** 时才会给出 finding。所以「带 classification」这件事本身就证明了「这条路由
75
+ * 在服务前缀缓存」,前缀 bug 四词(model-switch / tool-schema / tool-set / system-prefix)落 error
76
+ * 是有据的;`server-or-ttl` 在 agentic 任务里是常态(慢工具 ⇒ 5min+ 间隔,provider 缓存到期)⇒ warn。 */
77
+ export type PromptCacheBreakCause = CacheBreakFinding["cause"];
78
+ export declare const PROMPT_CACHE_BREAK_CAUSE_LEVEL: Readonly<Record<PromptCacheBreakCause, "error" | "warn">>;
63
79
  export interface RunnerDepsCtx {
64
80
  config: ServiceConfig;
65
81
  logger: Logger;
@@ -24,6 +24,14 @@ export const CONFIG_ADVISORY_CLASSIFICATIONS = new Set([
24
24
  "shared-memory-not-mounted",
25
25
  "shell-gate-off",
26
26
  ]);
27
+ export const PROMPT_CACHE_BREAK_CAUSE_LEVEL = {
28
+ "model-switch": "error",
29
+ "tool-schema": "error",
30
+ "tool-set": "error",
31
+ "system-prefix": "error",
32
+ "server-or-ttl": "warn",
33
+ };
34
+ const PROMPT_CACHE_BREAK_LEVELS = new Map(Object.entries(PROMPT_CACHE_BREAK_CAUSE_LEVEL));
27
35
  const CEILING_ORIGINS = new Set(["walltime", "turns"]);
28
36
  export function createEngineNoticeForwarder(logger, router = defaultEngineNoticeRouter, metrics) {
29
37
  return (notice) => {
@@ -224,9 +232,20 @@ export function createRunnerDeps(ctx) {
224
232
  return;
225
233
  }
226
234
  if (ctx.phase === "prompt-cache") {
227
- metrics.inc("prompt_cache_low_hit_total");
228
- const fields = { sessionId: ctx.sessionId, ...(ctx.classification ? { classification: ctx.classification } : {}), err: String(err) };
229
- if (ctx.classification === "server-or-ttl")
235
+ const classification = ctx.classification;
236
+ const causeLevel = classification === undefined ? undefined : PROMPT_CACHE_BREAK_LEVELS.get(classification);
237
+ const level = classification === undefined ? "warn" : (causeLevel ?? "error");
238
+ const fields = {
239
+ sessionId: ctx.sessionId,
240
+ confirmed: classification !== undefined,
241
+ ...(classification !== undefined ? { classification, classificationKnown: causeLevel !== undefined } : {}),
242
+ err: String(err),
243
+ };
244
+ metrics.inc("prompt_cache_low_hit_total", {
245
+ ...(classification !== undefined ? { classification: causeLevel !== undefined ? classification : "unknown" } : {}),
246
+ level,
247
+ });
248
+ if (level === "warn")
230
249
  logger.warn("prompt_cache_break", fields);
231
250
  else
232
251
  logger.error("prompt_cache_break", fields);
@@ -206,7 +206,7 @@ export function createMetrics() {
206
206
  m.histogram("task_cache_hit_rate", "Prefix-cache hit rate per task (core-computed stats.cacheHitRate, 0..1)", [
207
207
  0.05, 0.1, 0.25, 0.5, 0.75, 0.9, 0.95, 1,
208
208
  ]);
209
- m.counter("prompt_cache_low_hit_total", "Tasks the Runner flagged with a low prefix-cache hit rate");
209
+ m.counter("prompt_cache_low_hit_total", "Prefix-cache advisories the Runner handed the host, by {classification,level} (S-386: the design/31 break causes carry a classification; the task-end low-hit heuristic and the two cache-family wiring advisories carry none)");
210
210
  m.counter("model_cost_micro_usd_total", "Authoritative model spend (integer micro-USD) over all brain calls, by model");
211
211
  m.histogram("brain_first_token_ms", "Time to first content token per brain call (gateway-hang signal)", [
212
212
  50, 100, 250, 500, 1000, 2000, 5000, 10000, 30000,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@sema-agent/server",
3
- "version": "7.80.1",
3
+ "version": "7.80.2",
4
4
  "description": "Sema Server — the server/API implementation layer for Sema, wiring core, registry, model providers, and cloud agent execution. Built on @sema-agent/core.",
5
5
  "type": "module",
6
6
  "license": "BUSL-1.1",