@vellumai/assistant 0.8.11 → 0.8.12-staging.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ARCHITECTURE.md +15 -17
- package/README.md +0 -6
- package/bun.lock +6 -122
- package/node_modules/@vellumai/gateway-client/bun.lock +1 -0
- package/node_modules/@vellumai/gateway-client/package.json +3 -1
- package/node_modules/@vellumai/gateway-client/src/__tests__/gateway-client.test.ts +1 -1
- package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +87 -0
- package/node_modules/@vellumai/gateway-client/src/index.ts +3 -5
- package/openapi.yaml +633 -4
- package/package.json +1 -3
- package/src/__tests__/adaptive-thinking-repair.test.ts +185 -0
- package/src/__tests__/agent-loop-compaction-events.test.ts +7 -6
- package/src/__tests__/anthropic-provider.test.ts +129 -0
- package/src/__tests__/background-workers-disk-pressure.test.ts +4 -1
- package/src/__tests__/btw-routes.test.ts +7 -34
- package/src/__tests__/checker.test.ts +6 -12
- package/src/__tests__/config-loader-backfill.test.ts +4 -2
- package/src/__tests__/config-loader-quarantine-notice.test.ts +167 -0
- package/src/__tests__/config-watcher.test.ts +2 -2
- package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +1 -1
- package/src/__tests__/conversation-error.test.ts +2 -6
- package/src/__tests__/conversation-history-web-search.test.ts +8 -0
- package/src/__tests__/conversation-title-service.test.ts +2 -1
- package/src/__tests__/credential-security-invariants.test.ts +1 -1
- package/src/__tests__/disk-pressure-tools.test.ts +1 -1
- package/src/__tests__/exploration-drift-hook.test.ts +692 -0
- package/src/__tests__/filing-service.test.ts +8 -3
- package/src/__tests__/guardian-action-store.test.ts +0 -167
- package/src/__tests__/handlers-skills-memory-v2-reseed.test.ts +1 -1
- package/src/__tests__/heartbeat-disk-pressure.test.ts +4 -1
- package/src/__tests__/heartbeat-service.test.ts +5 -2
- package/src/__tests__/identity-intro-cache.test.ts +12 -5
- package/src/__tests__/identity-routes.test.ts +16 -57
- package/src/__tests__/injector-chain.test.ts +8 -3
- package/src/__tests__/injector-config-quarantine-notice.test.ts +115 -0
- package/src/__tests__/llm-catalog-parity.test.ts +16 -0
- package/src/__tests__/llm-usage-store.test.ts +11 -0
- package/src/__tests__/log-export-workspace.test.ts +468 -3
- package/src/__tests__/memory-v2-static-injector.test.ts +22 -0
- package/src/__tests__/model-intents.test.ts +1 -1
- package/src/__tests__/oauth-cli.test.ts +19 -8
- package/src/__tests__/openai-provider.test.ts +34 -0
- package/src/__tests__/prechat-onboarding-contract.test.ts +0 -1
- package/src/__tests__/recurrence-engine.test.ts +45 -0
- package/src/__tests__/schedule-routes.test.ts +34 -0
- package/src/__tests__/scheduler-disk-pressure.test.ts +1 -1
- package/src/__tests__/script-proxy-conversation-manager.test.ts +10 -5
- package/src/__tests__/secret-fixtures.ts +20 -0
- package/src/__tests__/skill-tool-factory.test.ts +49 -0
- package/src/__tests__/subagent-role-registry.test.ts +24 -1
- package/src/__tests__/subagent-tools.test.ts +1 -0
- package/src/__tests__/system-prompt.test.ts +109 -11
- package/src/__tests__/tool-approval-handler.test.ts +85 -0
- package/src/__tests__/tool-audit-listener.test.ts +86 -0
- package/src/__tests__/tool-error-hook.test.ts +1 -0
- package/src/__tests__/tool-result-spool.test.ts +337 -0
- package/src/__tests__/tool-result-truncate-hook.test.ts +1 -0
- package/src/__tests__/validate-input.test.ts +95 -1
- package/src/__tests__/workspace-migration-098-remove-stale-updates-bulletin-file.test.ts +65 -0
- package/src/__tests__/workspace-migration-099-disable-cache-one-shot-callsites.test.ts +139 -0
- package/src/__tests__/workspace-migration-100-upgrade-quality-profile-to-fable-5.test.ts +174 -0
- package/src/__tests__/workspace-migration-101-upgrade-balanced-economy-to-minimax-m3.test.ts +162 -0
- package/src/__tests__/workspace-release-notes-feature-flag-guard.test.ts +45 -95
- package/src/acp/__tests__/agent-process.test.ts +315 -2
- package/src/acp/__tests__/prepare-agent-env.test.ts +79 -5
- package/src/acp/agent-process.ts +163 -34
- package/src/acp/prepare-agent-env.ts +55 -15
- package/src/agent/loop.ts +81 -24
- package/src/api/events/usage-progress.ts +28 -0
- package/src/api/index.ts +6 -0
- package/src/background-wake/wake-intent-hooks.test.ts +2 -0
- package/src/bundler/app-bundler.ts +25 -42
- package/src/bundler/app-compiler.ts +8 -0
- package/src/calls/call-controller.ts +1 -1
- package/src/cli/commands/plugins.ts +248 -15
- package/src/cli/lib/__tests__/inspect-plugin.test.ts +318 -0
- package/src/cli/lib/__tests__/install-from-github.test.ts +16 -9
- package/src/cli/lib/__tests__/plugin-artifact.test.ts +183 -0
- package/src/cli/lib/__tests__/plugin-details.test.ts +158 -0
- package/src/cli/lib/__tests__/plugin-fingerprint.test.ts +245 -0
- package/src/cli/lib/__tests__/upgrade-plugin.test.ts +307 -0
- package/src/cli/lib/inspect-plugin.ts +252 -0
- package/src/cli/lib/install-from-github.ts +214 -21
- package/src/cli/lib/list-installed-plugins.ts +17 -6
- package/src/cli/lib/plugin-artifact.ts +103 -0
- package/src/cli/lib/plugin-details.ts +18 -1
- package/src/cli/lib/plugin-fingerprint.ts +197 -0
- package/src/cli/lib/upgrade-plugin.ts +225 -0
- package/src/config/bundled-skills/subagent/SKILL.md +2 -0
- package/src/config/bundled-skills/subagent/TOOLS.json +8 -2
- package/src/config/call-site-defaults.ts +13 -2
- package/src/config/feature-flag-registry.json +8 -16
- package/src/config/loader.ts +52 -59
- package/src/config/schema.ts +0 -2
- package/src/config/schemas/__tests__/memory-v2.test.ts +1 -0
- package/src/config/schemas/__tests__/memory-v3.test.ts +10 -0
- package/src/config/schemas/llm.ts +10 -0
- package/src/config/schemas/memory-v2.ts +13 -0
- package/src/config/schemas/memory-v3.ts +92 -0
- package/src/config/seed-inference-profiles.ts +4 -8
- package/src/context/post-turn-tool-result-truncation.ts +32 -18
- package/src/context/tool-result-spool.ts +104 -0
- package/src/credential-execution/feature-gates.ts +0 -1
- package/src/daemon/conversation-agent-loop-handlers.ts +41 -16
- package/src/daemon/conversation-error.ts +6 -15
- package/src/daemon/conversation.ts +9 -0
- package/src/daemon/disk-pressure-policy.ts +0 -1
- package/src/daemon/lifecycle.ts +1 -20
- package/src/daemon/message-types/conversations.ts +2 -15
- package/src/daemon/trust-context.ts +1 -1
- package/src/events/tool-audit-listener.ts +40 -9
- package/src/heartbeat/__tests__/heartbeat-service.test.ts +1 -1
- package/src/home/__tests__/home-content-refresh.test.ts +114 -0
- package/src/home/__tests__/suggested-prompts.test.ts +86 -5
- package/src/home/home-content-refresh.ts +43 -31
- package/src/home/home-greeting-cache.ts +8 -1
- package/src/home/home-greeting.ts +13 -9
- package/src/home/suggested-prompts.ts +77 -24
- package/src/ipc/routes/trust-rules.test.ts +66 -72
- package/src/media/image-credentials.ts +2 -2
- package/src/memory/__tests__/compaction-log-store-clickhouse.test.ts +432 -0
- package/src/memory/{compaction-log-writer-clickhouse.ts → compaction-log-store-clickhouse.ts} +264 -55
- package/src/memory/conversation-attention-store.ts +1 -0
- package/src/memory/conversation-bootstrap.ts +18 -9
- package/src/memory/conversation-crud.ts +12 -2
- package/src/memory/conversation-title-service.ts +53 -9
- package/src/memory/delivery-channels.ts +0 -69
- package/src/memory/graph/extraction-job.ts +0 -15
- package/src/memory/guardian-action-store.ts +1 -376
- package/src/memory/llm-usage-store.ts +5 -1
- package/src/memory/migrations/181-rename-thread-starters-checkpoints.ts +2 -2
- package/src/memory/v2/__tests__/consolidation-job.test.ts +183 -2
- package/src/memory/v2/__tests__/injection.test.ts +70 -0
- package/src/memory/v2/__tests__/static-context.test.ts +12 -0
- package/src/memory/v2/consolidation-job.ts +93 -9
- package/src/memory/v2/injection.ts +53 -0
- package/src/memory/v2/prompts/consolidation.ts +1 -0
- package/src/memory/v2/static-context.ts +13 -1
- package/src/memory/v2/sweep-job.ts +1 -1
- package/src/memory/v2/types.ts +5 -0
- package/src/plugin-api/types.ts +7 -0
- package/src/plugins/defaults/exploration-drift/hooks/post-tool-use.ts +300 -0
- package/src/plugins/defaults/exploration-drift/package.json +15 -0
- package/src/plugins/defaults/index.ts +25 -0
- package/src/plugins/defaults/memory-retrieval/injectors.ts +132 -4
- package/src/plugins/defaults/memory-v3-shadow/__tests__/card.test.ts +92 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +2 -1
- package/src/plugins/defaults/memory-v3-shadow/__tests__/fresh-set.test.ts +52 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +1 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +2 -1
- package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +136 -5
- package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +17 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +6 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +5 -1
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +68 -4
- package/src/plugins/defaults/memory-v3-shadow/card.ts +49 -5
- package/src/plugins/defaults/memory-v3-shadow/fresh-set.ts +59 -0
- package/src/plugins/defaults/memory-v3-shadow/injector.ts +4 -2
- package/src/plugins/defaults/memory-v3-shadow/learned-edges.test.ts +169 -0
- package/src/plugins/defaults/memory-v3-shadow/learned-edges.ts +178 -0
- package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +115 -26
- package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +13 -9
- package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +144 -22
- package/src/plugins/defaults/memory-v3-shadow/types.ts +24 -6
- package/src/plugins/defaults/title-generate/hooks/stop.ts +13 -0
- package/src/plugins/defaults/title-generate/hooks/user-prompt-submit.ts +16 -0
- package/src/prompts/cache-boundary.ts +17 -0
- package/src/prompts/sections.ts +50 -17
- package/src/prompts/system-prompt.ts +12 -4
- package/src/prompts/templates/system-sections.ts +22 -0
- package/src/providers/__tests__/unparseable-tool-args.test.ts +53 -0
- package/src/providers/anthropic/client.ts +74 -28
- package/src/providers/gemini/client.ts +5 -1
- package/src/providers/minimax/client.ts +9 -0
- package/src/providers/model-catalog.ts +28 -0
- package/src/providers/model-intents.ts +3 -3
- package/src/providers/openai/chat-completions-provider.ts +4 -2
- package/src/providers/openai/responses-provider.ts +7 -2
- package/src/providers/retry.ts +8 -0
- package/src/providers/types.ts +11 -0
- package/src/providers/unparseable-tool-args.ts +56 -0
- package/src/runtime/AGENTS.md +6 -0
- package/src/runtime/__tests__/agent-wake.test.ts +2 -2
- package/src/runtime/agent-wake.ts +5 -5
- package/src/runtime/background-job-runner.ts +2 -2
- package/src/runtime/migrations/__tests__/vbundle-legacy-user-md.test.ts +150 -3
- package/src/runtime/migrations/vbundle-import-analyzer.ts +29 -6
- package/src/runtime/migrations/vbundle-import-policy.ts +23 -0
- package/src/runtime/migrations/vbundle-importer.ts +9 -4
- package/src/runtime/migrations/vbundle-streaming-importer.ts +8 -3
- package/src/runtime/pre-first-message-gate.ts +1 -1
- package/src/runtime/routes/__tests__/conversation-compaction-routes.test.ts +241 -0
- package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +132 -0
- package/src/runtime/routes/__tests__/gateway-log-routes.test.ts +97 -185
- package/src/runtime/routes/__tests__/home-feed-routes.test.ts +17 -0
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +348 -0
- package/src/runtime/routes/__tests__/task-routes.test.ts +3 -3
- package/src/runtime/routes/btw-routes.ts +0 -14
- package/src/runtime/routes/conversation-compaction-routes.ts +86 -19
- package/src/runtime/routes/conversation-list-routes.ts +77 -5
- package/src/runtime/routes/conversation-management-routes.ts +54 -0
- package/src/runtime/routes/conversation-query-routes.ts +79 -4
- package/src/runtime/routes/gateway-log-routes.ts +14 -64
- package/src/runtime/routes/home-feed-routes.ts +10 -0
- package/src/runtime/routes/identity-intro-cache.ts +1 -1
- package/src/runtime/routes/identity-routes.ts +76 -20
- package/src/runtime/routes/inbound-message-handler.ts +0 -36
- package/src/runtime/routes/log-export-routes.ts +143 -96
- package/src/runtime/routes/plugins-routes.ts +380 -0
- package/src/runtime/routes/redact-staged-export.ts +259 -0
- package/src/runtime/routes/schedule-routes.ts +19 -2
- package/src/runtime/routes/trust-rules-routes.ts +14 -67
- package/src/schedule/recurrence-engine.ts +34 -0
- package/src/schedule/scheduler.ts +1 -0
- package/src/security/redact-json.ts +61 -0
- package/src/skills/validate-input.ts +41 -1
- package/src/subagent/types.ts +26 -1
- package/src/telemetry/types.ts +15 -1
- package/src/telemetry/usage-telemetry-reporter.test.ts +6 -1
- package/src/telemetry/usage-telemetry-reporter.ts +1 -0
- package/src/tools/apps/executors.ts +1 -1
- package/src/tools/skills/skill-tool-factory.ts +19 -8
- package/src/tools/tool-approval-handler.ts +31 -0
- package/src/usage/types.ts +8 -1
- package/src/util/platform.ts +16 -0
- package/src/watcher/engine.ts +1 -0
- package/src/workspace/adaptive-thinking-repair.ts +113 -0
- package/src/workspace/migrations/097-enable-adaptive-thinking-managed-profiles.ts +70 -67
- package/src/workspace/migrations/098-remove-stale-updates-bulletin-file.ts +31 -0
- package/src/workspace/migrations/099-disable-cache-one-shot-callsites.ts +81 -0
- package/src/workspace/migrations/100-upgrade-quality-profile-to-fable-5.ts +86 -0
- package/src/workspace/migrations/101-upgrade-balanced-economy-to-minimax-m3.ts +70 -0
- package/src/workspace/migrations/registry.ts +8 -0
- package/src/__tests__/config-loader-quarantine-bulletin.test.ts +0 -202
- package/src/__tests__/conversation-starters-cadence.test.ts +0 -161
- package/src/__tests__/guardian-action-followup-executor.test.ts +0 -322
- package/src/__tests__/guardian-action-followup-store.test.ts +0 -373
- package/src/__tests__/guardian-action-late-reply.test.ts +0 -1083
- package/src/__tests__/update-bulletin-job.test.ts +0 -292
- package/src/config/schemas/updates.ts +0 -14
- package/src/memory/__tests__/compaction-log-writer-clickhouse.test.ts +0 -227
- package/src/memory/conversation-starters-cadence.ts +0 -78
- package/src/prompts/update-bulletin-job.ts +0 -180
- package/src/runtime/guardian-action-followup-executor.ts +0 -306
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Default `post-tool-use` hook: when a turn's exploration tool calls (bash,
|
|
3
|
+
* file_read, file_list) show drift — a long unbroken run with no text sent to
|
|
4
|
+
* the user, or the model re-issuing the exact same call — surface a notice
|
|
5
|
+
* via `additionalContext` that coaches it to (a) give the user a brief
|
|
6
|
+
* progress summary and (b) delegate the rest of the investigation to an
|
|
7
|
+
* `investigator` subagent instead of continuing inline.
|
|
8
|
+
*
|
|
9
|
+
* Motivation: a root-cause request investigated inline ran 167 sequential
|
|
10
|
+
* bash calls in a single turn with no user-facing text, overflowed the
|
|
11
|
+
* conversation context before any findings were written up, and forced the
|
|
12
|
+
* user into repeated "Continue?" turns that re-explored the same files (11 of
|
|
13
|
+
* the 15 files read in the follow-up turn were re-reads). Delegation keeps
|
|
14
|
+
* the digging in a disposable subagent context. The notices are advisory —
|
|
15
|
+
* the model decides whether the current run is genuinely an investigation
|
|
16
|
+
* worth delegating.
|
|
17
|
+
*
|
|
18
|
+
* Two triggers, sharing one trailing-run computation:
|
|
19
|
+
*
|
|
20
|
+
* 1. **Long dig** (all models): an unbroken run of
|
|
21
|
+
* {@link EXPLORATION_NUDGE_THRESHOLD} exploration calls with no user-facing
|
|
22
|
+
* text. Repeat nudges are spaced one full threshold apart.
|
|
23
|
+
* 2. **Loop** (loop-prone models only, currently Kimi K2.6 and MiniMax M3):
|
|
24
|
+
* the current call
|
|
25
|
+
* is byte-identical (same tool, same input) to at least
|
|
26
|
+
* {@link EXPLORATION_LOOP_REPEAT_THRESHOLD}-1 prior calls within the
|
|
27
|
+
* trailing run. Re-issuing an identical read-only call inside an unbroken
|
|
28
|
+
* read-only run yields no new information — it is the earliest reliable
|
|
29
|
+
* sign the model is stuck, so this fires as soon as the repetition
|
|
30
|
+
* appears (potentially at call 3 of a run) rather than waiting for the
|
|
31
|
+
* long-dig threshold, and re-fires on every further duplicate while the
|
|
32
|
+
* model keeps looping. Gated by model so the aggressive trigger covers
|
|
33
|
+
* only models prone to this looping; the one legitimate identical-call pattern
|
|
34
|
+
* (polling an external process's output) is rare inside an unbroken
|
|
35
|
+
* read-only run and the nudge is advisory anyway.
|
|
36
|
+
*
|
|
37
|
+
* The trailing run is derived from conversation history on every call
|
|
38
|
+
* (mirroring the tool-error plugin) so the signal survives mid-run compaction
|
|
39
|
+
* rewriting the array. The run is bounded by:
|
|
40
|
+
* - a real user message (turn boundary),
|
|
41
|
+
* - any non-empty assistant text block (the model spoke to the user),
|
|
42
|
+
* - any non-exploration tool result (the model did something besides read).
|
|
43
|
+
*
|
|
44
|
+
* Nudges dedupe via a per-conversation high-water mark of the streak length
|
|
45
|
+
* at the last nudge: long-dig nudges require another full threshold of
|
|
46
|
+
* growth, loop nudges require any growth (each additional duplicate call
|
|
47
|
+
* re-nudges). The mark also dedupes parallel tool results of one batch (they
|
|
48
|
+
* all observe identical history and compute the same streak). Entries are
|
|
49
|
+
* dropped when the streak restarts.
|
|
50
|
+
*
|
|
51
|
+
* Subagent conversations are exempt: an investigator is *supposed* to dig at
|
|
52
|
+
* length, and subagents cannot nest (`SUBAGENT_LIMITS.maxDepth`), so the
|
|
53
|
+
* delegation advice would be wrong there. The check is a lazy import on the
|
|
54
|
+
* rare nudge path so the subagent manager's module graph stays out of the
|
|
55
|
+
* per-tool-result hot path.
|
|
56
|
+
*/
|
|
57
|
+
|
|
58
|
+
import type { PluginHookFn, PostToolUseContext } from "@vellumai/plugin-api";
|
|
59
|
+
|
|
60
|
+
import type { Message } from "../../../../providers/types.js";
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Canonical long-dig notice. Module-level constant so tests and wrapping
|
|
64
|
+
* plugins can match it without duplicating the string. Shown to the model as
|
|
65
|
+
* provider-only context, never to the user.
|
|
66
|
+
*/
|
|
67
|
+
export const EXPLORATION_DRIFT_NUDGE_TEXT =
|
|
68
|
+
"<system_notice>You have made a long unbroken run of exploration tool calls (shell/file reads) without sending the user any text. Do two things now: (1) send the user a brief summary of what you have found so far — do not keep working silently; (2) if you are tracing a root cause or exploring code/logs at length, stop exploring inline and delegate the remainder to a subagent: call subagent_spawn with role 'investigator' and a precise objective. It will investigate in its own context window and return a compact root-cause report. Continuing inline floods this conversation's context and risks losing your findings before you can report them.</system_notice>";
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Canonical loop notice, parameterized on the repeated call. Firmer than the
|
|
72
|
+
* long-dig text — by the time an identical read-only call repeats, the model
|
|
73
|
+
* is demonstrably not gaining information.
|
|
74
|
+
*/
|
|
75
|
+
export function explorationLoopNudgeText(
|
|
76
|
+
toolName: string,
|
|
77
|
+
repeatCount: number,
|
|
78
|
+
): string {
|
|
79
|
+
return `<system_notice>You have issued this exact ${toolName} call ${repeatCount} times in the current run of exploration tool calls. Repeating an identical read-only call yields no new information — you are likely stuck. Do two things now: (1) send the user a brief summary of what you have found so far and what you are still missing — do not keep working silently; (2) stop exploring inline and delegate the remaining investigation to a subagent: call subagent_spawn with role 'investigator' and a precise objective that includes what you have already checked and ruled out. It will investigate in its own context window and return a compact root-cause report. Do not re-issue this call again.</system_notice>`;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Exploration streak length that triggers the first long-dig nudge; repeat
|
|
84
|
+
* long-dig nudges fire each time the streak grows by another full threshold.
|
|
85
|
+
*/
|
|
86
|
+
export const EXPLORATION_NUDGE_THRESHOLD = 25;
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Number of byte-identical exploration calls (tool name + input) within one
|
|
90
|
+
* trailing run that triggers the loop nudge on loop-prone models. The count
|
|
91
|
+
* includes the current call, so 3 means "the current call is the third
|
|
92
|
+
* identical issue of this command".
|
|
93
|
+
*/
|
|
94
|
+
export const EXPLORATION_LOOP_REPEAT_THRESHOLD = 3;
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Models that get the early loop trigger: Kimi K2.6 and MiniMax M3, matched
|
|
98
|
+
* across provider naming conventions (Fireworks spells the dot as `p`, e.g.
|
|
99
|
+
* `accounts/fireworks/models/kimi-k2p6`; OpenRouter reports
|
|
100
|
+
* `moonshotai/kimi-k2.6` and `minimax/minimax-m3`). Extend the pattern as
|
|
101
|
+
* other models exhibit the same re-exploration looping.
|
|
102
|
+
*/
|
|
103
|
+
const LOOP_PRONE_MODEL_PATTERN = /kimi-k2[p.]6|minimax-m3/i;
|
|
104
|
+
|
|
105
|
+
/** Read-only exploration tools whose unbroken runs indicate inline drift. */
|
|
106
|
+
const EXPLORATION_TOOL_NAMES: ReadonlySet<string> = new Set([
|
|
107
|
+
"bash",
|
|
108
|
+
"file_read",
|
|
109
|
+
"file_list",
|
|
110
|
+
]);
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Streak length at the last nudge (either kind), per conversation. A
|
|
114
|
+
* high-water mark rather than a flag so long-dig nudges stay one full
|
|
115
|
+
* threshold apart, loop nudges fire only when the streak has grown since the
|
|
116
|
+
* last nudge, and parallel results of one batch (same observed history, same
|
|
117
|
+
* computed streak) dedupe to a single notice. Entries are dropped when the
|
|
118
|
+
* streak restarts.
|
|
119
|
+
*/
|
|
120
|
+
const lastNudgedStreakByConversation = new Map<string, number>();
|
|
121
|
+
|
|
122
|
+
/** Test-only: clear the per-conversation nudge high-water marks. */
|
|
123
|
+
export function resetExplorationDriftStateForTests(): void {
|
|
124
|
+
lastNudgedStreakByConversation.clear();
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/** A `tool_use` block's invocation: tool name plus its raw input. */
|
|
128
|
+
interface ToolInvocation {
|
|
129
|
+
readonly name: string;
|
|
130
|
+
readonly input: unknown;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Map every `tool_use` block id in history to its invocation. */
|
|
134
|
+
function toolUsesById(
|
|
135
|
+
messages: ReadonlyArray<Message>,
|
|
136
|
+
): Map<string, ToolInvocation> {
|
|
137
|
+
const uses = new Map<string, ToolInvocation>();
|
|
138
|
+
for (const message of messages) {
|
|
139
|
+
if (message.role !== "assistant") continue;
|
|
140
|
+
for (const block of message.content) {
|
|
141
|
+
if (block.type === "tool_use") {
|
|
142
|
+
uses.set(block.id, { name: block.name, input: block.input });
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
return uses;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Deterministic JSON encoding with object keys sorted recursively, so two
|
|
151
|
+
* semantically identical tool inputs hash to the same signature regardless of
|
|
152
|
+
* key order.
|
|
153
|
+
*/
|
|
154
|
+
function stableStringify(value: unknown): string {
|
|
155
|
+
if (value === null || typeof value !== "object") {
|
|
156
|
+
return JSON.stringify(value) ?? "undefined";
|
|
157
|
+
}
|
|
158
|
+
if (Array.isArray(value)) {
|
|
159
|
+
return `[${value.map(stableStringify).join(",")}]`;
|
|
160
|
+
}
|
|
161
|
+
const record = value as Record<string, unknown>;
|
|
162
|
+
const entries = Object.keys(record)
|
|
163
|
+
.sort()
|
|
164
|
+
.map((key) => `${JSON.stringify(key)}:${stableStringify(record[key])}`);
|
|
165
|
+
return `{${entries.join(",")}}`;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* The trailing unbroken run of exploration tool results in history: its
|
|
170
|
+
* length and the `tool_use` ids of the calls in it. Walks backwards from the
|
|
171
|
+
* most recent message and stops at a real user message, a non-empty assistant
|
|
172
|
+
* text block, or a non-exploration tool result. Text blocks inside
|
|
173
|
+
* tool-result user rows (e.g. coaching notices appended by other hooks) do
|
|
174
|
+
* not break the run — they are system notices, not the model speaking to the
|
|
175
|
+
* user.
|
|
176
|
+
*/
|
|
177
|
+
function trailingExplorationRun(
|
|
178
|
+
messages: ReadonlyArray<Message>,
|
|
179
|
+
usesById: ReadonlyMap<string, ToolInvocation>,
|
|
180
|
+
): { streak: number; toolUseIds: string[] } {
|
|
181
|
+
const toolUseIds: string[] = [];
|
|
182
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
183
|
+
const message = messages[i];
|
|
184
|
+
if (message.role === "assistant") {
|
|
185
|
+
const spokeToUser = message.content.some(
|
|
186
|
+
(block) => block.type === "text" && block.text.trim().length > 0,
|
|
187
|
+
);
|
|
188
|
+
if (spokeToUser) break;
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
if (message.role !== "user") continue;
|
|
192
|
+
const hasToolResult = message.content.some(
|
|
193
|
+
(block) => block.type === "tool_result",
|
|
194
|
+
);
|
|
195
|
+
if (!hasToolResult) break;
|
|
196
|
+
for (let j = message.content.length - 1; j >= 0; j--) {
|
|
197
|
+
const block = message.content[j];
|
|
198
|
+
if (block.type !== "tool_result") continue;
|
|
199
|
+
const use = usesById.get(block.tool_use_id);
|
|
200
|
+
if (use === undefined || !EXPLORATION_TOOL_NAMES.has(use.name)) {
|
|
201
|
+
return { streak: toolUseIds.length, toolUseIds };
|
|
202
|
+
}
|
|
203
|
+
toolUseIds.push(block.tool_use_id);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
return { streak: toolUseIds.length, toolUseIds };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* How many times the current call (tool name + input) has been issued within
|
|
211
|
+
* the trailing run, including the current call itself. Signatures are only
|
|
212
|
+
* computed for same-named calls, and only on the loop-prone-model path, to
|
|
213
|
+
* keep the per-tool-result cost bounded.
|
|
214
|
+
*/
|
|
215
|
+
function currentCallRepeatCount(
|
|
216
|
+
current: ToolInvocation,
|
|
217
|
+
runToolUseIds: ReadonlyArray<string>,
|
|
218
|
+
usesById: ReadonlyMap<string, ToolInvocation>,
|
|
219
|
+
): number {
|
|
220
|
+
const currentSignature = stableStringify(current.input);
|
|
221
|
+
let count = 1;
|
|
222
|
+
for (const id of runToolUseIds) {
|
|
223
|
+
const use = usesById.get(id);
|
|
224
|
+
if (
|
|
225
|
+
use !== undefined &&
|
|
226
|
+
use.name === current.name &&
|
|
227
|
+
stableStringify(use.input) === currentSignature
|
|
228
|
+
) {
|
|
229
|
+
count++;
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
return count;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const postToolUse: PluginHookFn<PostToolUseContext> = async (ctx) => {
|
|
236
|
+
const usesById = toolUsesById(ctx.messages);
|
|
237
|
+
const currentUse = usesById.get(ctx.toolResponse.tool_use_id);
|
|
238
|
+
if (currentUse === undefined || !EXPLORATION_TOOL_NAMES.has(currentUse.name))
|
|
239
|
+
return;
|
|
240
|
+
|
|
241
|
+
// The current result is not in history yet — count it explicitly.
|
|
242
|
+
const run = trailingExplorationRun(ctx.messages, usesById);
|
|
243
|
+
const streak = run.streak + 1;
|
|
244
|
+
|
|
245
|
+
let lastNudged = lastNudgedStreakByConversation.get(ctx.conversationId) ?? 0;
|
|
246
|
+
if (streak < lastNudged) {
|
|
247
|
+
// The streak restarted (new turn, intervening text, or compaction) since
|
|
248
|
+
// the last nudge — drop the stale high-water mark.
|
|
249
|
+
lastNudgedStreakByConversation.delete(ctx.conversationId);
|
|
250
|
+
lastNudged = 0;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
const longDigNudge = streak - lastNudged >= EXPLORATION_NUDGE_THRESHOLD;
|
|
254
|
+
|
|
255
|
+
// Loop detection: only on loop-prone models, and only when the streak has
|
|
256
|
+
// grown since the last nudge (dedupes parallel batches; re-fires on each
|
|
257
|
+
// further duplicate). Keyed to the *current* call's signature so the nudge
|
|
258
|
+
// stops as soon as the model moves on to fresh calls.
|
|
259
|
+
let loopRepeatCount = 0;
|
|
260
|
+
if (
|
|
261
|
+
!longDigNudge &&
|
|
262
|
+
streak > lastNudged &&
|
|
263
|
+
LOOP_PRONE_MODEL_PATTERN.test(ctx.model)
|
|
264
|
+
) {
|
|
265
|
+
loopRepeatCount = currentCallRepeatCount(
|
|
266
|
+
currentUse,
|
|
267
|
+
run.toolUseIds,
|
|
268
|
+
usesById,
|
|
269
|
+
);
|
|
270
|
+
}
|
|
271
|
+
const loopNudge = loopRepeatCount >= EXPLORATION_LOOP_REPEAT_THRESHOLD;
|
|
272
|
+
|
|
273
|
+
if (!longDigNudge && !loopNudge) return;
|
|
274
|
+
|
|
275
|
+
// Subagent conversations are exempt — see module doc.
|
|
276
|
+
const { getSubagentManager } = await import("../../../../subagent/index.js");
|
|
277
|
+
if (getSubagentManager().getParentInfo(ctx.conversationId) !== undefined) {
|
|
278
|
+
return;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
lastNudgedStreakByConversation.set(ctx.conversationId, streak);
|
|
282
|
+
const nudgeText = loopNudge
|
|
283
|
+
? explorationLoopNudgeText(currentUse.name, loopRepeatCount)
|
|
284
|
+
: EXPLORATION_DRIFT_NUDGE_TEXT;
|
|
285
|
+
ctx.logger.info(
|
|
286
|
+
{
|
|
287
|
+
plugin: "exploration-drift",
|
|
288
|
+
streak,
|
|
289
|
+
toolName: currentUse.name,
|
|
290
|
+
trigger: loopNudge ? "loop" : "long-dig",
|
|
291
|
+
...(loopNudge ? { repeatCount: loopRepeatCount } : {}),
|
|
292
|
+
},
|
|
293
|
+
"Exploration drift detected — nudging summary + investigator delegation",
|
|
294
|
+
);
|
|
295
|
+
ctx.additionalContext = ctx.additionalContext
|
|
296
|
+
? `${ctx.additionalContext}\n${nudgeText}`
|
|
297
|
+
: nudgeText;
|
|
298
|
+
};
|
|
299
|
+
|
|
300
|
+
export default postToolUse;
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "default-exploration-drift",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "First-party default plugin contributing a post-tool-use hook that nudges the model to summarize and delegate to an investigator subagent when a turn accumulates a long run of exploration tool calls, or (on loop-prone models) re-issues an identical exploration call.",
|
|
5
|
+
"private": true,
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"type": "module",
|
|
8
|
+
"main": "./register.ts",
|
|
9
|
+
"engines": {
|
|
10
|
+
"node": ">=20.12.0"
|
|
11
|
+
},
|
|
12
|
+
"peerDependencies": {
|
|
13
|
+
"@vellumai/plugin-api": "^0.8.0"
|
|
14
|
+
}
|
|
15
|
+
}
|
|
@@ -31,6 +31,10 @@ import emptyResponsePostModelCall from "./empty-response/hooks/post-model-call.j
|
|
|
31
31
|
import emptyResponseStop from "./empty-response/hooks/stop.js";
|
|
32
32
|
import { resetEmptyResponseNudgeStoreForTests } from "./empty-response/nudge-state-store.js";
|
|
33
33
|
import emptyResponsePkg from "./empty-response/package.json" with { type: "json" };
|
|
34
|
+
import explorationDriftPostToolUse, {
|
|
35
|
+
resetExplorationDriftStateForTests,
|
|
36
|
+
} from "./exploration-drift/hooks/post-tool-use.js";
|
|
37
|
+
import explorationDriftPkg from "./exploration-drift/package.json" with { type: "json" };
|
|
34
38
|
import historyRepairPostModelCall from "./history-repair/hooks/post-model-call.js";
|
|
35
39
|
import historyRepairStop from "./history-repair/hooks/stop.js";
|
|
36
40
|
import historyRepairUserPromptSubmit from "./history-repair/hooks/user-prompt-submit.js";
|
|
@@ -194,6 +198,25 @@ export const defaultToolErrorPlugin: Plugin = {
|
|
|
194
198
|
},
|
|
195
199
|
};
|
|
196
200
|
|
|
201
|
+
/**
|
|
202
|
+
* `exploration-drift` — a `post-tool-use` hook that detects exploration
|
|
203
|
+
* drift — a long unbroken run of exploration tool calls (bash, file_read,
|
|
204
|
+
* file_list) with no user-facing text, or (on loop-prone models such as Kimi
|
|
205
|
+
* K2.6 and MiniMax M3) the model re-issuing a byte-identical exploration call — and nudges
|
|
206
|
+
* the model via `additionalContext` to summarize progress for the user and
|
|
207
|
+
* delegate the remaining investigation to an `investigator` subagent rather
|
|
208
|
+
* than continuing inline.
|
|
209
|
+
*/
|
|
210
|
+
export const defaultExplorationDriftPlugin: Plugin = {
|
|
211
|
+
manifest: {
|
|
212
|
+
name: explorationDriftPkg.name,
|
|
213
|
+
version: explorationDriftPkg.version,
|
|
214
|
+
},
|
|
215
|
+
hooks: {
|
|
216
|
+
"post-tool-use": explorationDriftPostToolUse,
|
|
217
|
+
},
|
|
218
|
+
};
|
|
219
|
+
|
|
197
220
|
/**
|
|
198
221
|
* `tool-result-truncate` — a `post-tool-use` hook that tail-drops an oversized
|
|
199
222
|
* tool result down to a character budget derived from the model's context
|
|
@@ -221,6 +244,7 @@ function getAllDefaultPlugins(): readonly Plugin[] {
|
|
|
221
244
|
defaultToolResultTruncatePlugin,
|
|
222
245
|
defaultEmptyResponsePlugin,
|
|
223
246
|
defaultToolErrorPlugin,
|
|
247
|
+
defaultExplorationDriftPlugin,
|
|
224
248
|
defaultHistoryRepairPlugin,
|
|
225
249
|
defaultImageRecoveryPlugin,
|
|
226
250
|
defaultCompactionPlugin,
|
|
@@ -268,5 +292,6 @@ export function resetPluginRegistryAndRegisterDefaults(): void {
|
|
|
268
292
|
resetEmptyResponseNudgeStoreForTests();
|
|
269
293
|
resetRepairStateStoreForTests();
|
|
270
294
|
resetImageRecoveryStoreForTests();
|
|
295
|
+
resetExplorationDriftStateForTests();
|
|
271
296
|
registerDefaultPlugins();
|
|
272
297
|
}
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* | `disk-pressure-warning` | 5 | prepend-user-tail |
|
|
16
16
|
* | `workspace-context` | 10 | prepend-user-tail |
|
|
17
17
|
* | `unified-turn-context` | 20 | prepend-user-tail |
|
|
18
|
+
* | `config-quarantine-notice` | 25 | prepend-user-tail |
|
|
18
19
|
* | `pkb-context` | 30 | after-memory-prefix |
|
|
19
20
|
* | `pkb-reminder` | 35 | after-memory-prefix |
|
|
20
21
|
* | `memory-v2-static` | 38 | after-memory-prefix |
|
|
@@ -43,6 +44,7 @@
|
|
|
43
44
|
* through the registry.
|
|
44
45
|
*/
|
|
45
46
|
|
|
47
|
+
import { existsSync, readFileSync, rmSync } from "node:fs";
|
|
46
48
|
import { resolve } from "node:path";
|
|
47
49
|
|
|
48
50
|
import { getConfig } from "../../../config/loader.js";
|
|
@@ -65,7 +67,10 @@ import { getPkbRoot, PKB_WORKSPACE_SCOPE } from "../../../memory/pkb/types.js";
|
|
|
65
67
|
import { readMemoryV2StaticContent } from "../../../memory/v2/static-context.js";
|
|
66
68
|
import type { Message } from "../../../providers/types.js";
|
|
67
69
|
import { getLogger } from "../../../util/logger.js";
|
|
68
|
-
import {
|
|
70
|
+
import {
|
|
71
|
+
getConfigQuarantineNoticePath,
|
|
72
|
+
getSandboxWorkingDir,
|
|
73
|
+
} from "../../../util/platform.js";
|
|
69
74
|
import {
|
|
70
75
|
type InjectionBlock,
|
|
71
76
|
type Injector,
|
|
@@ -99,6 +104,7 @@ export const DEFAULT_INJECTOR_ORDER = {
|
|
|
99
104
|
workspaceContext: 10,
|
|
100
105
|
backgroundTurn: 15,
|
|
101
106
|
unifiedTurnContext: 20,
|
|
107
|
+
configQuarantineNotice: 25,
|
|
102
108
|
pkbContext: 30,
|
|
103
109
|
pkbReminder: 35,
|
|
104
110
|
memoryV2Static: 38,
|
|
@@ -277,6 +283,114 @@ const unifiedTurnContextInjector: Injector = {
|
|
|
277
283
|
},
|
|
278
284
|
};
|
|
279
285
|
|
|
286
|
+
/**
|
|
287
|
+
* Maximum age of a config-quarantine notice before it is considered stale.
|
|
288
|
+
* After this window the sentinel is deleted and nothing is injected — the
|
|
289
|
+
* event is no longer actionable context for the agent.
|
|
290
|
+
*/
|
|
291
|
+
const CONFIG_QUARANTINE_NOTICE_MAX_AGE_MS = 7 * 24 * 60 * 60 * 1000;
|
|
292
|
+
|
|
293
|
+
/** Shape of the config-quarantine notice sentinel written by the config loader. */
|
|
294
|
+
interface ConfigQuarantineNotice {
|
|
295
|
+
quarantinedAt: string;
|
|
296
|
+
quarantinePath: string;
|
|
297
|
+
originalPath: string;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/**
|
|
301
|
+
* Read and validate the config-quarantine notice sentinel. Returns the parsed
|
|
302
|
+
* notice when the file exists and carries the expected string fields, otherwise
|
|
303
|
+
* `null`. Best-effort: any read/parse error is swallowed and treated as absent.
|
|
304
|
+
*/
|
|
305
|
+
function readConfigQuarantineNotice(): ConfigQuarantineNotice | null {
|
|
306
|
+
const noticePath = getConfigQuarantineNoticePath();
|
|
307
|
+
if (!existsSync(noticePath)) return null;
|
|
308
|
+
try {
|
|
309
|
+
const parsed: unknown = JSON.parse(readFileSync(noticePath, "utf-8"));
|
|
310
|
+
if (parsed == null || typeof parsed !== "object") return null;
|
|
311
|
+
const { quarantinedAt, quarantinePath, originalPath } = parsed as Record<
|
|
312
|
+
string,
|
|
313
|
+
unknown
|
|
314
|
+
>;
|
|
315
|
+
if (
|
|
316
|
+
typeof quarantinedAt !== "string" ||
|
|
317
|
+
typeof quarantinePath !== "string" ||
|
|
318
|
+
typeof originalPath !== "string"
|
|
319
|
+
) {
|
|
320
|
+
return null;
|
|
321
|
+
}
|
|
322
|
+
return { quarantinedAt, quarantinePath, originalPath };
|
|
323
|
+
} catch {
|
|
324
|
+
return null;
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* `config-quarantine-notice` injector — order 25, prepend-user-tail.
|
|
330
|
+
*
|
|
331
|
+
* Surfaces a recent config-quarantine event to the agent. The config loader
|
|
332
|
+
* writes a JSON sentinel ({@link getConfigQuarantineNoticePath}) when it
|
|
333
|
+
* quarantines a corrupt `config.json` and falls back to defaults. This injector
|
|
334
|
+
* reads that sentinel and, when it is younger than
|
|
335
|
+
* {@link CONFIG_QUARANTINE_NOTICE_MAX_AGE_MS}, injects a short block telling the
|
|
336
|
+
* agent the user's settings were reset and where the original is preserved, so
|
|
337
|
+
* it can explain the change if the user asks about missing settings/API keys —
|
|
338
|
+
* or mention it proactively when relevant. A stale sentinel is deleted and
|
|
339
|
+
* nothing is injected.
|
|
340
|
+
*
|
|
341
|
+
* Active in both `full` and `minimal` mode — a settings reset is grounding the
|
|
342
|
+
* agent should not lose to injection downgrade. The block is non-persisted
|
|
343
|
+
* (re-evaluated every turn) so the notice naturally stops appearing once the
|
|
344
|
+
* sentinel ages out or is removed.
|
|
345
|
+
*
|
|
346
|
+
* Guardian-only: turns driven by non-guardian actors (trusted contacts,
|
|
347
|
+
* unknown channel senders) must not see workspace file paths or be told the
|
|
348
|
+
* guardian's settings were reset — the notice is only actionable in guardian
|
|
349
|
+
* conversations.
|
|
350
|
+
*/
|
|
351
|
+
const configQuarantineNoticeInjector: Injector = {
|
|
352
|
+
name: "config-quarantine-notice",
|
|
353
|
+
order: DEFAULT_INJECTOR_ORDER.configQuarantineNotice,
|
|
354
|
+
async produce(ctx: TurnContext): Promise<InjectionBlock | null> {
|
|
355
|
+
if (ctx.trust.trustClass !== "guardian") return null;
|
|
356
|
+
|
|
357
|
+
const notice = readConfigQuarantineNotice();
|
|
358
|
+
if (!notice) return null;
|
|
359
|
+
|
|
360
|
+
const quarantinedAtMs = Date.parse(notice.quarantinedAt);
|
|
361
|
+
const ageMs = Number.isNaN(quarantinedAtMs)
|
|
362
|
+
? Number.POSITIVE_INFINITY
|
|
363
|
+
: Date.now() - quarantinedAtMs;
|
|
364
|
+
if (ageMs > CONFIG_QUARANTINE_NOTICE_MAX_AGE_MS) {
|
|
365
|
+
try {
|
|
366
|
+
rmSync(getConfigQuarantineNoticePath(), { force: true });
|
|
367
|
+
} catch {
|
|
368
|
+
// Best-effort cleanup — a failed delete just means we re-check (and
|
|
369
|
+
// re-attempt deletion) next turn.
|
|
370
|
+
}
|
|
371
|
+
return null;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
const text =
|
|
375
|
+
`<config_reset_notice>\n` +
|
|
376
|
+
`The user's config.json was unreadable and was reset to defaults at ` +
|
|
377
|
+
`${notice.quarantinedAt}. The original file was preserved at ` +
|
|
378
|
+
`${notice.quarantinePath}. Any custom settings the user had (API keys, ` +
|
|
379
|
+
`model choices, voice preferences) are still in that file but are not ` +
|
|
380
|
+
`currently active.\n\n` +
|
|
381
|
+
`If the user asks why a setting, API key, or preference is missing or ` +
|
|
382
|
+
`changed, explain this reset and point them at the preserved file to ` +
|
|
383
|
+
`recover their settings. Otherwise mention it proactively only when it ` +
|
|
384
|
+
`is clearly relevant — do not interrupt unrelated work.\n` +
|
|
385
|
+
`</config_reset_notice>`;
|
|
386
|
+
return {
|
|
387
|
+
id: "config-quarantine-notice",
|
|
388
|
+
text,
|
|
389
|
+
placement: "prepend-user-tail",
|
|
390
|
+
};
|
|
391
|
+
},
|
|
392
|
+
};
|
|
393
|
+
|
|
280
394
|
/**
|
|
281
395
|
* `pkb-context` injector — order 30, after-memory-prefix.
|
|
282
396
|
*
|
|
@@ -389,9 +503,17 @@ function readGatedNowScratchpad(trust: TrustContext): string | null {
|
|
|
389
503
|
* otherwise `null`. {@link readMemoryV2StaticContent} self-gates on the v2
|
|
390
504
|
* flag + config, so the `memory-v2-static` injector owns its input rather than
|
|
391
505
|
* having it threaded in from the agent loop.
|
|
506
|
+
*
|
|
507
|
+
* `excludeBuffer` is forwarded for consolidation turns, whose contract is the
|
|
508
|
+
* buffer FILE itself — see {@link readMemoryV2StaticContent}.
|
|
392
509
|
*/
|
|
393
|
-
function readGatedMemoryV2Static(
|
|
394
|
-
|
|
510
|
+
function readGatedMemoryV2Static(
|
|
511
|
+
trust: TrustContext,
|
|
512
|
+
options: { excludeBuffer?: boolean } = {},
|
|
513
|
+
): string | null {
|
|
514
|
+
return isPersonalMemoryAllowed(trust)
|
|
515
|
+
? readMemoryV2StaticContent(options)
|
|
516
|
+
: null;
|
|
395
517
|
}
|
|
396
518
|
|
|
397
519
|
/**
|
|
@@ -629,7 +751,12 @@ const memoryV2StaticInjector: Injector = {
|
|
|
629
751
|
): Promise<InjectionBlock | null> {
|
|
630
752
|
const mode = ctx.mode ?? "full";
|
|
631
753
|
if (mode !== "full") return null;
|
|
632
|
-
|
|
754
|
+
// The consolidation agent reads and rewrites memory/buffer.md through
|
|
755
|
+
// file tools; injecting the buffer section here would duplicate the
|
|
756
|
+
// entire backlog into its context (and go stale as it edits the file).
|
|
757
|
+
const content = readGatedMemoryV2Static(ctx.trust, {
|
|
758
|
+
excludeBuffer: ctx.callSite === "memoryV2Consolidation",
|
|
759
|
+
});
|
|
633
760
|
if (!content) return null;
|
|
634
761
|
if (hasInjectedUserTextBlock(runMessages, MEMORY_V2_STATIC_BLOCK_MATCHERS))
|
|
635
762
|
return null;
|
|
@@ -921,6 +1048,7 @@ export const defaultInjectors: Injector[] = [
|
|
|
921
1048
|
workspaceContextInjector,
|
|
922
1049
|
backgroundTurnInjector,
|
|
923
1050
|
unifiedTurnContextInjector,
|
|
1051
|
+
configQuarantineNoticeInjector,
|
|
924
1052
|
pkbContextInjector,
|
|
925
1053
|
pkbReminderInjector,
|
|
926
1054
|
memoryV2StaticInjector,
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for `card.ts` — the compact card renderer. Focused on the annotation
|
|
3
|
+
* line: it must sit directly under the header (the always-rendered card
|
|
4
|
+
* surface) and leave the card untouched when absent.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { describe, expect, test } from "bun:test";
|
|
8
|
+
|
|
9
|
+
import { renderCard } from "../card.js";
|
|
10
|
+
|
|
11
|
+
const PAGE = `---
|
|
12
|
+
title: Page A
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
Lead paragraph for page a.
|
|
16
|
+
|
|
17
|
+
## Alpha
|
|
18
|
+
|
|
19
|
+
Body text.
|
|
20
|
+
`;
|
|
21
|
+
|
|
22
|
+
describe("renderCard — annotation line", () => {
|
|
23
|
+
test("renders the annotation directly under the header, before the head", () => {
|
|
24
|
+
const card = renderCard(
|
|
25
|
+
"page-a",
|
|
26
|
+
PAGE,
|
|
27
|
+
"[lane: fresh · updated 2026-06-10 14:23 UTC]",
|
|
28
|
+
);
|
|
29
|
+
expect(
|
|
30
|
+
card.startsWith(
|
|
31
|
+
"# memory/concepts/page-a.md\n[lane: fresh · updated 2026-06-10 14:23 UTC]\nLead paragraph for page a.",
|
|
32
|
+
),
|
|
33
|
+
).toBe(true);
|
|
34
|
+
expect(card).toContain("[sections: §Alpha]");
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
test("an absent or empty annotation leaves the card unchanged", () => {
|
|
38
|
+
const bare = renderCard("page-a", PAGE);
|
|
39
|
+
expect(renderCard("page-a", PAGE, "")).toBe(bare);
|
|
40
|
+
expect(bare).not.toContain("[lane:");
|
|
41
|
+
expect(
|
|
42
|
+
bare.startsWith(
|
|
43
|
+
"# memory/concepts/page-a.md\nLead paragraph for page a.",
|
|
44
|
+
),
|
|
45
|
+
).toBe(true);
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test("renders a `current:` frontmatter line first, before the lane annotation", () => {
|
|
49
|
+
const page = `---
|
|
50
|
+
title: Page A
|
|
51
|
+
current: "bridge check owed before thursday's dry-run (as of jun 10)"
|
|
52
|
+
---
|
|
53
|
+
|
|
54
|
+
Lead paragraph for page a.
|
|
55
|
+
`;
|
|
56
|
+
const card = renderCard("page-a", page, "[lane: fresh]");
|
|
57
|
+
expect(
|
|
58
|
+
card.startsWith(
|
|
59
|
+
"# memory/concepts/page-a.md\n[current: bridge check owed before thursday's dry-run (as of jun 10)]\n[lane: fresh]\nLead paragraph for page a.",
|
|
60
|
+
),
|
|
61
|
+
).toBe(true);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test("collapses whitespace and caps a runaway `current:` value", () => {
|
|
65
|
+
const long = `a line with breaks ${"x".repeat(400)}`;
|
|
66
|
+
const card = renderCard(
|
|
67
|
+
"page-a",
|
|
68
|
+
`---\ncurrent: "${long}"\n---\n\nLead.\n`,
|
|
69
|
+
);
|
|
70
|
+
const line = card.split("\n")[1]!;
|
|
71
|
+
expect(line.startsWith("[current: a line with breaks x")).toBe(true);
|
|
72
|
+
expect(line.endsWith("…]")).toBe(true);
|
|
73
|
+
expect(line.length).toBeLessThan(300);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test("the `status:` draft marker does NOT render as a card line", () => {
|
|
77
|
+
const card = renderCard("page-a", `---\nstatus: cc-draft\n---\n\nLead.\n`);
|
|
78
|
+
expect(card).not.toContain("cc-draft");
|
|
79
|
+
expect(card).not.toContain("[current:");
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
test("annotates a page with no head section without a dangling blank line", () => {
|
|
83
|
+
const card = renderCard(
|
|
84
|
+
"page-b",
|
|
85
|
+
"## Only Section\n\nBody.",
|
|
86
|
+
"[lane: core]",
|
|
87
|
+
);
|
|
88
|
+
expect(card.startsWith("# memory/concepts/page-b.md\n[lane: core]")).toBe(
|
|
89
|
+
true,
|
|
90
|
+
);
|
|
91
|
+
});
|
|
92
|
+
});
|
|
@@ -255,6 +255,7 @@ async function scriptedObserveTurn(conversationId: string, turnIndex: number) {
|
|
|
255
255
|
edgeGraph: lanes.edgeGraph,
|
|
256
256
|
coreSlugs: lanes.coreSlugs,
|
|
257
257
|
hotSlugs: lanes.hotSlugs,
|
|
258
|
+
freshSlugs: [],
|
|
258
259
|
prefixCards: lanes.prefixCards,
|
|
259
260
|
},
|
|
260
261
|
);
|
|
@@ -487,7 +488,7 @@ function candidateSlugs(messages: Message[]): Slug[] {
|
|
|
487
488
|
);
|
|
488
489
|
if (finder) {
|
|
489
490
|
for (const line of finder[1].split("\n")) {
|
|
490
|
-
const m = /^\[(\d+)\] (\S+)(?: — |$)/.exec(line);
|
|
491
|
+
const m = /^\[(\d+)\] (?:\([^)]*\) )?(\S+)(?: — |$)/.exec(line);
|
|
491
492
|
if (m) entries.push({ id: Number(m[1]), slug: m[2]! });
|
|
492
493
|
}
|
|
493
494
|
}
|