@arnilo/prism 0.0.18 → 0.0.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/dist/agents.js +17 -2
- package/dist/context-budget.d.ts +5 -1
- package/dist/context-budget.js +55 -9
- package/dist/contracts.d.ts +16 -0
- package/dist/index.d.ts +7 -1
- package/dist/index.js +4 -1
- package/dist/input.d.ts +6 -0
- package/dist/input.js +42 -8
- package/dist/skill-disclosure.d.ts +35 -0
- package/dist/skill-disclosure.js +101 -0
- package/dist/skill-load.d.ts +25 -0
- package/dist/skill-load.js +112 -0
- package/dist/tool-result-fold.d.ts +40 -0
- package/dist/tool-result-fold.js +164 -0
- package/docs/0.1.0-readiness.md +10 -8
- package/docs/agent-session-runtime.md +2 -0
- package/docs/compaction-observational-memory.md +52 -8
- package/docs/context-and-skills.md +83 -7
- package/docs/index.md +5 -5
- package/docs/migration.md +35 -0
- package/docs/release-and-install.md +56 -14
- package/package.json +1 -1
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
import { estimateTextBytes } from "./context-budget.js";
|
|
2
|
+
import { HARD_MAX_SKILL_INSTRUCTION_BYTES } from "./skill-disclosure.js";
|
|
3
|
+
export const DEFAULT_LOAD_SKILL_TOOL_NAME = "load_skill";
|
|
4
|
+
export const SKILL_LOAD_ERROR_CODE = "skill_load_failed";
|
|
5
|
+
export const MAX_LOAD_SKILL_RESULT_BYTES = 512;
|
|
6
|
+
export class SkillLoadError extends Error {
|
|
7
|
+
code = SKILL_LOAD_ERROR_CODE;
|
|
8
|
+
constructor(message) {
|
|
9
|
+
super(message);
|
|
10
|
+
this.name = "SkillLoadError";
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
export function isSkillLoadError(error) {
|
|
14
|
+
return error instanceof Error && error.code === SKILL_LOAD_ERROR_CODE;
|
|
15
|
+
}
|
|
16
|
+
export function resolveSkillLoad(options) {
|
|
17
|
+
const name = options.name.trim();
|
|
18
|
+
if (!name)
|
|
19
|
+
throw new SkillLoadError("Skill name is required");
|
|
20
|
+
let skill;
|
|
21
|
+
try {
|
|
22
|
+
skill = options.registry.resolve(name);
|
|
23
|
+
}
|
|
24
|
+
catch {
|
|
25
|
+
throw new SkillLoadError(`Unknown skill: ${name}`);
|
|
26
|
+
}
|
|
27
|
+
if (options.activeSkillNames && !options.activeSkillNames.includes(skill.name)) {
|
|
28
|
+
throw new SkillLoadError(`Skill ${skill.name} is not active for this run`);
|
|
29
|
+
}
|
|
30
|
+
const toolNames = new Set((options.tools ?? []).map((tool) => tool.name));
|
|
31
|
+
const missingTool = skill.toolNames?.find((toolName) => !toolNames.has(toolName));
|
|
32
|
+
if (missingTool)
|
|
33
|
+
throw new SkillLoadError(`Skill ${skill.name} requires inactive tool: ${missingTool}`);
|
|
34
|
+
if (!skill.instructions?.trim())
|
|
35
|
+
throw new SkillLoadError(`Skill ${skill.name} has no instructions to load`);
|
|
36
|
+
const bytes = estimateTextBytes(skill.instructions);
|
|
37
|
+
if (bytes > HARD_MAX_SKILL_INSTRUCTION_BYTES) {
|
|
38
|
+
throw new SkillLoadError(`Skill ${skill.name} instructions exceed hard cap (${HARD_MAX_SKILL_INSTRUCTION_BYTES} bytes)`);
|
|
39
|
+
}
|
|
40
|
+
if (options.loaded?.has(skill.name))
|
|
41
|
+
throw new SkillLoadError(`Skill ${skill.name} is already loaded`);
|
|
42
|
+
return skill;
|
|
43
|
+
}
|
|
44
|
+
function skillLoadMetadata(context) {
|
|
45
|
+
const metadata = context.metadata;
|
|
46
|
+
if (!metadata || typeof metadata !== "object")
|
|
47
|
+
return undefined;
|
|
48
|
+
return metadata;
|
|
49
|
+
}
|
|
50
|
+
function capLoadSkillText(text) {
|
|
51
|
+
const bytes = estimateTextBytes(text);
|
|
52
|
+
if (bytes <= MAX_LOAD_SKILL_RESULT_BYTES)
|
|
53
|
+
return text;
|
|
54
|
+
const suffix = "…";
|
|
55
|
+
let end = MAX_LOAD_SKILL_RESULT_BYTES - new TextEncoder().encode(suffix).length;
|
|
56
|
+
const encoded = new TextEncoder().encode(text);
|
|
57
|
+
while (end > 0 && (encoded[end] & 0xc0) === 0x80)
|
|
58
|
+
end--;
|
|
59
|
+
return new TextDecoder().decode(encoded.slice(0, end)) + suffix;
|
|
60
|
+
}
|
|
61
|
+
export function createLoadSkillTool(options) {
|
|
62
|
+
const toolName = options.name ?? DEFAULT_LOAD_SKILL_TOOL_NAME;
|
|
63
|
+
return {
|
|
64
|
+
name: toolName,
|
|
65
|
+
description: "Load the full instructions for an active skill by exact name.",
|
|
66
|
+
parameters: {
|
|
67
|
+
type: "object",
|
|
68
|
+
properties: {
|
|
69
|
+
name: { type: "string", description: "Exact skill name from the catalog." },
|
|
70
|
+
},
|
|
71
|
+
required: ["name"],
|
|
72
|
+
},
|
|
73
|
+
execute(args, context) {
|
|
74
|
+
const fail = (reason, text) => {
|
|
75
|
+
const bounded = capLoadSkillText(text);
|
|
76
|
+
return {
|
|
77
|
+
toolCallId: context.toolCallId,
|
|
78
|
+
name: toolName,
|
|
79
|
+
value: { ok: false, reason, text: bounded },
|
|
80
|
+
content: [{ type: "text", text: bounded }],
|
|
81
|
+
};
|
|
82
|
+
};
|
|
83
|
+
const name = typeof args.name === "string" ? args.name : "";
|
|
84
|
+
const metadata = skillLoadMetadata(context);
|
|
85
|
+
const loaded = metadata?.loadedSkills ?? options.loaded;
|
|
86
|
+
if (!loaded)
|
|
87
|
+
return fail("no_loaded_set", "Skill load state is unavailable for this session.");
|
|
88
|
+
try {
|
|
89
|
+
const skill = resolveSkillLoad({
|
|
90
|
+
registry: options.registry,
|
|
91
|
+
name,
|
|
92
|
+
tools: metadata?.activeTools ?? options.tools,
|
|
93
|
+
loaded,
|
|
94
|
+
activeSkillNames: metadata?.activeSkillNames,
|
|
95
|
+
});
|
|
96
|
+
loaded.add(skill.name);
|
|
97
|
+
const text = capLoadSkillText(`Loaded skill ${skill.name} for this session.`);
|
|
98
|
+
return {
|
|
99
|
+
toolCallId: context.toolCallId,
|
|
100
|
+
name: toolName,
|
|
101
|
+
value: { ok: true, name: skill.name, text },
|
|
102
|
+
content: [{ type: "text", text }],
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
catch (error) {
|
|
106
|
+
const message = error instanceof Error ? error.message : "Skill load failed";
|
|
107
|
+
return fail("skill_load_failed", message);
|
|
108
|
+
}
|
|
109
|
+
},
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
//# sourceMappingURL=skill-load.js.map
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { Message, ToolResult } from "./contracts.js";
|
|
2
|
+
export declare const DEFAULT_TOOL_RESULT_FOLD_MIN_AGE_TURNS = 2;
|
|
3
|
+
export declare const DEFAULT_TOOL_RESULT_FOLD_MIN_BYTES = 4096;
|
|
4
|
+
export declare const DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES = 512;
|
|
5
|
+
export declare const HARD_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES = 4096;
|
|
6
|
+
export declare const TOOL_RESULT_FOLD_TURN_METADATA_KEY: "prismToolResultTurn";
|
|
7
|
+
export interface ToolResultFoldInput {
|
|
8
|
+
readonly sessionId: string;
|
|
9
|
+
readonly runId: string;
|
|
10
|
+
readonly turn: number;
|
|
11
|
+
readonly toolCallId: string;
|
|
12
|
+
readonly toolName: string;
|
|
13
|
+
readonly text: string;
|
|
14
|
+
}
|
|
15
|
+
export interface ToolResultFoldOptions {
|
|
16
|
+
readonly minAgeTurns?: number;
|
|
17
|
+
readonly minBytes?: number;
|
|
18
|
+
readonly maxSummaryBytes?: number;
|
|
19
|
+
readonly summarize: (input: ToolResultFoldInput) => Promise<string> | string;
|
|
20
|
+
}
|
|
21
|
+
export interface ResolvedToolResultFoldOptions {
|
|
22
|
+
readonly minAgeTurns: number;
|
|
23
|
+
readonly minBytes: number;
|
|
24
|
+
readonly maxSummaryBytes: number;
|
|
25
|
+
readonly summarize: (input: ToolResultFoldInput) => Promise<string> | string;
|
|
26
|
+
}
|
|
27
|
+
export interface FoldToolResultsContext {
|
|
28
|
+
readonly sessionId: string;
|
|
29
|
+
readonly runId: string;
|
|
30
|
+
readonly turn: number;
|
|
31
|
+
readonly signal?: AbortSignal;
|
|
32
|
+
}
|
|
33
|
+
/** Run overrides agent; disabled when neither supplies `summarize`. */
|
|
34
|
+
export declare function resolveToolResultFold(run?: ToolResultFoldOptions, agent?: ToolResultFoldOptions): ResolvedToolResultFoldOptions | undefined;
|
|
35
|
+
/** Projection-only fold for history tool messages; does not mutate the input array. */
|
|
36
|
+
export declare function foldToolResultHistory(history: readonly Message[], options: ResolvedToolResultFoldOptions, context: FoldToolResultsContext): Promise<readonly Message[]>;
|
|
37
|
+
/** Projection-only fold for in-flight tool results before message conversion. */
|
|
38
|
+
export declare function foldToolResults(results: readonly ToolResult[], options: ResolvedToolResultFoldOptions, context: FoldToolResultsContext): Promise<readonly ToolResult[]>;
|
|
39
|
+
export declare function formatFoldedToolResult(summary: string): string;
|
|
40
|
+
export declare function foldedToolResultHeader(toolName: string, toolCallId: string, summary: string): string;
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
import { estimateTextBytes } from "./context-budget.js";
|
|
2
|
+
export const DEFAULT_TOOL_RESULT_FOLD_MIN_AGE_TURNS = 2;
|
|
3
|
+
export const DEFAULT_TOOL_RESULT_FOLD_MIN_BYTES = 4_096;
|
|
4
|
+
export const DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES = 512;
|
|
5
|
+
export const HARD_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES = 4_096;
|
|
6
|
+
export const TOOL_RESULT_FOLD_TURN_METADATA_KEY = "prismToolResultTurn";
|
|
7
|
+
/** Run overrides agent; disabled when neither supplies `summarize`. */
|
|
8
|
+
export function resolveToolResultFold(run, agent) {
|
|
9
|
+
const options = run ?? agent;
|
|
10
|
+
if (!options?.summarize)
|
|
11
|
+
return undefined;
|
|
12
|
+
const minAgeTurns = options.minAgeTurns ?? DEFAULT_TOOL_RESULT_FOLD_MIN_AGE_TURNS;
|
|
13
|
+
const minBytes = options.minBytes ?? DEFAULT_TOOL_RESULT_FOLD_MIN_BYTES;
|
|
14
|
+
const maxSummaryBytes = options.maxSummaryBytes ?? DEFAULT_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES;
|
|
15
|
+
assertPositiveInt(minAgeTurns, "minAgeTurns", 1, 1_024);
|
|
16
|
+
assertPositiveInt(minBytes, "minBytes", 1, 32 * 1024 * 1024);
|
|
17
|
+
assertPositiveInt(maxSummaryBytes, "maxSummaryBytes", 1, HARD_TOOL_RESULT_FOLD_MAX_SUMMARY_BYTES);
|
|
18
|
+
return { minAgeTurns, minBytes, maxSummaryBytes, summarize: options.summarize };
|
|
19
|
+
}
|
|
20
|
+
/** Projection-only fold for history tool messages; does not mutate the input array. */
|
|
21
|
+
export async function foldToolResultHistory(history, options, context) {
|
|
22
|
+
if (history.length === 0)
|
|
23
|
+
return history;
|
|
24
|
+
const turns = inferToolResultTurns(history);
|
|
25
|
+
const out = [];
|
|
26
|
+
for (let index = 0; index < history.length; index += 1) {
|
|
27
|
+
const message = history[index];
|
|
28
|
+
const folded = await foldToolResultMessage(message, options, {
|
|
29
|
+
...context,
|
|
30
|
+
toolResultTurn: turns[index] ?? context.turn,
|
|
31
|
+
});
|
|
32
|
+
out.push(folded);
|
|
33
|
+
}
|
|
34
|
+
return out;
|
|
35
|
+
}
|
|
36
|
+
/** Projection-only fold for in-flight tool results before message conversion. */
|
|
37
|
+
export async function foldToolResults(results, options, context) {
|
|
38
|
+
if (results.length === 0)
|
|
39
|
+
return results;
|
|
40
|
+
const out = [];
|
|
41
|
+
for (const result of results) {
|
|
42
|
+
const folded = await foldToolResultValue(result, options, { ...context, toolResultTurn: context.turn });
|
|
43
|
+
out.push(folded);
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
async function foldToolResultMessage(message, options, context) {
|
|
48
|
+
if (message.role !== "tool")
|
|
49
|
+
return message;
|
|
50
|
+
const block = message.content.find((part) => part.type === "tool_result");
|
|
51
|
+
if (!block || block.type !== "tool_result")
|
|
52
|
+
return message;
|
|
53
|
+
const text = toolResultText(block.result, block.error, message.content);
|
|
54
|
+
const folded = await maybeFold({
|
|
55
|
+
options,
|
|
56
|
+
context,
|
|
57
|
+
toolCallId: block.toolCallId,
|
|
58
|
+
toolName: block.name,
|
|
59
|
+
text,
|
|
60
|
+
apply: (summary) => ({
|
|
61
|
+
...message,
|
|
62
|
+
content: message.content.map((part) => part.type === "tool_result"
|
|
63
|
+
? { ...part, result: foldedToolResultHeader(block.name, block.toolCallId, summary), error: undefined }
|
|
64
|
+
: part),
|
|
65
|
+
metadata: { ...message.metadata, prismFolded: true },
|
|
66
|
+
}),
|
|
67
|
+
});
|
|
68
|
+
return folded ?? message;
|
|
69
|
+
}
|
|
70
|
+
async function foldToolResultValue(result, options, context) {
|
|
71
|
+
const text = toolResultText(result.value, result.error, result.content);
|
|
72
|
+
const folded = await maybeFold({
|
|
73
|
+
options,
|
|
74
|
+
context,
|
|
75
|
+
toolCallId: result.toolCallId,
|
|
76
|
+
toolName: result.name,
|
|
77
|
+
text,
|
|
78
|
+
apply: (summary) => ({
|
|
79
|
+
...result,
|
|
80
|
+
value: foldedToolResultHeader(result.name, result.toolCallId, summary),
|
|
81
|
+
error: undefined,
|
|
82
|
+
metadata: { ...result.metadata, prismFolded: true },
|
|
83
|
+
}),
|
|
84
|
+
});
|
|
85
|
+
return folded ?? result;
|
|
86
|
+
}
|
|
87
|
+
async function maybeFold(input) {
|
|
88
|
+
const age = input.context.turn - input.context.toolResultTurn;
|
|
89
|
+
if (age < input.options.minAgeTurns)
|
|
90
|
+
return undefined;
|
|
91
|
+
if (estimateTextBytes(input.text) < input.options.minBytes)
|
|
92
|
+
return undefined;
|
|
93
|
+
throwIfAborted(input.context.signal);
|
|
94
|
+
try {
|
|
95
|
+
const summary = await input.options.summarize({
|
|
96
|
+
sessionId: input.context.sessionId,
|
|
97
|
+
runId: input.context.runId,
|
|
98
|
+
turn: input.context.toolResultTurn,
|
|
99
|
+
toolCallId: input.toolCallId,
|
|
100
|
+
toolName: input.toolName,
|
|
101
|
+
text: input.text,
|
|
102
|
+
});
|
|
103
|
+
return input.apply(capSummaryBytes(String(summary), input.options.maxSummaryBytes));
|
|
104
|
+
}
|
|
105
|
+
catch {
|
|
106
|
+
return undefined;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
export function formatFoldedToolResult(summary) {
|
|
110
|
+
return summary;
|
|
111
|
+
}
|
|
112
|
+
export function foldedToolResultHeader(toolName, toolCallId, summary) {
|
|
113
|
+
return `Tool result ${toolName} [${toolCallId}]: ${summary}`;
|
|
114
|
+
}
|
|
115
|
+
function toolResultText(result, error, extra) {
|
|
116
|
+
const parts = [JSON.stringify(error ?? result ?? null)];
|
|
117
|
+
for (const block of extra ?? []) {
|
|
118
|
+
if (block.type === "text" && "text" in block && typeof block.text === "string")
|
|
119
|
+
parts.push(block.text);
|
|
120
|
+
}
|
|
121
|
+
return parts.join("\n");
|
|
122
|
+
}
|
|
123
|
+
function capSummaryBytes(summary, maxBytes) {
|
|
124
|
+
const bytes = estimateTextBytes(summary);
|
|
125
|
+
if (bytes <= maxBytes)
|
|
126
|
+
return summary;
|
|
127
|
+
const encoded = new TextEncoder().encode(summary);
|
|
128
|
+
const suffix = new TextEncoder().encode("…");
|
|
129
|
+
let end = Math.max(0, maxBytes - suffix.length);
|
|
130
|
+
while (end > 0 && (encoded[end] & 0xc0) === 0x80)
|
|
131
|
+
end--;
|
|
132
|
+
return new TextDecoder().decode(encoded.slice(0, end)) + "…";
|
|
133
|
+
}
|
|
134
|
+
function inferToolResultTurns(history) {
|
|
135
|
+
const turns = new Array(history.length).fill(1);
|
|
136
|
+
let providerTurn = 0;
|
|
137
|
+
let toolTurn = 1;
|
|
138
|
+
for (let index = 0; index < history.length; index += 1) {
|
|
139
|
+
const message = history[index];
|
|
140
|
+
const stamped = readToolResultTurn(message.metadata);
|
|
141
|
+
if (message.role === "assistant") {
|
|
142
|
+
providerTurn += 1;
|
|
143
|
+
toolTurn = providerTurn;
|
|
144
|
+
}
|
|
145
|
+
if (message.role === "tool") {
|
|
146
|
+
turns[index] = stamped ?? toolTurn;
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
return turns;
|
|
150
|
+
}
|
|
151
|
+
function readToolResultTurn(metadata) {
|
|
152
|
+
const value = metadata?.[TOOL_RESULT_FOLD_TURN_METADATA_KEY];
|
|
153
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined;
|
|
154
|
+
}
|
|
155
|
+
function assertPositiveInt(value, name, min, max) {
|
|
156
|
+
if (!Number.isSafeInteger(value) || value < min || value > max) {
|
|
157
|
+
throw new TypeError(`toolResultFold.${name} must be a safe integer from ${min} to ${max}`);
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
function throwIfAborted(signal) {
|
|
161
|
+
if (signal?.aborted)
|
|
162
|
+
throw signal.reason instanceof Error ? signal.reason : new Error("Tool result fold aborted");
|
|
163
|
+
}
|
|
164
|
+
//# sourceMappingURL=tool-result-fold.js.map
|
package/docs/0.1.0-readiness.md
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# 0.1.0 / 1.0 Readiness Gates
|
|
2
2
|
|
|
3
|
-
Status: **0.0.
|
|
3
|
+
Status: **0.0.20** is the current release line (Phase 3 skills progressive disclosure); **1.0** readiness remains operator-gated, not automatic.
|
|
4
4
|
|
|
5
5
|
This page distills runnable readiness gates into one command-per-gate table. The
|
|
6
6
|
**Last evidence** column records the 2026-07-26 **0.0.16** baseline snapshot
|
|
7
7
|
(Phase 11, Node v24.18.0, Linux x86_64). Treat it as historical floor evidence,
|
|
8
8
|
not the current release tag. Re-run each gate on the target release tree before
|
|
9
|
-
cutting 0.0.
|
|
9
|
+
cutting 0.0.19 / 1.0. The decision to cut 1.0 stays with the operator after
|
|
10
10
|
operator-gated legs run in a protected environment and Phase 12 demand evidence
|
|
11
11
|
exists.
|
|
12
12
|
|
|
@@ -14,13 +14,13 @@ Evidence trail: [`docs/review-coverage-2026-07-26-phase-11.md`](./review-coverag
|
|
|
14
14
|
(addenda 0–9), [`docs/release-and-install.md`](./release-and-install.md),
|
|
15
15
|
[`docs/migration.md`](./migration.md), [`docs/performance.md`](./performance.md).
|
|
16
16
|
|
|
17
|
-
## Current line (0.0.
|
|
17
|
+
## Current line (0.0.20)
|
|
18
18
|
|
|
19
19
|
| Item | Status |
|
|
20
20
|
|---|---|
|
|
21
|
-
| Published graph | **44** publishable manifests at **0.0.
|
|
22
|
-
| Phase
|
|
23
|
-
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.
|
|
21
|
+
| Published graph | **44** publishable manifests at **0.0.20** (`docs/release-and-install.md`) |
|
|
22
|
+
| Phase 3 progressive disclosure | `skillsDisclosure`, `load_skill`, empty registry default + `activateAllSkills`, priority budget demotion, optional `toolResultFold` |
|
|
23
|
+
| Docs tripwires | `node --test dist/__tests__/docs.test.js` — migration section `0.0.19 → 0.0.20 skills and context progressive disclosure` documents intentional breaks |
|
|
24
24
|
| Readiness table below | **0.0.16 measured values** — refresh evidence columns when 1.0 RC gates are recorded |
|
|
25
25
|
|
|
26
26
|
## Gate table
|
|
@@ -63,7 +63,9 @@ signatures can drift on a TypeScript bump without any real API change).
|
|
|
63
63
|
|
|
64
64
|
`docs/migration.md` carries release migration sections tripwired by
|
|
65
65
|
`docs.test.ts` (headings and key phrases). A missing or gutted section fails
|
|
66
|
-
the suite. **0.0.
|
|
66
|
+
the suite. **0.0.19** adds observational-memory lifecycle and nested settings in
|
|
67
|
+
`@arnilo/prism-compaction-observational-memory` — see `0.0.18 → 0.0.19 observational memory lifecycle`.
|
|
68
|
+
**0.0.18** adds intentional pre-1.0 breaks (`repo_search` literal-only,
|
|
67
69
|
`cache_aware` default layout, oldest-first history eviction, atomic write/edit
|
|
68
70
|
durability, MCP SDK bump) — see `0.0.17 → 0.0.18 restore integrity`.
|
|
69
71
|
|
|
@@ -133,7 +135,7 @@ Exact prerequisites that must be satisfied before cutting 1.0:
|
|
|
133
135
|
checked-in baseline.
|
|
134
136
|
7. **Phase 12 demand evidence** (below) recorded for any capability that 1.0
|
|
135
137
|
is expected to anchor.
|
|
136
|
-
8. **0.0.
|
|
138
|
+
8. **0.0.19+ lifecycle gates green** on the release candidate (docs suite,
|
|
137
139
|
MCP SDK advisory cleared, coding-tool durability fixes, layout/eviction defaults).
|
|
138
140
|
|
|
139
141
|
## Phase 12 demand-evidence entry criteria
|
|
@@ -51,6 +51,8 @@ string | Message | readonly Message[]
|
|
|
51
51
|
|
|
52
52
|
`RunOptions.model` can override the request model for a run. Model overrides append a `model_change` entry. `AgentConfig.inputLayout` selects the default input assembly layout (`"cache_aware"` by default, or opt-in `"legacy"`); `RunOptions.inputLayout` wins for one run. `AgentConfig.providerOptions`/`RunOptions.providerOptions` supply generic provider request options; `timeoutMs`, `maxRetries`, and `maxRetryDelayMs` are deprecated inert provider-level hints in first-party providers. Use `RunOptions.signal`/host abort controllers for timeouts and `AgentConfig.retry`/`RunOptions.retry` for retry. `AgentConfig.providerRequestPolicies`/`RunOptions.providerRequestPolicies` run before `AIProvider.generate()` and before `provider_request` middleware. `AgentConfig.systemPrompt` and `RunOptions.systemPrompt` add explicit layered system prompt contributions; `RunOptions.systemPrompt: false` disables configured prompt layers for that run while keeping `AgentConfig.instructions` as the base path. `RunOptions.compaction` can enable auto-compaction for that run or use `false` to disable configured auto-compaction. `RunOptions.retry` can enable provider-turn retry for that run or use `false` to disable configured retry. `RunOptions.metadata` is merged with agent/session metadata for assembly, provider requests, and tool contexts. Deprecated `RunOptions.maxToolRounds` narrows `limits.maxToolRounds`. `RunOptions.signal` is bridged into the per-run abort signal passed to assembly, providers, tools, auto-compaction, and retry backoff.
|
|
53
53
|
|
|
54
|
+
`RunOptions.activeSkills` selects named skills from a configured `SkillRegistry`; `RunOptions.skills` replaces a plain `Skill[]` config for one run. When `AgentConfig.skills` is a registry and neither is set, **no skills activate** unless `activateAllSkills: true` (run or agent). `skillsDisclosure` (`"progressive"` default, `"eager"` opt-in; run wins) controls catalog vs full instruction bodies; the session-owned `LoadedSkillSet` is populated by `load_skill` when the host registers `createLoadSkillTool`. `toolResultFold` (off unless the host supplies `summarize`) optionally folds aged large tool results in provider input only. See [Context and skills](context-and-skills.md).
|
|
55
|
+
|
|
54
56
|
## Outputs / response / events
|
|
55
57
|
|
|
56
58
|
`session.run()` / `session.prompt()` resolve to an `AgentRunResult` with `sessionId`, `runId`, `status`, `text`, `content`, optional `message`/`usage`/`leafId`, and terminal `error`/`abortReason` when applicable. Callers may ignore the return value. Failed and aborted runs still emit their terminal events, then reject with `AgentRunError` whose `.result` carries the same shape.
|
|
@@ -8,6 +8,21 @@ Current status: ledger/projection/render/recall utilities, explicit worker runti
|
|
|
8
8
|
|
|
9
9
|
This package is distinct from `@arnilo/prism-memory` working/semantic memory: observational memory compresses and recalls source-backed observations/reflections; semantic memory retrieves embeddings; working memory stores the current structured profile/state. Hosts may compose both.
|
|
10
10
|
|
|
11
|
+
## Four-layer provider context
|
|
12
|
+
|
|
13
|
+
Observational memory composes four independent layers for long sessions (Mastra-style):
|
|
14
|
+
|
|
15
|
+
| Layer | What it holds | How it is produced |
|
|
16
|
+
| --- | --- | --- |
|
|
17
|
+
| **Recent exact messages** | Last `context.recentMessages` user/assistant/tool entries in branch order (optional `recentMessageMaxTokens` trim, oldest first) | `buildObservationalMemoryContextBlocks()` → `recent-messages` ContextBlock; aligned with compaction `keepRecentEntries` |
|
|
18
|
+
| **Observation log** | Source-backed facts with 12-hex ids and `sourceEntryIds` | Observer worker on eligible unscanned `message` entries after `observation.messageTokens`; coverage advances even on empty passes |
|
|
19
|
+
| **Reflections** | Higher-level summaries over observation ids | Reflector worker on observations after last reflection coverage when `reflection.observationTokens` met |
|
|
20
|
+
| **Raw-source retrieval** | Exact branch messages behind a memory id or cursor page | `recallObservationalMemory()` / `recallObservationalMemoryBranchPage()` / `createRecallMemoryTool()` — exact-id or cursor paging only; no semantic search |
|
|
21
|
+
|
|
22
|
+
Activation is explicit: `createObservationalMemory().attach()` coordinates post-run observe/reflect/drop and `context.compactAfterTokens` compaction. Import and extension `setup` start nothing. Recall, commands, and utilities fail closed on invalid ids, wrong `sessionId`, ambiguous tool input, or oversized pages. Pass `secrets` for exact-value redaction in render/recall/worker paths. Branch isolation: hosts supply current-branch `appendEntry` and `getEntries`; mismatched store/session pairs fail closed after append.
|
|
23
|
+
|
|
24
|
+
See `examples/observational-memory-lifecycle.ts` for attach → turn → projection/recall/page without live credentials.
|
|
25
|
+
|
|
11
26
|
## When to use it
|
|
12
27
|
|
|
13
28
|
Use it when a host wants to opt in to long-session memory that records observations/reflections as session custom entries, renders prepared memory during compaction, and supports exact-id recall.
|
|
@@ -20,7 +35,7 @@ Memory records use `SessionEntry.kind: "custom"` with `entry.data.type` markers:
|
|
|
20
35
|
|
|
21
36
|
| Type | Payload |
|
|
22
37
|
| --- | --- |
|
|
23
|
-
| `om.observations.recorded` | `{ observations, coversUpToId? }` |
|
|
38
|
+
| `om.observations.recorded` | `{ observations, coversUpToId? }` — successful observer runs append coverage even when `observations` is empty. |
|
|
24
39
|
| `om.reflections.recorded` | `{ reflections, coversUpToId? }` |
|
|
25
40
|
| `om.observations.dropped` | `{ observationIds, coversUpToId? }` |
|
|
26
41
|
| `om.folded` | Compaction `data.memory` folded details. |
|
|
@@ -38,6 +53,10 @@ Worker limits are finite positive safe integers:
|
|
|
38
53
|
| `maxWorkerResultBytes` | 64 KiB | 1 MiB | Full tool result and replayed value/error payload |
|
|
39
54
|
| `maxWorkerMessageBytes` | 1 MiB | 8 MiB | System/prompt plus assistant-call/tool-result transcript |
|
|
40
55
|
| `maxWorkerErrorBytes` | 1 KiB | 8 KiB | Provider/tool/runtime error text after exact known-secret redaction |
|
|
56
|
+
| Rendered memory projection | — | 256 KiB | `renderObservationalMemory()` / context block text |
|
|
57
|
+
| Folded compaction payload | — | 512 KiB | `data.memory` JSON; strategy trims lowest-relevance observations before failing |
|
|
58
|
+
| Recent-message window | — | 512 KiB | `renderRecentMessageWindow()` hard cap |
|
|
59
|
+
| Recall page size | 20 | 100 | `retrieval.pageLimit` / recall tool `limit` |
|
|
41
60
|
|
|
42
61
|
Direct `runObserver()` / `runReflector()` / `runDropper()` calls retain required `maxTurns` and accept the corresponding shorter worker fields (`maxToolCalls`, `maxResultBytes`, etc.). Named default/hard constants and `resolveMemoryWorkerLimits()` are exported.
|
|
43
62
|
|
|
@@ -48,20 +67,26 @@ Key exports:
|
|
|
48
67
|
| Export | Purpose |
|
|
49
68
|
| --- | --- |
|
|
50
69
|
| `foldObservationalMemoryLedger()` | Fold custom memory entries into observations, reflections, drops, and coverage markers. |
|
|
70
|
+
| `isEligibleObservationSourceEntry()` / `eligibleObservationSources()` | Select user/assistant/tool `message` entries for observer input. |
|
|
71
|
+
| `unscannedEntries()` / `observationsUncoveredByReflection()` | Dual coverage helpers for observation scan and reflection windows. |
|
|
51
72
|
| `buildObservationalMemoryProjection()` | Build active/full/folded projections from current branch entries. |
|
|
73
|
+
| `buildObservationalMemoryContextBlocks()` | Render observational-memory + recent-messages context blocks for provider input. |
|
|
74
|
+
| `selectRecentMessageEntries()` / `renderRecentMessageWindow()` | Bounded exact recent-message suffix; count via `keepRecentEntries`, optional token trim via `estimateEntryTokens`. |
|
|
52
75
|
| `createFoldedMemoryDetails()` | Create JSON details for compaction `data.memory`. |
|
|
53
76
|
| `renderObservationalMemory()` | Render reflections and observations into a prepared memory summary. |
|
|
54
77
|
| `recallObservationalMemory()` | Recover source evidence for a known observation/reflection id from supplied current-branch entries. |
|
|
78
|
+
| `recallObservationalMemoryBranchPage()` | Page eligible user/assistant/tool messages around a cursor entry id (`forward`/`backward`, optional `detail: summary|full`). |
|
|
55
79
|
| `createMemoryId()` / `isMemoryId()` | Create/check 12-character ids. |
|
|
56
80
|
| `resolveObservationalMemorySettings()` | Merge `observational-memory` settings with defaults and overrides. |
|
|
57
|
-
| `
|
|
81
|
+
| `createObservationalMemory()` / `attach()` | One activation wires post-run observe/reflect/drop and `compactAfterTokens` compaction; returns proxied session, runtime, context provider, and strategy. |
|
|
82
|
+
| `createObservationalMemoryRuntime()` | Low-level explicit flush for advanced hosts or tests. |
|
|
58
83
|
| `createObservationalMemoryCompactionStrategy()` | Render existing folded memory as a standard Prism compaction summary with `data.memory`. |
|
|
59
84
|
| `createObservationalMemoryExtension()` | Inert extension helper that registers the strategy contribution unless disabled. |
|
|
60
|
-
| `createRecallMemoryTool()` | Optional
|
|
85
|
+
| `createRecallMemoryTool()` | Optional `recall` tool factory: exact id lookup or current-branch message paging via host-supplied entries. |
|
|
61
86
|
| `createMemoryStatusCommand()` / `createMemoryViewCommand()` | Optional `om:status` and `om:view` command factories. |
|
|
62
87
|
| `createObservationalMemoryCommands()` | Convenience factory returning status and view commands. |
|
|
63
88
|
|
|
64
|
-
Pure utilities create no events, workers, tools, commands, credentials, or provider requests. `
|
|
89
|
+
Pure utilities create no events, workers, tools, commands, credentials, or provider requests. `createObservationalMemoryExtension()` and import alone start nothing. `createObservationalMemory().attach()` runs workers only after proxied `run`/`prompt`/`stream`/`compact` complete (or after `wrapResumeRun` / `wrapResumeStream`). `createObservationalMemoryRuntime().flush()` remains for manual/advanced use. Attached `contextProvider` renders two blocks each turn: `observational-memory` (active reflections/observations aligned to the recent-message boundary) and `recent-messages` (last `keepRecentEntries` message entries in branch order, optionally trimmed by `recentMessageMaxTokens` using `estimateEntryTokens`; oldest dropped first). Compaction uses the same `keepRecentEntries` setting. Observer input includes only eligible `message` entries (`user`, `assistant`, `tool`); memory/compaction/bookkeeping entries advance `coversUpToId` scan coverage without entering the observer prompt. Successful observer/reflector runs append coverage markers even when they record zero facts. Reflection uses only active observations recorded after the last `om.reflections.recorded` entry unless `flush({ fullReflectionRebuild: true })`. Attached `flush()` skips with `run_active` while a proxied run is in flight. The compaction strategy is O(n) over supplied entries and makes no provider call. Tool and command factories are inert until a host registers/selects them.
|
|
65
90
|
|
|
66
91
|
## Request/response example
|
|
67
92
|
|
|
@@ -74,6 +99,7 @@ Pure utilities create no events, workers, tools, commands, credentials, or provi
|
|
|
74
99
|
```ts
|
|
75
100
|
import {
|
|
76
101
|
buildObservationalMemoryProjection,
|
|
102
|
+
createObservationalMemory,
|
|
77
103
|
createObservationalMemoryCompactionStrategy,
|
|
78
104
|
createObservationalMemoryExtension,
|
|
79
105
|
createObservationalMemoryCommands,
|
|
@@ -83,6 +109,18 @@ import {
|
|
|
83
109
|
renderObservationalMemory,
|
|
84
110
|
} from "@arnilo/prism-compaction-observational-memory";
|
|
85
111
|
|
|
112
|
+
const om = createObservationalMemory({
|
|
113
|
+
observation: { provider: observerProvider, model: observerModel, messageTokens: 10_000 },
|
|
114
|
+
reflection: { provider: reflectorProvider, model: reflectorModel, observationTokens: 20_000 },
|
|
115
|
+
context: { compactAfterTokens: 81_000, recentMessages: 8 },
|
|
116
|
+
retrieval: { pageLimit: 20 },
|
|
117
|
+
});
|
|
118
|
+
const attached = om.attach(session, {
|
|
119
|
+
appendEntry: (entry, options) => store.append(entry, options),
|
|
120
|
+
sessionModel: agent.config.model,
|
|
121
|
+
});
|
|
122
|
+
await attached.session.run("Continue from prior work");
|
|
123
|
+
|
|
86
124
|
const entries = await session.entries();
|
|
87
125
|
const projection = buildObservationalMemoryProjection(entries);
|
|
88
126
|
const summary = renderObservationalMemory(projection.reflections, projection.observations);
|
|
@@ -111,13 +149,19 @@ await kernel.load([createObservationalMemoryExtension({ recallTool: { getEntries
|
|
|
111
149
|
|
|
112
150
|
## Extension and configuration notes
|
|
113
151
|
|
|
114
|
-
Settings
|
|
152
|
+
Settings resolve to nested `observation` / `reflection` / `dropper` / `context` / `retrieval` groups via `resolveObservationalMemorySettings()`. Defaults: `observation.messageTokens: 10000`, `reflection.observationTokens: 20000`, `context.compactAfterTokens: 81000`, `context.recentMessages: 8`, `context.observationsPoolMaxTokens: 20000`, `dropper.targetTokens: 10000` (from `context.observationsPoolTargetTokens`), `retrieval.pageLimit: 20`, `agentMaxTurns: 16`, `passive: false`, `debugLog: false`. Optional `context.recentMessageMaxTokens` trims the recent-message context window (oldest first) after the count limit.
|
|
153
|
+
|
|
154
|
+
Legacy flat keys still map for pre-1.0 hosts (`observeAfterTokens` → `observation.messageTokens`, `reflectAfterTokens` → `reflection.observationTokens`, `compactAfterTokens` → `context.compactAfterTokens`, `keepRecentEntries` → `context.recentMessages`, flat `workerModel` → all workers when nested models absent). Conflicting flat+nested values throw.
|
|
155
|
+
|
|
156
|
+
Observer/reflector/dropper may use separate providers, models, instructions, thinking levels, credentials, and `requireExplicitModel`. `dropper.policy: "lowest-relevance"` drops deterministically without a model call; default is `"model"`. Top-level `workerProvider` / `workerModel` remain as deprecated aliases.
|
|
157
|
+
|
|
158
|
+
Token counting uses `estimateEntryTokens()` / `estimateMessageTokens()`.
|
|
115
159
|
|
|
116
|
-
The runtime requires host-supplied `session`, an `appendEntry` callback bound to that session's owning store/branch, and `workerProvider
|
|
160
|
+
The runtime requires host-supplied `session`, an `appendEntry` callback bound to that session's owning store/branch, and at least one worker provider (`observation.provider` or legacy `workerProvider`). Model selection uses [use-case model selection](use-case-model-selection.md): pass per-worker `model` (or settings `observation.model` / `reflection.model` / `dropper.model`) to override, and `sessionModel: agent.config.model` so workers fall back to the session model when no worker model is configured. `requireExplicitModel: true` restores the historical `missing_model` skip when no explicit worker model is set. It no longer accepts a separate `store` option because mismatched session/store pairs can append memory entries outside the active branch. After each memory append, the runtime checks the appended entry is visible at the session leaf and fails closed/restores the previous checkout if the callback points elsewhere. Optional credential resolution is explicit; missing requested credentials skip worker execution. Default credential requests use the **resolved** model's provider id.
|
|
117
161
|
|
|
118
|
-
`createObservationalMemoryCompactionStrategy()` keeps recent message entries like the default compaction strategy, renders existing observations/reflections as the summary, and returns a standard Prism compaction entry. Its `data` includes `throughEntryId`, `keepEntryIds`, `strategy`, `trigger`, and `memory: { type: "om.folded", version: 1, fullFold, observations, reflections, droppedObservationIds }`. When active observations exceed `observationsPoolMaxTokens`, it performs a full fold
|
|
162
|
+
`createObservationalMemoryCompactionStrategy()` keeps recent message entries like the default compaction strategy, renders existing observations/reflections as the summary, and returns a standard Prism compaction entry. Its `data` includes `throughEntryId`, `keepEntryIds`, `strategy`, `trigger`, and `memory: { type: "om.folded", version: 1, fullFold, observations, reflections, droppedObservationIds }`. When active observations exceed `context.observationsPoolMaxTokens`, it performs a full fold and synchronously trims lowest-relevance observations until the folded payload fits hard byte/token caps (or throws a typed error).
|
|
119
163
|
|
|
120
|
-
`createRecallMemoryTool()`
|
|
164
|
+
`createRecallMemoryTool()` accepts either `{ id }` for exact memory recall or `{ cursor, limit?, direction?, detail? }` for current-branch raw-message paging (default limit 20, hard cap 100). Reflection recall resolves supporting observations from the full ledger and reports `droppedSupportingObservationIds` / `missingSupportingObservationIds`; dropped supports still return available raw sources. Invalid ids, ambiguous requests, wrong `sessionId`, missing cursors, non-message cursors, and oversized pages fail closed. It does not search by topic.
|
|
121
165
|
|
|
122
166
|
`createMemoryStatusCommand()` reports recorded/dropped/active/visible observations, recorded/visible reflections, pool token counts, and optional runtime in-flight/last-error state. `createMemoryViewCommand()` renders visible memory by default or full active recorded memory with `{ mode: "full" }`; other modes return `Usage: /om:view [full]`.
|
|
123
167
|
|