@agentwhy/cli 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapter/codex/contract/assumptions.js +4 -4
- package/dist/adapter/codex/contract/deliveries.js +12 -0
- package/dist/adapter/codex/contract/messages.js +8 -1
- package/dist/adapter/codex/events/cell-commands.js +191 -0
- package/dist/adapter/codex/events/delivered-output.js +30 -0
- package/dist/adapter/codex/events/rollout-scan.js +99 -7
- package/dist/core/access/command-line.js +4 -1
- package/dist/core/access/listing.js +10 -3
- package/dist/core/access/protected-access.js +94 -1
- package/dist/core/access/recorded-effect.js +22 -1
- package/dist/report/build-report.js +125 -19
- package/dist/report/render/report-page/file-story.js +33 -2
- package/dist/report/render/report-page/files-view.js +176 -55
- package/dist/report/render/report-page/files.js +51 -8
- package/dist/report/render/report-page/fix-wizard-script.js +17 -7
- package/dist/report/render/report-page/fix-wizard.js +4 -1
- package/dist/report/render/report-page/helpers.js +2 -1
- package/dist/report/render/report-page/report-page-renderer.js +51 -12
- package/dist/report/render/report-page/story-window.js +33 -12
- package/dist/report/render/report-page/to-do-view.js +42 -10
- package/dist/report/render/ui/app-sidebar.js +31 -2
- package/dist/report/render/ui/confirm-dialog.js +69 -4
- package/dist/report/render/ui/fold-line.js +2 -0
- package/dist/report/render/ui/popup.js +1 -1
- package/dist/report/render/ui/project-list.js +49 -10
- package/dist/report/render/ui/words/app-words.js +6 -0
- package/dist/report/render/ui/words/conversations-words.js +12 -0
- package/dist/report/render/ui/words/projects-words.js +9 -0
- package/dist/report/render/ui/words/report-words.js +114 -12
- package/dist/report/render/ui/words/settings-words.js +12 -9
- package/dist/report/start/conversations/calendar-view.js +1 -1
- package/dist/report/start/conversations/conversation-columns.js +6 -4
- package/dist/report/start/conversations/period-section.js +33 -17
- package/dist/report/start/conversations/periods.js +7 -1
- package/dist/report/start/render/session-status.js +14 -9
- package/dist/report/start/session-start.js +6 -3
- package/dist/report/start/settings/alerts-tab.js +2 -1
- package/package.json +1 -1
|
@@ -23,16 +23,16 @@ export const ASSUMPTIONS = {
|
|
|
23
23
|
measured: '65 paginated and 5 legacy files by first metadata on 2026-09-29; 7 paginated terminal sessions (§2.8).',
|
|
24
24
|
},
|
|
25
25
|
actions: {
|
|
26
|
-
rule: 'One event per action item, by item.id; a command line only from [shell, -lc|-c, line]; parsed_cmd never read.',
|
|
27
|
-
measured: 'Item joins to turn and thread in every item (§2.3, §2.8). Refused attempts and failed image views leave no item (XB7, XB1), so the action stream is never whole.',
|
|
26
|
+
rule: 'One event per action item, by item.id; a command line only from [shell, -lc|-c, line]; parsed_cmd never read. Where a record holds no command or change item, each command a cell wrote out as text is one event (X23a).',
|
|
27
|
+
measured: 'Item joins to turn and thread in every item (§2.3, §2.8). Refused attempts and failed image views leave no item (XB7, XB1), so the action stream is never whole. The VS Code panel records no command item at all; 230 of 230 commands in 224 cells without items were written out as text (§2.11).',
|
|
28
28
|
},
|
|
29
29
|
access: {
|
|
30
30
|
rule: 'A status says a process ran, never what it reached: only a completed write, or one reader that exited 0, or a line of output naming the file.',
|
|
31
31
|
measured: 'A failed cat records its diagnostic in stdout with exit 1 (§2.8); exit codes of other programs are not profiled.',
|
|
32
32
|
},
|
|
33
33
|
delivery: {
|
|
34
|
-
rule: 'The model receives the cell output as a whole, joined to its cell by call_id and to no item; recorded execution output is not delivered by itself.',
|
|
35
|
-
measured: 'XB5 on 0.157.0: forwarded markers are in the cell output, withheld ones only in the item; outputs over 1 MiB are cut with a textual notice only.',
|
|
34
|
+
rule: 'The model receives the cell output as a whole, joined to its cell by call_id and to no item; recorded execution output is not delivered by itself, and is the model\'s only where one cell return carries it whole and alone (§2.10).',
|
|
35
|
+
measured: 'XB5 on 0.157.0: forwarded markers are in the cell output, withheld ones only in the item; outputs over 1 MiB are cut with a textual notice only. On 0.160.0 paginated, of 79 non-empty command outputs 15 reached a cell whole (13 of them uniquely), 36 left one long line, 11 a first token and 17 nothing.',
|
|
36
36
|
},
|
|
37
37
|
messages: {
|
|
38
38
|
rule: 'Assistant response messages are canonical; AgentMessage items join them by id; task_complete joins the one final answer of its turn.',
|
|
@@ -19,6 +19,18 @@ export const CODE_CELL = {
|
|
|
19
19
|
partText: 'text',
|
|
20
20
|
textPart: 'input_text',
|
|
21
21
|
};
|
|
22
|
+
/**
|
|
23
|
+
* What a cell's code ran, read only where the record keeps no action item (XD4, amended 2026-10-05; §2.11): the VS Code
|
|
24
|
+
* panel's records hold none. The code calls `tools.exec_command({ cmd: "…" })`; its return to the model opens with one
|
|
25
|
+
* of two headers (the probe's vocabulary, measured on 0.157.0-0.160.0), the script's state, never the command's exit.
|
|
26
|
+
*/
|
|
27
|
+
export const CELL_COMMANDS = {
|
|
28
|
+
toolsObject: 'tools',
|
|
29
|
+
call: 'exec_command',
|
|
30
|
+
command: 'cmd',
|
|
31
|
+
completedHeader: 'Script completed',
|
|
32
|
+
failedHeader: 'Script failed',
|
|
33
|
+
};
|
|
22
34
|
export const FUNCTION_CALL = {
|
|
23
35
|
name: 'name',
|
|
24
36
|
callId: 'call_id',
|
|
@@ -29,14 +29,21 @@ export const ROLES = { assistant: 'assistant', user: 'user', developer: 'develop
|
|
|
29
29
|
export const TEXT_BLOCKS = { said: 'output_text', given: 'input_text' };
|
|
30
30
|
/** `phase` on assistant messages and `AgentMessage` items (§2.8): what the core calls a channel. */
|
|
31
31
|
export const PHASES = { commentary: 'commentary', final_answer: 'final' };
|
|
32
|
-
/**
|
|
32
|
+
/**
|
|
33
|
+
* Message items (§2.3): copies of response records, joined by `id` where one is (§2.8). A `HookPrompt` is what a hook
|
|
34
|
+
* asked of the model when it blocked a Stop: of 84, on 0.155-0.160 in the terminal, VS Code, the Codex app and the
|
|
35
|
+
* ChatGPT app, each came right after a `user` message with its `id` whose parts held its `fragments`' text (§2.12).
|
|
36
|
+
*/
|
|
33
37
|
export const MESSAGE_ITEMS = {
|
|
34
38
|
agent: 'AgentMessage',
|
|
35
39
|
user: 'UserMessage',
|
|
36
40
|
reasoning: 'Reasoning',
|
|
37
41
|
compaction: 'ContextCompaction',
|
|
38
42
|
functionOutput: 'FunctionCallOutput',
|
|
43
|
+
hookPrompt: 'HookPrompt',
|
|
39
44
|
};
|
|
45
|
+
/** A `HookPrompt` item's words: one fragment per hook that asked, each with its `text` (§2.12). */
|
|
46
|
+
export const HOOK_PROMPT = { fragments: 'fragments', text: 'text' };
|
|
40
47
|
/** An `AgentMessage` or `UserMessage` item's content blocks: `Text` and `text` (§2.8). */
|
|
41
48
|
export const ITEM_TEXT_BLOCKS = ['Text', 'text'];
|
|
42
49
|
/**
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
// Copyright 2026 Nessprim Karol Kozer
|
|
2
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
import { CELL_COMMANDS } from '../contract/deliveries.js';
|
|
4
|
+
export function cellCommands(code) {
|
|
5
|
+
const commands = [];
|
|
6
|
+
let calls = 0;
|
|
7
|
+
let other = false;
|
|
8
|
+
for (const call of toolCalls(code)) {
|
|
9
|
+
if (call.name !== CELL_COMMANDS.call) {
|
|
10
|
+
other = true;
|
|
11
|
+
continue;
|
|
12
|
+
}
|
|
13
|
+
calls += 1;
|
|
14
|
+
const command = literalArgument(code, call.argumentsAt);
|
|
15
|
+
if (command !== undefined)
|
|
16
|
+
commands.push(command);
|
|
17
|
+
}
|
|
18
|
+
return { commands, unread: calls - commands.length, alone: calls === 1 && commands.length === 1 && !other };
|
|
19
|
+
}
|
|
20
|
+
/** Every `tools.<name>(` the code holds outside a string or a comment, with where its arguments start. */
|
|
21
|
+
function toolCalls(code) {
|
|
22
|
+
const found = [];
|
|
23
|
+
const prefix = CELL_COMMANDS.toolsObject + '.';
|
|
24
|
+
for (let at = 0; at < code.length;) {
|
|
25
|
+
const skipped = skipQuoted(code, at);
|
|
26
|
+
if (skipped !== at) {
|
|
27
|
+
at = skipped;
|
|
28
|
+
continue;
|
|
29
|
+
}
|
|
30
|
+
if (code.startsWith(prefix, at) && !isWordCharacter(code[at - 1])) {
|
|
31
|
+
let end = at + prefix.length;
|
|
32
|
+
while (end < code.length && isWordCharacter(code[end]))
|
|
33
|
+
end += 1;
|
|
34
|
+
const name = code.slice(at + prefix.length, end);
|
|
35
|
+
let open = end;
|
|
36
|
+
while (open < code.length && /\s/.test(code[open] ?? ''))
|
|
37
|
+
open += 1;
|
|
38
|
+
if (name !== '' && code[open] === '(')
|
|
39
|
+
found.push({ name, argumentsAt: open + 1 });
|
|
40
|
+
at = end;
|
|
41
|
+
continue;
|
|
42
|
+
}
|
|
43
|
+
at += 1;
|
|
44
|
+
}
|
|
45
|
+
return found;
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* The command of `({ cmd: "…", … })` where it is written text: a quoted string, or a template with no `${`. Anything
|
|
49
|
+
* else - a variable, a call, a template that interpolates, an argument that is no object - is not read.
|
|
50
|
+
*/
|
|
51
|
+
function literalArgument(code, at) {
|
|
52
|
+
let position = skipSpace(code, at);
|
|
53
|
+
if (code[position] !== '{')
|
|
54
|
+
return undefined;
|
|
55
|
+
position += 1;
|
|
56
|
+
for (let depth = 1; position < code.length && depth > 0;) {
|
|
57
|
+
position = skipSpace(code, position);
|
|
58
|
+
const character = code[position];
|
|
59
|
+
if (character === '}') {
|
|
60
|
+
depth -= 1;
|
|
61
|
+
position += 1;
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
if (character === '{' || character === '[' || character === '(') {
|
|
65
|
+
depth += 1;
|
|
66
|
+
position += 1;
|
|
67
|
+
continue;
|
|
68
|
+
}
|
|
69
|
+
if (character === ']' || character === ')') {
|
|
70
|
+
depth -= 1;
|
|
71
|
+
position += 1;
|
|
72
|
+
continue;
|
|
73
|
+
}
|
|
74
|
+
if (depth === 1) {
|
|
75
|
+
const key = keyAt(code, position);
|
|
76
|
+
if (key !== undefined) {
|
|
77
|
+
const colon = skipSpace(code, key.end);
|
|
78
|
+
if (code[colon] === ':' && key.name === CELL_COMMANDS.command) {
|
|
79
|
+
// The text must be the whole value: `"rg " + pattern` is built while the code runs.
|
|
80
|
+
const value = skipSpace(code, colon + 1);
|
|
81
|
+
const after = code[skipSpace(code, skipQuoted(code, value))];
|
|
82
|
+
return after === ',' || after === '}' ? stringAt(code, value) : undefined;
|
|
83
|
+
}
|
|
84
|
+
position = key.end;
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
const skipped = skipQuoted(code, position);
|
|
89
|
+
position = skipped !== position ? skipped : position + 1;
|
|
90
|
+
}
|
|
91
|
+
return undefined;
|
|
92
|
+
}
|
|
93
|
+
/** A key of an object literal: a bare name, or a quoted one. */
|
|
94
|
+
function keyAt(code, at) {
|
|
95
|
+
const quote = code[at];
|
|
96
|
+
if (quote === '"' || quote === "'") {
|
|
97
|
+
const end = skipQuoted(code, at);
|
|
98
|
+
const name = stringAt(code, at);
|
|
99
|
+
return name === undefined ? undefined : { name, end };
|
|
100
|
+
}
|
|
101
|
+
let end = at;
|
|
102
|
+
while (end < code.length && isWordCharacter(code[end]))
|
|
103
|
+
end += 1;
|
|
104
|
+
return end === at ? undefined : { name: code.slice(at, end), end };
|
|
105
|
+
}
|
|
106
|
+
/** The value of a string literal starting at `at`, or `undefined` where there is none, or it interpolates. */
|
|
107
|
+
function stringAt(code, at) {
|
|
108
|
+
const quote = code[at];
|
|
109
|
+
if (quote !== '"' && quote !== "'" && quote !== '`')
|
|
110
|
+
return undefined;
|
|
111
|
+
let text = '';
|
|
112
|
+
for (let position = at + 1; position < code.length; position += 1) {
|
|
113
|
+
const character = code[position];
|
|
114
|
+
if (character === quote)
|
|
115
|
+
return text;
|
|
116
|
+
if (quote === '`' && character === '$' && code[position + 1] === '{')
|
|
117
|
+
return undefined;
|
|
118
|
+
if (quote !== '`' && character === '\n')
|
|
119
|
+
return undefined;
|
|
120
|
+
if (character !== '\\') {
|
|
121
|
+
text += character;
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
const escaped = escapeAt(code, position);
|
|
125
|
+
if (escaped === undefined)
|
|
126
|
+
return undefined;
|
|
127
|
+
text += escaped.text;
|
|
128
|
+
position = escaped.end - 1;
|
|
129
|
+
}
|
|
130
|
+
return undefined;
|
|
131
|
+
}
|
|
132
|
+
/** One escape of a JavaScript string, as the language reads it: `\n`, `é`, `\x41`, a line continuation, `\q` as `q`. */
|
|
133
|
+
function escapeAt(code, at) {
|
|
134
|
+
const next = code[at + 1];
|
|
135
|
+
if (next === undefined)
|
|
136
|
+
return undefined;
|
|
137
|
+
const simple = { n: '\n', t: '\t', r: '\r', b: '\b', f: '\f', v: '\v', 0: '\0' };
|
|
138
|
+
if (next in simple && !(next === '0' && /\d/.test(code[at + 2] ?? '')))
|
|
139
|
+
return { text: simple[next], end: at + 2 };
|
|
140
|
+
if (next === '\n')
|
|
141
|
+
return { text: '', end: at + 2 };
|
|
142
|
+
if (next === 'x') {
|
|
143
|
+
const hex = code.slice(at + 2, at + 4);
|
|
144
|
+
return /^[0-9a-fA-F]{2}$/.test(hex) ? { text: String.fromCharCode(parseInt(hex, 16)), end: at + 4 } : undefined;
|
|
145
|
+
}
|
|
146
|
+
if (next === 'u') {
|
|
147
|
+
const braced = /^\{([0-9a-fA-F]{1,6})\}/.exec(code.slice(at + 2));
|
|
148
|
+
if (braced !== null) {
|
|
149
|
+
const point = parseInt(braced[1], 16);
|
|
150
|
+
return point > 0x10ffff ? undefined : { text: String.fromCodePoint(point), end: at + 2 + braced[0].length };
|
|
151
|
+
}
|
|
152
|
+
const hex = code.slice(at + 2, at + 6);
|
|
153
|
+
return /^[0-9a-fA-F]{4}$/.test(hex) ? { text: String.fromCharCode(parseInt(hex, 16)), end: at + 6 } : undefined;
|
|
154
|
+
}
|
|
155
|
+
// An octal escape is legacy and a strict module refuses it: not read.
|
|
156
|
+
if (/[1-9]/.test(next))
|
|
157
|
+
return undefined;
|
|
158
|
+
return { text: next, end: at + 2 };
|
|
159
|
+
}
|
|
160
|
+
/** Past a string, a template or a comment that starts at `at`; `at` itself where none does. */
|
|
161
|
+
function skipQuoted(code, at) {
|
|
162
|
+
const character = code[at];
|
|
163
|
+
if (character === '/' && code[at + 1] === '/') {
|
|
164
|
+
const end = code.indexOf('\n', at);
|
|
165
|
+
return end === -1 ? code.length : end + 1;
|
|
166
|
+
}
|
|
167
|
+
if (character === '/' && code[at + 1] === '*') {
|
|
168
|
+
const end = code.indexOf('*/', at + 2);
|
|
169
|
+
return end === -1 ? code.length : end + 2;
|
|
170
|
+
}
|
|
171
|
+
if (character !== '"' && character !== "'" && character !== '`')
|
|
172
|
+
return at;
|
|
173
|
+
for (let position = at + 1; position < code.length; position += 1) {
|
|
174
|
+
if (code[position] === '\\') {
|
|
175
|
+
position += 1;
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
if (code[position] === character)
|
|
179
|
+
return position + 1;
|
|
180
|
+
}
|
|
181
|
+
return code.length;
|
|
182
|
+
}
|
|
183
|
+
function skipSpace(code, at) {
|
|
184
|
+
let position = at;
|
|
185
|
+
while (position < code.length && /\s/.test(code[position] ?? ''))
|
|
186
|
+
position += 1;
|
|
187
|
+
return position;
|
|
188
|
+
}
|
|
189
|
+
function isWordCharacter(character) {
|
|
190
|
+
return character !== undefined && /[A-Za-z0-9_$]/.test(character);
|
|
191
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
// Copyright 2026 Nessprim Karol Kozer
|
|
2
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
/**
|
|
4
|
+
* Whether a command's recorded output is what the model was handed (XB5, `2026-09-27-what-codex-wrote.md` §7).
|
|
5
|
+
*
|
|
6
|
+
* In code mode the model never receives what a process printed: it receives the cell's return value, and the agent's
|
|
7
|
+
* own JavaScript stands between the two and may keep any part of it. Measured 2026-10-05 over the 79 non-empty
|
|
8
|
+
* `CommandExecution` outputs of Codex 0.160.0 `paginated`: 15 reached a cell whole, 36 left one long line in it, 11 only
|
|
9
|
+
* a first token, and 17 nothing at all. An action item is therefore no evidence of delivery, which is why the adapter
|
|
10
|
+
* records one at the execution stage.
|
|
11
|
+
*
|
|
12
|
+
* Where the output does occur whole in exactly one of the session's cell returns, that one return carried it, and the
|
|
13
|
+
* stage is the model's. Two returns holding the same text name neither: of the 15, 13 were unique and 2 were not. The
|
|
14
|
+
* join is by the text itself, not by an id - no id joins an item to a cell (§2.3) - so it is made only on an exact,
|
|
15
|
+
* unique occurrence, and never on an empty output, which every cell would hold.
|
|
16
|
+
*/
|
|
17
|
+
export function deliveredWhole(output, cellReturns) {
|
|
18
|
+
const text = output.trim();
|
|
19
|
+
if (text === '')
|
|
20
|
+
return false;
|
|
21
|
+
let found = 0;
|
|
22
|
+
for (const returned of cellReturns) {
|
|
23
|
+
if (!returned.includes(text))
|
|
24
|
+
continue;
|
|
25
|
+
found += 1;
|
|
26
|
+
if (found > 1)
|
|
27
|
+
return false;
|
|
28
|
+
}
|
|
29
|
+
return found === 1;
|
|
30
|
+
}
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { FileAccessError } from '../../../ports/file-access-error.js';
|
|
2
2
|
import { isJsonObject, parseJsonObject } from '../../../shared/json.js';
|
|
3
|
-
import { ITEM, ITEM_EVENT } from '../contract/actions.js';
|
|
3
|
+
import { ACTION_ITEMS, ITEM, ITEM_EVENT } from '../contract/actions.js';
|
|
4
4
|
import { ACTIVITY, AGENT_MESSAGE, DELEGATION_TOOLS, SPAWN_ARGUMENTS } from '../contract/delegations.js';
|
|
5
|
-
import { CODE_CELL, FUNCTION_CALL } from '../contract/deliveries.js';
|
|
5
|
+
import { CELL_COMMANDS, CODE_CELL, FUNCTION_CALL } from '../contract/deliveries.js';
|
|
6
6
|
import { ENVELOPE, LINE_TYPES, PASSIVE_EVENTS, PASSIVE_LINE_TYPES } from '../contract/envelope.js';
|
|
7
|
-
import { EVENT_MESSAGES, ITEM_TEXT_BLOCKS, MESSAGE, MESSAGE_ITEMS, PHASES, REASONING, REASONING_ITEM, RESPONSE_ITEMS, ROLES, TEXT_BLOCKS, } from '../contract/messages.js';
|
|
7
|
+
import { EVENT_MESSAGES, HOOK_PROMPT, ITEM_TEXT_BLOCKS, MESSAGE, MESSAGE_ITEMS, PHASES, REASONING, REASONING_ITEM, RESPONSE_ITEMS, ROLES, TEXT_BLOCKS, } from '../contract/messages.js';
|
|
8
8
|
import { VERDICT } from '../contract/reviews.js';
|
|
9
9
|
import { SESSION } from '../contract/session.js';
|
|
10
10
|
import { RESPONSE_TURN, TURN_CONTEXT, TURN_EVENTS } from '../contract/turns.js';
|
|
@@ -13,6 +13,8 @@ import { PRE_TOOL_USE } from '../contract/hooks.js';
|
|
|
13
13
|
import { isActionItem, readAction } from './action-items.js';
|
|
14
14
|
import { hookRefusalsIn } from './hook-refusals.js';
|
|
15
15
|
import { capabilitiesOf } from './capability-records.js';
|
|
16
|
+
import { cellCommands } from './cell-commands.js';
|
|
17
|
+
import { deliveredWhole } from './delivered-output.js';
|
|
16
18
|
import { permissionsOf } from './turn-permissions.js';
|
|
17
19
|
/**
|
|
18
20
|
* One rollout file, read line by line into the collection. Only top-level lines are records (X13); order is the line's
|
|
@@ -43,6 +45,8 @@ class FileState {
|
|
|
43
45
|
#outputs = [];
|
|
44
46
|
#itemIds = new Set();
|
|
45
47
|
#actionCalls = [];
|
|
48
|
+
/** Items that record a command or a change: where there are none, a cell's code is all there is of what ran (XD4). */
|
|
49
|
+
#ranItems = 0;
|
|
46
50
|
#said = [];
|
|
47
51
|
#reasoning = new Map();
|
|
48
52
|
#agentItems = [];
|
|
@@ -51,6 +55,8 @@ class FileState {
|
|
|
51
55
|
#verdicts = [];
|
|
52
56
|
#unrecognised = false;
|
|
53
57
|
#interrupted = false;
|
|
58
|
+
/** The ids of `user` messages read so far: a `HookPrompt` with one of them is a copy of that message (§2.12). */
|
|
59
|
+
#givenIds = new Set();
|
|
54
60
|
#hiddenReasoning = false;
|
|
55
61
|
#permissionsRecorded = false;
|
|
56
62
|
#permissionsIncomplete = false;
|
|
@@ -169,6 +175,9 @@ class FileState {
|
|
|
169
175
|
const author = this.#role.kind === 'reviewer'
|
|
170
176
|
? (role === ROLES.assistant ? 'reviewer' : 'runtime')
|
|
171
177
|
: role === ROLES.user ? 'person' : role === ROLES.developer ? 'developer' : 'unknown';
|
|
178
|
+
const id = payload[MESSAGE.id];
|
|
179
|
+
if (role === ROLES.user && typeof id === 'string' && id !== '')
|
|
180
|
+
this.#givenIds.add(id);
|
|
172
181
|
this.#context('conversation', author, words.text, words.completeness, evidence, turnId);
|
|
173
182
|
}
|
|
174
183
|
/** X31: readable summary text is reasoning, never the whole of it; encrypted content is never read or guessed. */
|
|
@@ -228,6 +237,7 @@ class FileState {
|
|
|
228
237
|
this.#reviewerActions = true;
|
|
229
238
|
this.#calls.set(callId, {
|
|
230
239
|
kind, id: callId, evidence, toolName: typeof name === 'string' ? name : 'unrecognised call', input, ...(turnId === undefined ? {} : { turnId }),
|
|
240
|
+
...(cell && typeof code === 'string' ? { code } : {}),
|
|
231
241
|
});
|
|
232
242
|
if (this.#role.kind !== 'agent')
|
|
233
243
|
return;
|
|
@@ -299,7 +309,9 @@ class FileState {
|
|
|
299
309
|
if (typeof text !== 'string')
|
|
300
310
|
return;
|
|
301
311
|
if (type === EVENT_MESSAGES.agent && this.#role.kind === 'agent') {
|
|
302
|
-
|
|
312
|
+
// A copy of what the agent said, joined to no utterance: the question it leaves open is its own words, never what
|
|
313
|
+
// it ran or reached - said so, so a page can tell it from an action left unjoined (found 2026-10-05).
|
|
314
|
+
this.#collected.gaps.push({ kind: 'relation-unresolved', question: 'own-words', agentId: this.#role.agentId });
|
|
303
315
|
this.#context('conversation', 'agent', text, 'complete', evidence, turnId);
|
|
304
316
|
}
|
|
305
317
|
else
|
|
@@ -339,6 +351,15 @@ class FileState {
|
|
|
339
351
|
}
|
|
340
352
|
if (type === MESSAGE_ITEMS.compaction || type === MESSAGE_ITEMS.functionOutput)
|
|
341
353
|
return;
|
|
354
|
+
// §2.12: a hook's request to the model, never an action. The `user` message with its id holds its words already; one
|
|
355
|
+
// that joins none is kept as the runtime's words, so nothing it carried goes unread.
|
|
356
|
+
if (type === MESSAGE_ITEMS.hookPrompt) {
|
|
357
|
+
if (typeof id !== 'string' || !this.#givenIds.has(id)) {
|
|
358
|
+
const words = hookPromptText(item);
|
|
359
|
+
this.#context('conversation', 'runtime', words.text, words.completeness, evidence, turnId);
|
|
360
|
+
}
|
|
361
|
+
return;
|
|
362
|
+
}
|
|
342
363
|
// X6: an action item; one the contract does not name is read as an unknown tool, with a gap.
|
|
343
364
|
if (!isActionItem(type))
|
|
344
365
|
this.#unrecognised = true;
|
|
@@ -354,6 +375,8 @@ class FileState {
|
|
|
354
375
|
return;
|
|
355
376
|
}
|
|
356
377
|
this.#itemIds.add(id);
|
|
378
|
+
if (item[ITEM.type] === ACTION_ITEMS.command || item[ITEM.type] === ACTION_ITEMS.fileChange)
|
|
379
|
+
this.#ranItems += 1;
|
|
357
380
|
const action = readAction(item);
|
|
358
381
|
for (const question of action.unanswered) {
|
|
359
382
|
this.#collected.gaps.push({ kind: 'capability-absent', question, agentId: this.#role.agentId });
|
|
@@ -365,11 +388,13 @@ class FileState {
|
|
|
365
388
|
commands: action.commands, resultShape: action.resultShape, toolKnown: action.toolKnown,
|
|
366
389
|
...(action.written === undefined ? {} : { written: action.written }), ...(turnId === undefined ? {} : { turnId }), evidence,
|
|
367
390
|
},
|
|
368
|
-
// The item is both the call and what came back (X6). What it printed is the execution stage: never delivered by
|
|
391
|
+
// The item is both the call and what came back (X6). What it printed is the execution stage: never delivered by
|
|
392
|
+
// itself - `#finishAgent` raises it to the model's only where a cell is shown to have returned it (XB5).
|
|
369
393
|
result: {
|
|
370
394
|
callId, stage: 'execution', completeness: action.output?.completeness ?? 'complete', execution: action.execution, evidence,
|
|
371
395
|
...(action.output === undefined ? {} : { content: action.output.text }),
|
|
372
396
|
},
|
|
397
|
+
command: item[ITEM.type] === ACTION_ITEMS.command,
|
|
373
398
|
});
|
|
374
399
|
}
|
|
375
400
|
/** X16, X17: `started` names the agent a spawn started, `interacted` the agent a later call reached, by the call's id. */
|
|
@@ -455,9 +480,59 @@ class FileState {
|
|
|
455
480
|
permissionsRecorded: this.#permissionsRecorded,
|
|
456
481
|
}));
|
|
457
482
|
}
|
|
483
|
+
/**
|
|
484
|
+
* XD4, amended 2026-10-05 by the maintainer (§2.11): where an agent's record keeps no item of what ran - every VS Code
|
|
485
|
+
* panel record, measured - the commands its cells' code wrote out as text are its actions. A command the code builds
|
|
486
|
+
* while it runs is not read, and leaves the action stream a gap. What a cell returned is given to its command only
|
|
487
|
+
* where the cell ran that one command and called no other tool; it is the model's input, and the cell's header says
|
|
488
|
+
* whether the script ran to its end - never the command's exit code, which is not recorded. A cell whose return holds
|
|
489
|
+
* a refusal by agentwhy's hook is read by `hookRefusalsIn` alone, so a stopped command is never also a run one.
|
|
490
|
+
*/
|
|
491
|
+
#cellCommands() {
|
|
492
|
+
const agentId = this.#role.agentId;
|
|
493
|
+
const returned = new Map(this.#outputs.filter(({ kind }) => kind === 'cell').map(({ output }) => [output.callId, output]));
|
|
494
|
+
const found = [];
|
|
495
|
+
let unread = 0;
|
|
496
|
+
for (const cell of this.#calls.values()) {
|
|
497
|
+
if (cell.kind !== 'cell' || cell.code === undefined)
|
|
498
|
+
continue;
|
|
499
|
+
const output = returned.get(cell.id);
|
|
500
|
+
if (output !== undefined && hookRefusalsIn(output.text).length > 0)
|
|
501
|
+
continue;
|
|
502
|
+
const { commands, unread: notText, alone } = cellCommands(cell.code);
|
|
503
|
+
unread += notText;
|
|
504
|
+
const execution = output === undefined ? undefined : scriptState(output.text);
|
|
505
|
+
commands.forEach((command, at) => {
|
|
506
|
+
const id = `${this.#scoped(cell.id)}:command:${at + 1}`;
|
|
507
|
+
found.push({
|
|
508
|
+
call: {
|
|
509
|
+
id, agentId, toolName: CELL_COMMANDS.call, input: { command }, targets: [], commands: [command], resultShape: 'listing',
|
|
510
|
+
toolKnown: true, evidence: cell.evidence, ...(cell.turnId === undefined ? {} : { turnId: cell.turnId }),
|
|
511
|
+
},
|
|
512
|
+
...(output === undefined || execution === undefined ? {} : {
|
|
513
|
+
result: {
|
|
514
|
+
callId: id, stage: 'model', execution, evidence: output.evidence,
|
|
515
|
+
// Only a cell that ran this one command returned what it printed; any other's return is no one command's.
|
|
516
|
+
...(alone ? { content: output.text, completeness: output.completeness } : { completeness: 'unknown' }),
|
|
517
|
+
},
|
|
518
|
+
}),
|
|
519
|
+
});
|
|
520
|
+
});
|
|
521
|
+
}
|
|
522
|
+
if (unread > 0)
|
|
523
|
+
this.#collected.capabilities.push({ question: 'actions', state: 'unmeasured', source: this.#role.source, agentId });
|
|
524
|
+
return found;
|
|
525
|
+
}
|
|
458
526
|
#finishAgent() {
|
|
459
527
|
const agentId = this.#role.agentId;
|
|
460
|
-
|
|
528
|
+
// XB5: a command's output is the model's only where one cell return is shown to carry it, which is known once
|
|
529
|
+
// every record of this agent has been read. Measured on CommandExecution alone, so no other item type is raised.
|
|
530
|
+
const cellReturns = this.#outputs.filter(({ kind }) => kind === 'cell').map(({ output }) => output.text);
|
|
531
|
+
const calls = this.#actionCalls.map(({ call, result, command }) => (command && result.content !== undefined && deliveredWhole(result.content, cellReturns)
|
|
532
|
+
? { call, result: { ...result, stage: 'model' } }
|
|
533
|
+
: { call, result }));
|
|
534
|
+
if (this.#ranItems === 0 && this.#role.kind === 'agent')
|
|
535
|
+
calls.push(...this.#cellCommands());
|
|
461
536
|
for (const call of this.#calls.values()) {
|
|
462
537
|
// A direct call its action item answers is that item: one action, never two (X6).
|
|
463
538
|
if (call.kind === 'cell' || call.kind === 'follow-up' || (call.kind === 'direct' && this.#itemIds.has(call.id)))
|
|
@@ -546,7 +621,8 @@ class FileState {
|
|
|
546
621
|
#uncertainCopy(text, evidence) {
|
|
547
622
|
if (text === '')
|
|
548
623
|
return;
|
|
549
|
-
|
|
624
|
+
// A copy of the agent's words that joins none of them: what it leaves open is its words, as with the editor's copies.
|
|
625
|
+
this.#collected.gaps.push({ kind: 'relation-unresolved', question: 'own-words', agentId: this.#role.agentId });
|
|
550
626
|
this.#context('conversation', 'agent', text, 'complete', evidence);
|
|
551
627
|
}
|
|
552
628
|
/** X19, X20: every verdict with its own evidence, on the reviewed turn its reviewer's turn names, or on none. */
|
|
@@ -606,6 +682,22 @@ function textOf(content, types) {
|
|
|
606
682
|
}
|
|
607
683
|
return { text: texts.join('\n'), completeness: whole ? 'complete' : 'partial' };
|
|
608
684
|
}
|
|
685
|
+
/** The script's state, by the header its return opens with (XD4): never the exit code of a command it ran. */
|
|
686
|
+
function scriptState(text) {
|
|
687
|
+
if (text.startsWith(CELL_COMMANDS.completedHeader))
|
|
688
|
+
return { status: 'completed' };
|
|
689
|
+
if (text.startsWith(CELL_COMMANDS.failedHeader))
|
|
690
|
+
return { status: 'failed' };
|
|
691
|
+
return { status: 'unrecognised' };
|
|
692
|
+
}
|
|
693
|
+
/** A `HookPrompt`'s words, one fragment after another (§2.12); a fragment of another shape makes them partial. */
|
|
694
|
+
function hookPromptText(item) {
|
|
695
|
+
const fragments = item[HOOK_PROMPT.fragments];
|
|
696
|
+
if (!Array.isArray(fragments))
|
|
697
|
+
return { text: '', completeness: 'unknown' };
|
|
698
|
+
const texts = fragments.flatMap((fragment) => (isJsonObject(fragment) && typeof fragment[HOOK_PROMPT.text] === 'string' ? [fragment[HOOK_PROMPT.text]] : []));
|
|
699
|
+
return { text: texts.join('\n'), completeness: texts.length === fragments.length ? 'complete' : 'partial' };
|
|
700
|
+
}
|
|
609
701
|
/** A cell's or a call's output: a string, or parts whose text is under `input_text` (§2.8). */
|
|
610
702
|
function outputText(output) {
|
|
611
703
|
if (typeof output === 'string')
|
|
@@ -50,6 +50,9 @@ const PRINTS_CONTENT = new Set([
|
|
|
50
50
|
'strings',
|
|
51
51
|
'xxd',
|
|
52
52
|
'od',
|
|
53
|
+
// A filter of the lines it is given: `head customers.csv | cut -c1-80` prints the file's rows, cut short. Found on
|
|
54
|
+
// 2026-10-05 in a Codex session whose agent printed a tracked file's header and first row through it.
|
|
55
|
+
'cut',
|
|
53
56
|
]);
|
|
54
57
|
/**
|
|
55
58
|
* Programs that print nothing when they succeed (`said-where-the-person-is` SWO1): `cd apps && cat .env` prints the
|
|
@@ -63,7 +66,7 @@ const PRINTS_NOTHING = new Set(['cd']);
|
|
|
63
66
|
* …` in one line read the key and traced nothing, so the person heard "no value found" over a value in the chat.
|
|
64
67
|
* Not a search: a search prints a file's lines, and is read as one (`search-hits-are-reads`).
|
|
65
68
|
*/
|
|
66
|
-
const LISTS_NAMES = new Set(['ls', 'find']);
|
|
69
|
+
const LISTS_NAMES = new Set(['ls', 'find', 'file']);
|
|
67
70
|
/** `NAME=value` before a program sets a variable; it runs nothing. */
|
|
68
71
|
const ASSIGNMENT = /^[A-Za-z_][A-Za-z0-9_]*=/;
|
|
69
72
|
/** Unquoted, each of these ends a simple command. `&` directly before `>` redirects instead (`&>file`). */
|
|
@@ -38,7 +38,10 @@ export function listingPathCandidates(text) {
|
|
|
38
38
|
// `grep -n` without a file name writes `49:matched text`, and that leading number is not part of a path.
|
|
39
39
|
// Dropping it is what stops `49:packages/app/.env` from being reported as a file.
|
|
40
40
|
const withoutLineNumber = trimmed.replace(/^\d+:/, '');
|
|
41
|
-
|
|
41
|
+
// `ls -l` opens a line with a permission string nothing else begins with, and prints no search hit: the colon of
|
|
42
|
+
// its time of day (`Oct 5 12:26`) is no path's, so no field before it is offered.
|
|
43
|
+
const long = LONG_LISTING.test(withoutLineNumber);
|
|
44
|
+
const colon = long ? -1 : withoutLineNumber.indexOf(':');
|
|
42
45
|
const fields = withoutLineNumber.split(/\s+/);
|
|
43
46
|
// In order of preference. A whole grep line matches a pattern as readily as the path that starts it, so
|
|
44
47
|
// the narrower candidate is offered first and only the first match of a line is taken: one line of a
|
|
@@ -51,8 +54,12 @@ export function listingPathCandidates(text) {
|
|
|
51
54
|
{ text: colon === -1 && PATH_START.test(withoutLineNumber) ? asWholeLine(withoutLineNumber) : '', positional: true },
|
|
52
55
|
// A line that is one word is that word: `ls -1` and `git status --porcelain` write listings like that.
|
|
53
56
|
{ text: fields.length === 1 ? asWholeLine(withoutLineNumber) : '', positional: true },
|
|
54
|
-
// `ls -l` writes the name last, after a permission string that nothing else begins with
|
|
55
|
-
|
|
57
|
+
// `ls -l` writes the name last, after a permission string that nothing else begins with - a path by where it sits
|
|
58
|
+
// (`paths-not-fragments` R2), so every pattern of a policy may match it. Offered as not positional, it matched
|
|
59
|
+
// only a pattern naming one file: `ls -la` listed `demo.env` and `customers.csv`, and only the tracked
|
|
60
|
+
// `**/customers.csv` was found, never `**/*.env` (found 2026-10-05). A name holding a space is still cut to its
|
|
61
|
+
// last word, as it always was.
|
|
62
|
+
{ text: long ? asField(fields[fields.length - 1] ?? '') : '', positional: true, ...(long ? { listed: withoutLineNumber.startsWith('d') ? 'directory' : 'file' } : {}) },
|
|
56
63
|
];
|
|
57
64
|
return candidates.filter((candidate) => candidate.text !== '');
|
|
58
65
|
})
|